[
    {
        "id": "osp-23064",
        "type": "article-journal",
        "title": "The Hard Part Comes After Search: Benchmarking Web Agents on Synthesizing, Organizing, and Displaying Knowledge",
        "author": [
            {
                "family": "Gill",
                "given": "Alexander"
            },
            {
                "family": "Ishmam",
                "given": "Md Farhan"
            },
            {
                "family": "Nguyen",
                "given": "Xuyen"
            },
            {
                "family": "Bhat",
                "given": "Neha"
            },
            {
                "family": "DeYoung",
                "given": "Parker Henry"
            },
            {
                "family": "Chaleshtori",
                "given": "Fateme Hashemi"
            },
            {
                "family": "Stringham",
                "given": "Nathan"
            },
            {
                "family": "Marino",
                "given": "Kenneth"
            },
            {
                "family": "Marasović",
                "given": "Ana"
            }
        ],
        "URL": "https://omanscience.com/en/articles/the-hard-part-comes-after-search-benchmarking-web-agents-on-synthesizing-organizing-and-displaying-knowledge",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Existing computer-use agent benchmarks do not fully evaluate agents acting as assistants. A useful assistant retrieves information across complex, multi-step workflows, synthesizes it into artifacts (documents, presentations, spreadsheets), and navigates program interfaces to produce a coherent final product. Such workflows demand reasoning and synthesis, decomposition of complex tasks, as well as visual and spatial understanding. To study agents on workflows like these, we introduce KNOWS, a benchmark of open-ended, complex, browser-based tasks that jointly evaluate these capabilities, with each task culminating in a produced artifact. To write tasks, we develop a task design rubric and a protocol for ensuring that tasks meet the requirements. Each task is paired with an evaluator, a program that combines deterministic checks with LLM judgments to balance the richness, reliability, and automation tradeoff inherent to agent evaluation. We evaluate and analyze frontier computer-use agents and browser-based harnesses. They achieve moderate scores on partial-success metrics, but the best performer fully succeeds in fewer than 3% of our complex, long-horizon tasks. Failures on visual steps render the resulting artifacts unusable, even when agents complete more than 50% of other evaluation steps. Our results expose limitations of current agents acting as end-to-end assistants, and call for progress on tool use, visual understanding, and long-horizon reasoning."
    }
]