[
    {
        "id": "osp-17789",
        "type": "article-journal",
        "title": "HeiCo-FOCUS: A Clinically Grounded Dataset for Long-Context Video Understanding",
        "author": [
            {
                "family": "Mayer",
                "given": "Leon"
            },
            {
                "family": "Luttner",
                "given": "Lucas"
            },
            {
                "family": "Godau",
                "given": "Patrick"
            },
            {
                "family": "Fritzsche",
                "given": "Kai"
            },
            {
                "family": "Reinke",
                "given": "Annika"
            },
            {
                "family": "Boland",
                "given": "Leonie"
            },
            {
                "family": "Brandt",
                "given": "Jule"
            },
            {
                "family": "Heinecke",
                "given": "Janne"
            },
            {
                "family": "Nobuhara",
                "given": "Chloe K."
            },
            {
                "family": "Holzwarth",
                "given": "Niklas"
            },
            {
                "family": "Christodoulou",
                "given": "Evangelia"
            },
            {
                "family": "Knopp",
                "given": "Marcel"
            },
            {
                "family": "Michael",
                "given": "Dominik"
            },
            {
                "family": "Piermarco",
                "given": "Pascale"
            },
            {
                "family": "Neyaz",
                "given": "Saliq"
            },
            {
                "family": "Özarslan",
                "given": "Korhan Derin"
            },
            {
                "family": "Hennighausen",
                "given": "Jakob"
            },
            {
                "family": "Aumente-Maestro",
                "given": "Carlos"
            },
            {
                "family": "Rädsch",
                "given": "Tim"
            },
            {
                "family": "Baji",
                "given": "Dheeraj"
            },
            {
                "family": "Full",
                "given": "Peter Maximilian"
            },
            {
                "family": "Aichholz",
                "given": "Finn"
            },
            {
                "family": "Erpenbeck",
                "given": "Justus Veit"
            },
            {
                "family": "Schott",
                "given": "Linus Finn"
            },
            {
                "family": "Winkelhausen",
                "given": "Bastian"
            },
            {
                "family": "de Boer",
                "given": "Claas"
            },
            {
                "family": "Güttner",
                "given": "Bianca"
            },
            {
                "family": "Hummel",
                "given": "Anneli"
            },
            {
                "family": "Just",
                "given": "Gregor"
            },
            {
                "family": "Kirchner",
                "given": "Max"
            },
            {
                "family": "Li",
                "given": "Chenyang"
            },
            {
                "family": "Raffaut",
                "given": "Rozenn"
            },
            {
                "family": "Rodriguez",
                "given": "Ariel"
            },
            {
                "family": "Venkatesh",
                "given": "Danush Kumar"
            },
            {
                "family": "Wang",
                "given": "Kevin"
            },
            {
                "family": "Xu",
                "given": "Jinjing"
            },
            {
                "family": "Zeinoddin",
                "given": "Mona Sheikh"
            },
            {
                "family": "Khan",
                "given": "Salman"
            },
            {
                "family": "Pausch",
                "given": "Thomas M."
            },
            {
                "family": "Speidel",
                "given": "Stefanie"
            },
            {
                "family": "Stoyanov",
                "given": "Danail"
            },
            {
                "family": "Hashimoto",
                "given": "Daniel A."
            },
            {
                "family": "Kolbinger",
                "given": "Fiona R."
            },
            {
                "family": "Weiser",
                "given": "Thomas G."
            },
            {
                "family": "Maier-Hein",
                "given": "Lena"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/heico-focus-a-clinically-grounded-dataset-for-long-context-video-understanding",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Recent advances in Vision-Language Models (VLMs) have led to rapid progress in video understanding across a wide range of benchmark tasks. However, existing evaluations largely focus on short-term reasoning, failing to assess a critical capability: maintaining cumulative temporal consistency over extended time horizons. To close this evaluation gap, we introduce HeiCo-FOCUS, a clinically grounded dataset for evaluating long-context video understanding through the task of Foreign Object Contextual Understanding in Surgery. Built on a dataset of Heidelberg Colorectal surgeries, this task requires models to continuously track multiple objects as they are inserted, manipulated, occluded, and removed over procedures lasting up to hours. HeiCo-FOCUS comprises 30,000 visual question answering (VQA) pairs covering five core capabilities: object recognition, temporal grounding, aggregation, event and procedural understanding, and complex reasoning. The dataset was constructed through a rigorous multi-stage annotation pipeline involving large-scale crowd annotation and 39 surgical domain experts to ensure high quality and clinical relevance. To systematically probe model behavior, we introduce a multi-track evaluation framework that progressively increases temporal and contextual demands from single frames to full procedures. Experiments with ten frontier VLMs show that HeiCo-FOCUS tasks are far from solved: only around half of the models clearly outperform a text-only baseline. Across the video tracks, models perform best on event and procedural understanding (mean Accuracy: 56.5% across all models), while temporal grounding remains particularly challenging for all evaluated models (mean Accuracy: 19.7%). We therefore expect HeiCo-FOCUS to serve as a catalyst for the development of models capable of reliable, temporally consistent reasoning over hours-long videos."
    }
]