[
    {
        "id": "osp-15749",
        "type": "article-journal",
        "title": "Logbook: Extremely Long-form Audio Event Understanding",
        "author": [
            {
                "family": "Choi",
                "given": "Kwanghee"
            },
            {
                "family": "Shon",
                "given": "Suwon"
            },
            {
                "family": "Serdyuk",
                "given": "Dmitriy"
            },
            {
                "family": "Lan",
                "given": "Guitang"
            },
            {
                "family": "Huang",
                "given": "Chao-Wei"
            },
            {
                "family": "Rasooli",
                "given": "Mohammad Sadegh"
            },
            {
                "family": "Srivastava",
                "given": "Sangeeta"
            },
            {
                "family": "Lin",
                "given": "Zhaojiang"
            },
            {
                "family": "Adya",
                "given": "Saurabh"
            },
            {
                "family": "Sun",
                "given": "Ming"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/logbook-extremely-long-form-audio-event-understanding",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Audio benchmarks are built around short, pre-segmented clips, limiting model design to brief inputs or fixed vocabularies. To close this gap, we introduce Logbook, a benchmark for hour-scale audio understanding, with recordings ranging from ten minutes to six days. Given a continuous audio recording and an event label vocabulary, a system must predict a gap-free segmentation with an event label and a description per segment. We compare 52 systems, end-to-end and cascaded, and ablate fine-tuning, context length, and reasoning budget. We find the task tractable, though the best systems remain below the human reference. Also, over-segmentation is pervasive, and fine-tuning partially mitigates it. Finally, end-to-end are often better than cascaded systems, but degrades with longer context."
    }
]