[
    {
        "id": "osp-24535",
        "type": "article-journal",
        "title": "ORAV: Benchmarking Audio-Video Generation from Multimodal Contexts",
        "author": [
            {
                "family": "Hua",
                "given": "Jiacheng"
            },
            {
                "family": "Feng",
                "given": "Xiaokun"
            },
            {
                "family": "Hua",
                "given": "Jiaqi"
            },
            {
                "family": "Liu",
                "given": "Chang"
            },
            {
                "family": "Wang",
                "given": "Biao"
            },
            {
                "family": "Liu",
                "given": "Miao"
            }
        ],
        "URL": "https://omanscience.com/en/articles/orav-benchmarking-audio-video-generation-from-multimodal-contexts",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Audio-video generation using heterogeneous multimodal references has emerged as a new challenge, requiring both compositional control over generation and grounded understanding of multimodal context. In this paper, we introduce ORAV Bench for Omni Reference Audio-Video Generation, comprising 380 task instances with 2-10 references, 9 semantic roles, and 30 role compositions. Instructions specify the relationships among references; the media supply the identities, dynamics, and audio characteristics to be realized. To evaluate these open-ended outputs, we develop a reference-aware pairwise protocol that prepares visual and auditory evidence, compares the intended contribution of each reference, and checks the overall verdict in both presentation orders. On held-out instances, it achieves 86.08% effective agreement with human judgments. Across 5 frontier systems, overall rankings conceal distinct strengths across reference compositions. A recurring failure is to reproduce unintended source content in place of the requested result, despite closely resembling a reference. Reproducible pointwise diagnostics of quality, reference affinity, and speech reveal distinct dimensions of model behavior. ORAV thus offers a benchmark for tracking progress toward controllable, compositional, and reference-faithful audio-video generation."
    }
]