[
    {
        "id": "osp-25870",
        "type": "article-journal",
        "title": "Evaluation Is All You Need for Multi-Modal Autonomous Driving",
        "author": [
            {
                "family": "He",
                "given": "Zeyu"
            },
            {
                "family": "Liu",
                "given": "Shiqi"
            },
            {
                "family": "Chen",
                "given": "Ke"
            },
            {
                "family": "Yan",
                "given": "Yun"
            },
            {
                "family": "Wu",
                "given": "Jinzi"
            },
            {
                "family": "Lei",
                "given": "Dianqiao"
            },
            {
                "family": "Wang",
                "given": "Sirui"
            },
            {
                "family": "Peng",
                "given": "ShuRui"
            },
            {
                "family": "Chen",
                "given": "Tao"
            },
            {
                "family": "Huang",
                "given": "Zhuo"
            },
            {
                "family": "Wu",
                "given": "Yu"
            },
            {
                "family": "Shao",
                "given": "Yadong"
            },
            {
                "family": "Li",
                "given": "Zhichao"
            },
            {
                "family": "Sun",
                "given": "Ke"
            },
            {
                "family": "Guan",
                "given": "Yang"
            },
            {
                "family": "Li",
                "given": "Keqiang"
            },
            {
                "family": "Li",
                "given": "Shengbo Eben"
            }
        ],
        "URL": "https://omanscience.com/en/articles/evaluation-is-all-you-need-for-multi-modal-autonomous-driving",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Multi-modal planning is promising for autonomous driving by representing multiple plausible behaviors in ambiguous and long-tail scenarios. Existing methods mainly focus on improving trajectory multi-modality, enhancing trajectory representations, or reshaping the candidate distribution. Nevertheless, we identify a pronounced generation-evaluation asymmetry in multi-modal planning: despite strong oracle performance, existing planners often fail to reliably select the best available candidate, leaving substantial planning potential unrealized. To address this challenge, we propose iDriveVLA, a multi-modal planning framework that improves the candidate trajectory space while enabling more reliable and context-aware trajectory evaluation. Specifically, iDriveVLA introduces a unified trajectory evaluator comprising a Safety-aware Scorer for quality and risk estimation, together with a VLM-guided Modulator for scene-adaptive criterion weighting. We further develop an oracle-aligned progressive training strategy consisting of candidate imitation pretraining, candidate space refinement, and semantic ranking alignment. On the public NAVSIM v1 leaderboard, iDriveVLA achieves a new state-of-the-art performance of 94.95 PDMS, surpassing the human-expert reference."
    }
]