[
    {
        "id": "osp-15429",
        "type": "article-journal",
        "title": "DIVA: Dual-Space Intent-Aware Visual Attenuation for Vision-Language-Action Policies",
        "author": [
            {
                "family": "Feng",
                "given": "Kaixi"
            },
            {
                "family": "Sun",
                "given": "Guoheng"
            },
            {
                "family": "Wang",
                "given": "Ziyao"
            },
            {
                "family": "He",
                "given": "Yexiao"
            },
            {
                "family": "Shen",
                "given": "Zheyu"
            },
            {
                "family": "Li",
                "given": "Ang"
            }
        ],
        "URL": "https://omanscience.com/en/articles/diva-dual-space-intent-aware-visual-attenuation-for-vision-language-action-policies",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Vision-language-action (VLA) policies typically feed dense visual patch tokens into a language-action backbone, preserving scene context but offering no explicit mechanism to regulate how strongly different visual tokens influence policy computation. We introduce DIVA, a Dual-Space Intent-Aware Visual Attenuation module with an anchor-then-attenuate design. DIVA combines high-level task intent with low-level visual evidence to estimate patch-wise relevance anchors, then applies them in two complementary spaces: it reweights projected visual tokens before backbone entry and persistently attenuates low-relevance visual states within the backbone. DIVA preserves the full visual token sequence and requires no external grounding supervision. On LIBERO, DIVA improves OpenVLA-OFT from 96.6% to 98.0% average success and raises its zero-shot LIBERO-Plus score from 69.6 to 72.6. Real-world experiments further show consistent gains under task-irrelevant visual perturbations, supporting the robustness of intent-aware visual attenuation beyond simulation."
    }
]