[
    {
        "id": "osp-18397",
        "type": "article-journal",
        "title": "ProactiveVLA: Augmenting Embodied Memory through Proactive Environment Exploration",
        "author": [
            {
                "family": "Tian",
                "given": "Shizuo"
            },
            {
                "family": "Luo",
                "given": "Haodong"
            },
            {
                "family": "Li",
                "given": "Yutong"
            },
            {
                "family": "Song",
                "given": "Yuebing"
            },
            {
                "family": "Liu",
                "given": "Yunxin"
            },
            {
                "family": "Li",
                "given": "Yuanchun"
            }
        ],
        "URL": "https://omanscience.com/en/articles/proactivevla-augmenting-embodied-memory-through-proactive-environment-exploration",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Rapid adaptation to a new environment requires a robot to acquire useful knowledge about local objects, states, and interactions from limited experience. Systems that combine a reasoning agent with a frozen vision-language-action model (VLA) can adapt through execution feedback and memory, making the choice of experience central to their effectiveness. Repeated practice of a target task may refine a familiar solution while leaving other interactions relevant to changed conditions untested. We introduce ProactiveVLA, which uses proactive environment exploration to acquire reusable knowledge for deployment-time adaptation. After completing an initial task, the agent allocates the remaining interaction budget to self-proposed goals covering object affordances, state-changing interactions, and compositions of interactions. It verifies execution outcomes and consolidates both task-directed and exploratory experience into memory that guides subsequent planning and control. ProactiveVLA outperforms the baselines under the same turn budget on LIBERO-Pro and RoboCasa365 Composite-Seen. On LIBERO-Pro Goal-T, with at most one VLA primitive invocation allowed during evaluation, ProactiveVLA completes 48% of instances, compared with 19% for the state-of-the-art task-refinement baseline."
    }
]