[
    {
        "id": "osp-25808",
        "type": "article-journal",
        "title": "Affordance-Conditioned Decision Making: Bridging the Semantic-Spatial Gap in Zero-Shot Cross-Floor Vision-and-Language Navigation",
        "author": [
            {
                "family": "Yang",
                "given": "Xuekang"
            },
            {
                "family": "Chen",
                "given": "Lu"
            },
            {
                "family": "Luo",
                "given": "Shuang"
            },
            {
                "family": "Zhu",
                "given": "Jialing"
            },
            {
                "family": "Zhang",
                "given": "Qi"
            },
            {
                "family": "Gao",
                "given": "Yue"
            },
            {
                "family": "Zhang",
                "given": "Xiang"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/affordance-conditioned-decision-making-bridging-the-semantic-spatial-gap-in-zero-shot-cross-floor-vision-and-language-navigation",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Vision-and-language navigation increasingly relies on general-purpose semantic planners, yet translating correct high-level intent into reliable physical execution remains difficult in spatially constrained transitions. Reaching a staircase, doorway, or narrow passage does not ensure traversal; the agent must identify an executable affordance pose and recover from accumulated action errors. We propose PACE (Preference-refined Affordance-Conditioned Execution), a supervised local execution module that augments frozen zero-shot semantic planners for reliable cross-floor navigation. PACE grounds transition-related semantics into a long-horizon, agent-centric traversable affordance pose and conditions short-horizon action generation on this spatial target, thereby aligning semantic goals with physical execution. We further post-train PACE through failure-aware preference refinement using rollout-derived pairs that contrast normal or recovery behaviors with deviation-amplifying behaviors, thereby improving closed-loop correction. We integrate PACE into six open-source zero-shot VLN navigators and demonstrate consistent improvements on the cross-floor subsets of R2R-CE and RxR-CE, increasing the average success rate from 16.35% to 27.65% and from 4.76% to 12.06%, respectively. Real-world experiments further demonstrate PACE's applicability in unseen environments, highlighting the potential of traversable affordances to bridge semantic intent and reliable embodied behavior."
    }
]