[
    {
        "id": "osp-26455",
        "type": "article-journal",
        "title": "PIVOT: Physically Informed Vision-Language Off-Road Traversability for Field Robot Navigation",
        "author": [
            {
                "family": "Jiao",
                "given": "Aoran"
            },
            {
                "family": "Zhao",
                "given": "Wenda"
            },
            {
                "family": "Sahak",
                "given": "Hshmat"
            },
            {
                "family": "Barfoot",
                "given": "Timothy D."
            }
        ],
        "URL": "https://omanscience.com/en/articles/pivot-physically-informed-vision-language-off-road-traversability-for-field-robot-navigation",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Terrain assessment is a critical capability for off-road mobile robots, enabling safe and reliable navigation through unstructured and geometrically complex environments. Conventional geometry-based terrain assessment is fast to compute but often overly conservative in unstructured environments. We present PIVOT: a Physically Informed Vision-Language Off-Road Traversability navigation system that augments conventional geometry-based planning with vision-language-model (VLM)-based semantic reasoning for field robots. To physically ground this assessment, we quantify how strongly the VLM's predicted traversal energy cost, robot vibration, and wheel slip correlate with real-world measurements and introduce a unified traversability score that weights each modality by its prediction-measurement correlation. For efficiency, we design a two-level navigation architecture that retains geometry-based planning as the nominal mode and invokes semantic replanning only when that mode fails to find a path. Across five repeated closed-loop trials on a mixed-terrain route totalling around $6.4$ km, the proposed system increases overall autonomy from $59.6\\%$ to $97.0\\%$, reduces human interventions from $11$ to $3$, and increases the mean distance between interventions (MDBI) from $69.2$ m to $412.9$ m compared with geometry-only navigation. These results demonstrate that physically grounded VLM-based terrain assessment can substantially extend autonomous navigation beyond the limitations of geometry alone, while preserving efficient geometric planning as the nominal mode."
    }
]