[
    {
        "id": "osp-24349",
        "type": "article-journal",
        "title": "InsightMap: Structured Spatial Modeling for Embodied Multimodal Reasoning",
        "author": [
            {
                "family": "Zheng",
                "given": "Hongpei"
            },
            {
                "family": "Yin",
                "given": "Hujun"
            }
        ],
        "URL": "https://omanscience.com/en/articles/insightmap-structured-spatial-modeling-for-embodied-multimodal-reasoning",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Language-guided navigation requires connecting partial observations to a persistent spatial reference and learning how actions change that representation. We introduce InsightMap, a framework that uses top-down maps as both explicit spatial memory and action-conditioned prediction targets. Historical views are linked to labeled map locations, and a shared multimodal backbone jointly learns navigation action prediction and post-action map generation. Map prediction provides auxiliary training supervision, while navigation inference decodes actions from the observed spatial context. An aligned RGB-D data pipeline supports a common interface for navigation, visual question answering, situated reasoning, and 3D grounding. On the validation-unseen splits of R2R-CE and RxR-CE, InsightMap achieves success rates (SR) of 56.9% and 54.9%, respectively. Adding map-prediction supervision improves R2R-CE SR by 4.3 and success weighted by path length (SPL) by 3.2 percentage points. On static spatial tasks, InsightMap achieves 103.7 CIDEr on ScanQA, 60.1% exact-match accuracy on SQA3D, and 53.1% grounding accuracy at 0.5 IoU on ScanRefer with detected object proposals. On Unitree Go2, it outperforms NaVid and NaVILA in hallway, lab, and office environments."
    }
]