[
    {
        "id": "osp-24178",
        "type": "article-journal",
        "title": "UniWAM Technical Report: Unified Mobile Manipulation via Mixed-Stream World-Action Modeling and Manipulation Anchor Pose Supervision",
        "author": [
            {
                "family": "Xue",
                "given": "Wei"
            },
            {
                "family": "Liu",
                "given": "Keliang"
            },
            {
                "family": "Cui",
                "given": "Mingzhang"
            },
            {
                "family": "Xie",
                "given": "Jinhua"
            },
            {
                "family": "Wei",
                "given": "Jinjie"
            },
            {
                "family": "Hou",
                "given": "Jianan"
            },
            {
                "family": "Lu",
                "given": "Jingcheng"
            },
            {
                "family": "Wang",
                "given": "Lintao"
            },
            {
                "family": "Qiu",
                "given": "Kaixiang"
            },
            {
                "family": "Liu",
                "given": "Yizhou"
            },
            {
                "family": "Ye",
                "given": "Xinghai"
            },
            {
                "family": "Han",
                "given": "Jinghang"
            },
            {
                "family": "Li",
                "given": "Mingcheng"
            },
            {
                "family": "Gu",
                "given": "Jie"
            },
            {
                "family": "Wang",
                "given": "Shunli"
            },
            {
                "family": "Zhang",
                "given": "Lihua"
            },
            {
                "family": "Yang",
                "given": "Dingkang"
            }
        ],
        "URL": "https://omanscience.com/en/articles/uniwam-technical-report-unified-mobile-manipulation-via-mixed-stream-world-action-modeling-and-manipulation-anchor-pose-supervision",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Mobile manipulation requires precise navigation to a manipulation-ready pose followed by reliable object interaction. These two stages differ in action spaces and visual requirements, which complicates unified policy learning. In addition, collecting diverse real-world navigation data with explicit manipulation-ready pose supervision remains costly and difficult to scale. We introduce UniWAM, a unified mixed-stream world-action model with separate action encoders and output heads for navigation and manipulation, sharing a common backbone. This design supports joint representation learning on independently sampled navigation and manipulation data. UniWAM supports independent inference for either stream and batch-parallel inference for both. We further introduce Manipulation Anchor Pose (MAP) supervision for where to stop and how to orient for manipulation. An automated pipeline constructs MAP-Data from large-scale 3D scenes, yielding over 1.5 million episodes and 7,500 hours. MAP-Data provides per-frame target-object bounding boxes and image-plane MAP coordinates as auxiliary navigation supervision. Together with projected end-effector trajectories for manipulation, these prediction targets provide stream-specific image-plane supervision for action learning from egocentric observations. With large-scale MAP-Data, UniWAM outperforms the strongest external baselines on our MAP-Bench by 30.1\\% in position error and 44.0\\% in heading error. Across 24 real-robot tasks, UniWAM achieves leading results in MAP navigation and mobile manipulation, with competitive manipulation performance. We have released code, data, and benchmark."
    }
]