[
    {
        "id": "osp-18300",
        "type": "article-journal",
        "title": "Beyond Retargeting: Low-Latency and Robust Humanoid Whole-Body Teleoperation with Learned Atomic Motion Primitives",
        "author": [
            {
                "family": "Xu",
                "given": "Xiayan"
            },
            {
                "family": "Yu",
                "given": "Jiyu"
            },
            {
                "family": "Chen",
                "given": "Xingzhou"
            },
            {
                "family": "Qian",
                "given": "Siyi"
            },
            {
                "family": "Ma",
                "given": "Zongyu"
            },
            {
                "family": "Liu",
                "given": "Lilu"
            },
            {
                "family": "Shi",
                "given": "Ling"
            },
            {
                "family": "Zhang",
                "given": "Haodong"
            }
        ],
        "URL": "https://omanscience.com/en/articles/beyond-retargeting-low-latency-and-robust-humanoid-whole-body-teleoperation-with-learned-atomic-motion-primitives",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Humanoid whole-body teleoperation translates human motion into stable robot behavior in real time. Existing systems typically rely on online motion retargeting to bridge human--robot morphological differences, but this process adds latency and can produce physically infeasible targets. Meanwhile, diverse, noisy, and partial human-motion observations often fall outside the training distribution, potentially causing unstable robot behavior. We propose a retargeting-free policy that maps raw human motion directly to robot joint commands in a single forward pass, eliminating online kinematic adaptation. To improve robustness, we learn a codebook of full-body motion primitives that projects out-of-distribution observations onto plausible motion prototypes and recovers full-body motion from partial inputs. Experiments on a Unitree~G1 in simulation and on hardware, using virtual reality, optical mocap, text-to-motion generation, and monocular video inputs, show that our method outperforms baselines in latency and robustness."
    }
]