[
    {
        "id": "osp-25012",
        "type": "article-journal",
        "title": "Rolling-WAM: World Action Models with Rolling Imagination",
        "author": [
            {
                "family": "Zhou",
                "given": "Yinghua"
            },
            {
                "family": "Ye",
                "given": "Junjie"
            },
            {
                "family": "Zhao",
                "given": "Yiqi"
            },
            {
                "family": "Dong",
                "given": "Hao"
            },
            {
                "family": "Wang",
                "given": "Celina Shiyu"
            },
            {
                "family": "Ge",
                "given": "Ruohai"
            },
            {
                "family": "Yang",
                "given": "Tingyi"
            },
            {
                "family": "Van Hoorick",
                "given": "Basile"
            },
            {
                "family": "Sukhatme",
                "given": "Gaurav"
            },
            {
                "family": "Guizilini",
                "given": "Vitor"
            },
            {
                "family": "Wang",
                "given": "Yue"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/rolling-wam-world-action-models-with-rolling-imagination",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "World Action Models (WAMs) couple action generation with future visual prediction for robotic manipulation. However, completing the joint video-action denoising process at each replanning cycle incurs substantial latency, delaying action updates and limiting closed-loop responsiveness. We present Rolling-WAM, a formulation that distributes joint denoising across successive replanning cycles. Our method maintains a sliding window of video-action chunks at staggered noise levels. At each step, a rolling noise schedule fully denoises the imminent action chunk for execution, while partially refining farther-future chunks. As the window advances with new camera observations, the retained future chunks continue their denoising process. This distributes the computational cost over time while carrying an evolving visual-action context across chunk boundaries. Evaluations on LIBERO, RoboTwin, and a real-world Unitree G1 humanoid show that Rolling-WAM achieves competitive manipulation performance. By removing the need to denoise the entire prediction horizon from scratch, it delivers a 4.5x steady-state replanning speedup over standard joint WAMs."
    }
]