[
    {
        "id": "osp-15960",
        "type": "article-journal",
        "title": "Imagine to Act: High-Fidelity Data Synthesis via Image Editing World Model for Scalable GUI Agent Training",
        "author": [
            {
                "family": "Ning",
                "given": "Yongxin"
            },
            {
                "family": "Niu",
                "given": "Runliang"
            },
            {
                "family": "Xing",
                "given": "Qianli"
            },
            {
                "family": "Duan",
                "given": "Zhiyi"
            },
            {
                "family": "He",
                "given": "Qingzu"
            },
            {
                "family": "Wang",
                "given": "Pan"
            },
            {
                "family": "Wang",
                "given": "Qi"
            }
        ],
        "URL": "https://omanscience.com/en/articles/imagine-to-act-high-fidelity-data-synthesis-via-image-editing-world-model-for-scalable-gui-agent-training",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Graphical User Interface (GUI) agents have emerged as a promising paradigm for automating complex digital workflows across diverse applications. However, training highly capable and generalizable agents fundamentally relies on massive, high-fidelity visual-action trajectories, which are notoriously difficult to acquire. While human demonstrations are unscalable, existing GUI world models rely on text descriptions or HTML rendering, discarding crucial pixel-level visual details like icons and layout styles. To address this issue, we introduce Infinite-Dreamer, a simulation-free data synthesis method powered by a pixel-level Image Editing World Model. By conceptualizing GUI transitions as image editing tasks, we leverage Vision-Language Models (VLMs) to describe action-induced UI changes as structured delta-text. We then fine-tune an image editing backbone to controllably synthesize realistic screenshot transitions. We utilize this model to generate both single-frame visual robustness data and multi-step imaginary trajectories. To validate the effectiveness of our approach, we fine-tune the Qwen3-VL baseline solely on the synthesized data to obtain Infinite-Actor, and evaluate it on AndroidWorld, MobileWorld, and AndroidControl-Curated benchmarks. Infinite-Actor consistently outperforms the Qwen3-VL baselines across scales: Infinite-Actor-8B improves AndroidWorld Pass@1 by +4.45 and nearly doubles the MobileWorld Pass@3 success rate, while Infinite-Actor-2B improves Pass@1 by +9.05. Code is available at https://github.com/swaydy-n/Infinite-Dreamer."
    }
]