[
    {
        "id": "osp-24650",
        "type": "article-journal",
        "title": "ZeroBot: Learning from Scratch in Minutes with Generative Real2Sim",
        "author": [
            {
                "family": "Kapelyukh",
                "given": "Ivan"
            },
            {
                "family": "Zhang",
                "given": "Xiaohan"
            },
            {
                "family": "James",
                "given": "Stephen"
            },
            {
                "family": "Herlant",
                "given": "Laura"
            },
            {
                "family": "Johns",
                "given": "Edward"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/zerobot-learning-from-scratch-in-minutes-with-generative-real2sim",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "We present ZeroBot, a real2sim framework for learning a robot manipulation task from scratch in minutes under challenging conditions: zero human demonstrations, zero policy pre-training, and zero known object models. Given only a single view of an object and a goal pose for that object, ZeroBot uses image-to-3D generative models to obtain a complete object mesh, which is used in simulation for large-scale parallel reinforcement learning. To accelerate training, we introduce an action space which leverages the generated geometry and learned value function to sample states involving robot-object contact. When evaluated on real-world tasks including grasping, pushing, articulated object interaction, and multi-stage manipulation, ZeroBot achieves an 87% success rate with an average training time of 119 seconds. These results show the value of using image-to-3D models in a real2sim framework for rapid, autonomous robot learning."
    }
]