[
    {
        "id": "osp-26189",
        "type": "article-journal",
        "title": "Zeva-Ego: Egocentric Mid-Training with In-Context Causal Learning for Robot Manipulation",
        "author": [
            {
                "family": "Huang",
                "given": "Bingjia"
            },
            {
                "family": "Ding",
                "given": "Xin"
            },
            {
                "family": "Chen",
                "given": "Fu"
            },
            {
                "family": "Li",
                "given": "Kun"
            },
            {
                "family": "Sun",
                "given": "Wei"
            },
            {
                "family": "Wu",
                "given": "Hao"
            },
            {
                "family": "Liu",
                "given": "Yunxin"
            },
            {
                "family": "Cao",
                "given": "Ting"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/zeva-ego-egocentric-mid-training-with-in-context-causal-learning-for-robot-manipulation",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Egocentric video offers a scalable source of physical interaction experience, yet translating it into robot-executable knowledge and enabling continual adaptation remain challenging. We introduce Zeva-Ego, a unified framework that learns physical priors from human experience and evolves through robot interaction. An Action-Centric Encoder (ACE) converts egocentric visual transitions into action-centered supervision for VLA mid-training, while In-Context Causal Learning (ICCL) enables parameter-free adaptation from action-effect feedback at deployment. Scaling Ego data to 10K hours improves RoboTwin success from 63.8% to 75.3%, matching 2K hours of robot demonstrations (74.7%), corresponding to an empirical data ratio of roughly 4-5:1. With accumulated interaction experience, ICCL further improves success from 58% to 89% within four attempts without parameter updates. These results demonstrate a scalable path toward embodied intelligence that learns from human experience and continuously improves through its own interaction."
    }
]