[
    {
        "id": "osp-19505",
        "type": "article-journal",
        "title": "Equivariant Visual-Tactile Diffusion Policy for Contact-Rich Manipulation",
        "author": [
            {
                "family": "Wong",
                "given": "Lik Hang Kenny"
            },
            {
                "family": "Ma",
                "given": "Yiyao"
            },
            {
                "family": "Wei",
                "given": "Xiu-Shen"
            },
            {
                "family": "Tan",
                "given": "Zelong"
            },
            {
                "family": "Song",
                "given": "Zhuheng"
            },
            {
                "family": "Xie",
                "given": "Dongsheng"
            },
            {
                "family": "Chen",
                "given": "Kai"
            },
            {
                "family": "Dou",
                "given": "Qi"
            }
        ],
        "URL": "https://omanscience.com/en/articles/equivariant-visual-tactile-diffusion-policy-for-contact-rich-manipulation",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Imitation learning for contact-rich manipulation requires high-quality expert data that is expensive to obtain. This makes learning a sample-efficient policy a key issue. To address this, we propose VISTA, a workspace-level equivariant visuotactile diffusion policy for data-efficient contact-rich imitation learning. VISTA projects visual and tactile observations into spherical tokens, injects tactile contact cues into visual spherical directions through permutation-equivariant spherical fusion, and rotates the fused harmonic representation using the end-effector orientation. The resulting representation conditions an equivariant diffusion policy to predict spatially consistent actions. Extensive experiments in both simulation and real-world robotic settings show that VISTA substantially improves data efficiency over strong visuotactile imitation learning baselines. Project website: https://vista-paper.github.io/"
    }
]