[
    {
        "id": "osp-17720",
        "type": "article-journal",
        "title": "OmniDex: Scaling Dexterous Hand Grasping to Diverse Cluttered Scenes",
        "author": [
            {
                "family": "Fang",
                "given": "Naiyu"
            },
            {
                "family": "Luo",
                "given": "Zhongjin"
            },
            {
                "family": "Mo",
                "given": "Yuxin"
            },
            {
                "family": "Huang",
                "given": "Siyuan"
            },
            {
                "family": "Liu",
                "given": "Jianbo"
            },
            {
                "family": "Liu",
                "given": "Yufei"
            },
            {
                "family": "Zhou",
                "given": "Zheyuan"
            },
            {
                "family": "Jin",
                "given": "Chenkai"
            },
            {
                "family": "Wang",
                "given": "Xiaogang"
            },
            {
                "family": "Li",
                "given": "Hongsheng"
            }
        ],
        "URL": "https://omanscience.com/en/articles/omnidex-scaling-dexterous-hand-grasping-to-diverse-cluttered-scenes",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Dexterous grasping is the foundational primitive in embodied AI, demanding massive data to train robust models. As real-world data collection is expensive, simulation has become the mainstream paradigm. Yet, while cluttered scenes best reflect real-world applications, learning to grasp within them is bottlenecked by a critical scarcity of large-scale data. To resolve this, we curate high-quality 3D objects and supporting bases, proposing a scalable seed-and-filter strategy that bypasses sluggish scene-level optimization. This yields an unprecedented benchmark comprising over 2.6 million scenes and 0.4B scene-specific grasp ground truths, featuring diverse realistic layouts paired with rich semantic and geometric observations. Furthermore, we introduce the OmniDex model to overcome the grasp multimodality and last-millimeter precision errors plaguing current generative models. By coupling Soft Winner-Takes-All learning with human-inspired physical constraints during training, and utilizing physics-driven ranking, our approach achieves robust dexterous grasping without the latency of post-optimization. Experimental results show that OmniDex model achieves state-of-the-art performance and strong generalization across diverse scenes, views, and unseen objects."
    }
]