[
    {
        "id": "osp-23637",
        "type": "article-journal",
        "title": "Uni-LaDiR: Latent Diffusion Unifies Multimodal Reasoning",
        "author": [
            {
                "family": "Kang",
                "given": "Haoqiang"
            },
            {
                "family": "Zhang",
                "given": "Yizhe"
            },
            {
                "family": "Kuang",
                "given": "Nikki Lijing"
            },
            {
                "family": "Gu",
                "given": "Jiatao"
            },
            {
                "family": "Ma",
                "given": "Yi-An"
            },
            {
                "family": "Qin",
                "given": "Lianhui"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/uni-ladir-latent-diffusion-unifies-multimodal-reasoning",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Multimodal models increasingly think with different modalities such as images, 3D point clouds, and robot states, not just text. Yet each modality is still encoded into its own representation space, creating a modality-switching gap whenever reasoning moves from one modality to another. In this paper, we introduce Uni-LaDiR (Unified Latent Diffusion Reasoner), a framework that unifies different modalities into a shared latent space for multimodal reasoning. A unified encoder maps teacher reasoning steps from different modalities into latent thought tokens in a shared space, trained to extract the information needed for later reasoning steps and the final output. A diffusion reasoner, trained jointly with the encoder, generates these tokens at inference without teacher reasoning steps. Across eleven vision-language model (VLM) benchmarks and two vision-language-action (VLA) suites, Uni-LaDiR achieves relative gains over the strongest baselines of 7.3% on four mathematical and logical VLM benchmarks and 6.1% on RLBench manipulation tasks. Controlled comparisons show increasing gains as more teacher modalities are unified. These results suggest that unification improves multimodal reasoning by weaving it into a single thread, where the model predicts successive thoughts in a common representation space."
    }
]