[
    {
        "id": "osp-19634",
        "type": "article-journal",
        "title": "Large Language Continuous Diffusion Models",
        "author": [
            {
                "family": "Yang",
                "given": "Zhihan"
            },
            {
                "family": "Guo",
                "given": "Wei"
            },
            {
                "family": "Lemercier",
                "given": "Jean-Marie"
            },
            {
                "family": "Welker",
                "given": "Simon"
            },
            {
                "family": "Fu",
                "given": "Yonggan"
            },
            {
                "family": "Kamani",
                "given": "Mohammad Mahdi"
            },
            {
                "family": "Norouzi",
                "given": "Sajad"
            },
            {
                "family": "Berner",
                "given": "Julius"
            },
            {
                "family": "Geffner",
                "given": "Tomas"
            },
            {
                "family": "Kreis",
                "given": "Karsten"
            },
            {
                "family": "Chen",
                "given": "Yongxin"
            },
            {
                "family": "Tao",
                "given": "Molei"
            },
            {
                "family": "Thickstun",
                "given": "John"
            },
            {
                "family": "Molchanov",
                "given": "Pavlo"
            },
            {
                "family": "Jukić",
                "given": "Ante"
            },
            {
                "family": "Vahdat",
                "given": "Arash"
            },
            {
                "family": "Mardani",
                "given": "Morteza"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/large-language-continuous-diffusion-models",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Despite the success of discrete diffusion language models (dLMs) for fast parallel decoding, their non-smooth, high-dimensional space hinders trajectory steering for reasoning and inference acceleration. To overcome this, we present Sigma, the first large-scale (3B/8B) continuous dLM built on steerable, low-dimensional ODE/SDE latent trajectories. Trained blockwise via likelihood optimization, Sigma jointly denoises Gaussian-corrupted token embeddings while learning an optimal embedding geometry. To accelerate training, Sigma leverages pre-trained weights from autoregressive (AR) models for warm-starting. During inference, we identify classifier-free guidance and score temperature as essential for high-fidelity reasoning and coding. Across comprehensive math reasoning and coding evaluations against state-of-the-art discrete counterparts (masked dLMs and AR baselines), Sigma achieves competitive performance with discrete models on standard benchmarks (e.g., GSM8K, Minerva, HumanEval, MBPP) after pre-training and on challenging reasoning tasks (e.g., MATH-500, AIME) after supervised fine-tuning. Beyond performance parity, we uncover key structural properties unique to continuous dLMs: (i) embedding-space steering effectively governs the quality-diversity trade-off, yielding strong pass@k performance and (ii) continuous trajectories enable graceful degradation for low NFEs and efficient distillation. These establish continuous dLMs as a promising paradigm for efficient language generation."
    }
]