[
    {
        "id": "osp-19716",
        "type": "article-journal",
        "title": "Hierarchical Continuous Diffusion Language Models",
        "author": [
            {
                "family": "Ren",
                "given": "Hui"
            },
            {
                "family": "Li",
                "given": "Zihan"
            },
            {
                "family": "Liu",
                "given": "Chang"
            },
            {
                "family": "Liu",
                "given": "Huidong"
            },
            {
                "family": "Schwing",
                "given": "Alexander"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/hierarchical-continuous-diffusion-language-models",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Discrete diffusion language models offer a compelling alternative to autoregressive generation for tasks demanding bidirectional reasoning and global constraint satisfaction. Yet they share a structural bottleneck: when decoding in parallel, each token is sampled independently from its marginal, severing the statistical dependencies among the tokens decoded together. Continuous diffusion language models avoid this by denoising a shared continuous state, but their denoiser sees only that state, so nothing ties it to a valid token configuration until it is finally decoded. To address this, we propose Hierarchical Continuous Diffusion Language Models (HC-DLM), which couple discrete token generation with a continuous latent trajectory in a single, principled denoising process, whose training objective is derived from a variational bound on the token likelihood. In contrast to recent methods that attach continuous context to a self-contained discrete chain, HC-DLM makes the latent the only persistent generative state: tokens are read out from it at every step and feed back as a scaffold for the next latent update. On structured reasoning (Sudoku), mathematical planning (Countdown) and language modeling (LM1B), HC-DLM improves over discrete and continuous diffusion baselines at matched model size, in puzzle accuracy on Sudoku and Countdown and in generative perplexity on LM1B. Project page: https://hc-dlm.github.io/."
    }
]