[
    {
        "id": "osp-20654",
        "type": "article-journal",
        "title": "Salt++: Context-Aligned Post-Training for Few-Step Streaming Multimodal Generation",
        "author": [
            {
                "family": "Ge",
                "given": "Xingtong"
            },
            {
                "family": "Wang",
                "given": "Yutong"
            },
            {
                "family": "Zhu",
                "given": "Lunjie"
            },
            {
                "family": "Lin",
                "given": "Haitao"
            },
            {
                "family": "Lin",
                "given": "Fangyu"
            },
            {
                "family": "Huang",
                "given": "Yushi"
            },
            {
                "family": "Zhang",
                "given": "Xin"
            },
            {
                "family": "Zhang",
                "given": "Yi"
            },
            {
                "family": "Liu",
                "given": "Yu"
            },
            {
                "family": "Zhang",
                "given": "Jun"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/salt-context-aligned-post-training-for-few-step-streaming-multimodal-generation",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Few-step streaming audio--video generation requires both causal modeling and step distillation, yet standard training recipes face two context-related challenges. Teacher forcing pairs clean history with a noisy target, but supervises predictive contextual representations only indirectly through velocity prediction. Meanwhile, directly reusing bidirectional score models in causal Distribution Matching Distillation (DMD) creates a mismatch between generation and scoring contexts. We address these challenges with Salt++, a two-stage post-training framework comprising Causal Self-Flow (CSF) and context-aligned autoregressive DMD. CSF exploits contextual information asymmetry by varying the history while keeping the noisy target fixed: a noise-mixed-history student aligns its intermediate representations with those of a clean-history exponential-moving-average teacher. This self-supervised signal encourages the student to extract semantic information and improves cross-modal alignment. Context-aligned AR DMD shares the causal mask and prefix across generator sampling, fake-score training, and real-score evaluation to match generated and reference distributions under a block-conditional KL objective. With calibrated teacher guidance, it performs clean-prefix few-step distillation and then adapts to generated histories without switching objectives or requiring separate consistency distillation. At 480p, Salt++ improves visual and motion quality by 57% and 45% over OmniForcing on JavisBench under the same 4-step causal setting. A separate scale-wise post-training stage extends Salt++ to 4-step $1664\\times960$ generation, outperforming bidirectional LTX-2 on six of seven reported metrics. Project page: https://xingtongge.github.io/Saltpp"
    }
]