[
    {
        "id": "osp-23308",
        "type": "article-journal",
        "title": "HySparse2: Hybrid Sparse Attention with Two-Level KV Sharing",
        "author": [
            {
                "family": "Wei",
                "given": "Jianyu"
            },
            {
                "family": "Gao",
                "given": "Yizhao"
            },
            {
                "family": "Zhang",
                "given": "Qihao"
            },
            {
                "family": "Chen",
                "given": "Shimao"
            },
            {
                "family": "Tang",
                "given": "Zhengju"
            },
            {
                "family": "Cheng",
                "given": "Yu"
            },
            {
                "family": "Zhou",
                "given": "Shengjie"
            },
            {
                "family": "Jiang",
                "given": "Zihan"
            },
            {
                "family": "Song",
                "given": "Yifan"
            },
            {
                "family": "Zhang",
                "given": "Hailin"
            },
            {
                "family": "Zhao",
                "given": "Liang"
            },
            {
                "family": "Yang",
                "given": "Bo"
            },
            {
                "family": "Wang",
                "given": "Gang"
            },
            {
                "family": "Cao",
                "given": "Shijie"
            },
            {
                "family": "Luo",
                "given": "Fuli"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/hysparse2-hybrid-sparse-attention-with-two-level-kv-sharing",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Long-horizon and multi-turn agents typically generate short actions and process long observations from tools and environments. This growing context demands efficient prefill, compact KV-cache storage, and accurate long-context retrieval. To meet these demands, we introduce HySparse2, a hybrid sparse attention architecture with two-level KV sharing. At the outer level, KV Bridging adopts a YOCO-style self-decoder and cross-decoder structure, but bridges only full-attention layers. The self-decoder uses hybrid sliding-window attention (SWA), while the cross-decoder uses hybrid sparse attention. The KV caches for full-attention layers in the cross-decoder are generated from the hidden states of full-attention layers in the self-decoder. At the inner level, HySparse2 retains HySparse's core KV Reuse design with two refinements. First, it replaces block-level sparsity with token-level sparsity for finer long-context retrieval. Second, it removes the separate SWA branch from sparse layers and instead forces a sliding window of recent tokens into the sparse selection. This two-level KV sharing allows all cross-decoder KV caches to be constructed from self-decoder hidden states. Prefill can therefore exit after the self-decoder, skipping all cross-decoder layers. On an 80B-A3B MoE model, HySparse2 outperforms HySparse and Hybrid SWA on long-context retrieval and multi-turn agentic tasks, while substantially reducing prefill computation and KV-cache storage."
    }
]