[
    {
        "id": "osp-23496",
        "type": "article-journal",
        "title": "Block-Sparse Attention with Semantic-Geometric Decoupled Routing",
        "author": [
            {
                "family": "Long",
                "given": "Xinwei"
            },
            {
                "family": "Sun",
                "given": "Weigao"
            },
            {
                "family": "Gao",
                "given": "Weibo"
            },
            {
                "family": "Jiao",
                "given": "Pengkun"
            },
            {
                "family": "Qi",
                "given": "Biqing"
            },
            {
                "family": "Zhu",
                "given": "Feida"
            },
            {
                "family": "Zhong",
                "given": "Yiran"
            },
            {
                "family": "Hoi",
                "given": "Steven"
            },
            {
                "family": "Zhou",
                "given": "Bowen"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/block-sparse-attention-with-semantic-geometric-decoupled-routing",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Long-context inference has become a defining capability of large language models, but exact dense attention remains costly due to its quadratic scaling with sequence length. Block-sparse attention offers a hardware-friendly alternative by routing each query block to a small set of relevant key blocks, yet accurate training-free block routing remains difficult. Existing routers often pool post-RoPE token representations, which entangles semantic aggregation with RoPE-induced geometry and attenuates local positional cues through high-frequency phase cancellation. To resolve this mismatch, we propose \\textbf{Semantic-Geometric Decoupled Routing}, a training-free block routing framework that shifts semantic aggregation to the pre-RoPE space and reconstructs geometric bias with an offline structural prior and relative block distances. This decomposition yields an explicit closed-form block routing score without token-level search or post-hoc calibration. Experiments on long-context text and video tasks show that our method approaches full-attention accuracy across 4K--128K contexts, keeps routing overhead below 3.4 ms, and achieves a 5.03$\\times$ speedup over FlashAttn at a 128K context length."
    }
]