[
    {
        "id": "osp-16821",
        "type": "article-journal",
        "title": "PHBA: Prefix-State Hybrid Block Attention",
        "author": [
            {
                "family": "Li",
                "given": "Ruijie"
            },
            {
                "family": "Hu",
                "given": "Jiaxi"
            },
            {
                "family": "Wang",
                "given": "Shiyu"
            },
            {
                "family": "Liang",
                "given": "Yuxuan"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/phba-prefix-state-hybrid-block-attention",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Hybrid architectures combining linear sequence models with softmax attention provide an effective balance between efficient long-context modeling and precise token retrieval. Existing designs such as Native Hybrid Attention (NHA) combine compressed long-term states with sliding-window attention, but their exact attention is restricted to a fixed local window. In this work, we introduce Prefix-State Hybrid Block Attention (PHBA), which replaces local sliding-window attention with top-k block-sparse retrieval and couples each retrieved block with a compact prefix state summarizing its preceding context. The prefix states are constructed by a gated linear recurrence at block boundaries and retrieved together with the corresponding token blocks, allowing the model to combine precise long-range evidence with compressed historical context within a unified layer. We further develop a hardware-aware Triton implementation that streams routed token blocks and prefix states without materializing large intermediate tensors. Experiments show that PHBA improves long-context and retrieval performance over strong linear and hybrid baselines while retaining efficient training and inference."
    }
]