[
    {
        "id": "osp-15292",
        "type": "article-journal",
        "title": "RollVerify: Bridging Efficiency and Accuracy in Long-Tail Rollout Reinforcement Learning",
        "author": [
            {
                "family": "Yao",
                "given": "Yongqiang"
            },
            {
                "family": "Tan",
                "given": "Jinru"
            },
            {
                "family": "Liang",
                "given": "Kaihuan"
            },
            {
                "family": "Yin",
                "given": "Zixin"
            },
            {
                "family": "Niu",
                "given": "Yazhe"
            },
            {
                "family": "Gong",
                "given": "Ruihao"
            },
            {
                "family": "Lin",
                "given": "Dahua"
            },
            {
                "family": "Xu",
                "given": "Ningyi"
            }
        ],
        "URL": "https://omanscience.com/en/articles/rollverify-bridging-efficiency-and-accuracy-in-long-tail-rollout-reinforcement-learning",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Reinforcement learning is crucial for improving large language models' reasoning and generalization. It relies on massive rollouts whose lengths become increasingly long-tailed as context windows grow. In on-policy training, these long-tail rollouts can result in GPU bubbles, reducing system utilization and limiting RL scalability. Asynchronous or partial-rollout methods improve throughput by relaxing synchronization, but inevitably introduce stale off-policy samples (trajectories) that may hurt final accuracy. Existing approaches mainly mitigate this off-policy issue by reweighting off-policy samples during training, yet they can still leave a performance gap compared to fully on-policy training. In this work, rather than passively reweighting samples during training, we propose RollVerify, a lightweight RL framework built on partial rollout that actively verifies and repairs samples before they enter training. Specifically, it introduces an off-policy shift metric OPS, to quantify the off-policy deviation of partially generated trajectories. Guided by the OPS constraint, RollVerify performs both sequence-level and token-level verification to identify and truncate invalid suffixes of trajectories. This yields high-quality samples that protect the models' accuracy while preserving the efficiency gains of partial rollout. Experiments on mathematical and tool-assisted mathematical reasoning show that RollVerify achieves accuracy comparable to on-policy training while reducing training cost. Additional code-generation results provide preliminary evidence beyond mathematics."
    }
]