[
    {
        "id": "osp-22563",
        "type": "article-journal",
        "title": "Settle: Learning When to Stop Reasoning",
        "author": [
            {
                "family": "Brown",
                "given": "Ryan"
            },
            {
                "family": "Fu",
                "given": "Zihao"
            },
            {
                "family": "Russell",
                "given": "Chris"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/settle-learning-when-to-stop-reasoning",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Reasoning models often continue generating after their answers have settled. Settle learns when to stop from answer stability in completed traces. It trains the existing end-of-reasoning token while keeping other predictions close to the base model, and requires only ordinary decoding at inference. On MATH-500 with Qwen3-4B, Settle reduces token count by 40% with a 0.5-percentage-point decrease in accuracy. It gains 6.16 percentage points over supervised fine-tuning on the same traces shortened at their first stable answer, at nearly identical token counts. Its stopping score predicts whether a correct answer will remain correct. Settle extends the accuracy-token-count Pareto frontier of the evaluated stopping methods."
    }
]