[
    {
        "id": "osp-23900",
        "type": "article-journal",
        "title": "Don't Count the Edits, Judge by the Outcome Alone: Reward-Based Evaluation for Grammatical Error Correction",
        "author": [
            {
                "family": "Ryu",
                "given": "Hayeong"
            },
            {
                "family": "Jo",
                "given": "Sunhee"
            },
            {
                "family": "Yu",
                "given": "Seunguk"
            },
            {
                "family": "Kim",
                "given": "YoungBin"
            }
        ],
        "URL": "https://omanscience.com/en/articles/don-t-count-the-edits-judge-by-the-outcome-alone-reward-based-evaluation-for-grammatical-error-correction",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Grammatical error correction (GEC) evaluation has traditionally relied on reference or edit overlap, which can penalize valid rewrites that differ from gold corrections. Reference-free metrics reduce this dependence, but evaluating whether a fluent output is a valid correction of the source remains challenging. We propose SURE, a source-conditioned reward evaluator trained on within-source preferences spanning minimal-edit and rewrite-oriented corrections. SURE jointly learns an overall reward with criteria-level supervision for grammaticality, faithfulness, and fluency, together with span-level grounding for source-side error resolution. Experiments on SEEDA show that SURE performs competitively against strong baselines, with particular gains on rewrite-style corrections and more disentangled criteria-level diagnostics. Our code is available at https://github.com/hayeonggg/SURE."
    }
]