[
    {
        "id": "osp-15895",
        "type": "article-journal",
        "title": "Cross-lingual Calibration of Pre-Generation Success Probes for Multilingual LLM Routing",
        "author": [
            {
                "family": "Paganelli",
                "given": "Andrea"
            },
            {
                "family": "Civelli",
                "given": "Stefano"
            },
            {
                "family": "Bernardelle",
                "given": "Pietro"
            },
            {
                "family": "Demartini",
                "given": "Gianluca"
            }
        ],
        "URL": "https://omanscience.com/en/articles/cross-lingual-calibration-of-pre-generation-success-probes-for-multilingual-llm-routing",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Pre-generation success probes estimate response correctness from a language model's hidden activations before decoding, enabling cost-aware routing. While prior work has demonstrated their utility primarily on English inputs, we study their reliability across languages along three dimensions: (1) whether they preserve the ranking of likely successes and failures (DISCRIMINATION); (2) whether they retain probabilities that match observed success frequencies (CALIBRATION); and (3) whether they produce scores comparable enough across candidate models for cost-aware multilingual routing (UTILITY). Using 3,000 MATH problems in 10 languages and 8 open-weight model configurations, we compare cross-lingual transfer from English-trained probes and equal-budget pooled multilingual probes. English-trained probes retain useful cross-lingual discrimination but become less well calibrated after transfer. Pooled multilingual supervision improves both properties and yields more reliable estimates of success. In routing experiments, the pooled router achieves a 0.7% higher test success rate while reducing modeled cost by 13.0% relative to always selecting the model with the highest average success. These results show that multilingual routing requires success estimates that remain well calibrated and comparable across languages and models."
    }
]