[
    {
        "id": "osp-22494",
        "type": "article-journal",
        "title": "My FAULT: Self-Diagnosis as Credit Assignment in Self-Evolving Agentic Reinforcement Learning",
        "author": [
            {
                "family": "Zhu",
                "given": "Yihua"
            },
            {
                "family": "Liu",
                "given": "Qianying"
            },
            {
                "family": "Qiao",
                "given": "Weixu"
            },
            {
                "family": "Ren",
                "given": "Xuan"
            },
            {
                "family": "Xu",
                "given": "Weiwei"
            },
            {
                "family": "Li",
                "given": "Wenbo"
            },
            {
                "family": "Wang",
                "given": "Wei"
            },
            {
                "family": "Chen",
                "given": "Ruijia"
            },
            {
                "family": "Luan",
                "given": "Xinmiao"
            },
            {
                "family": "Luo",
                "given": "Yin"
            },
            {
                "family": "Huang",
                "given": "Hao"
            },
            {
                "family": "Zheng",
                "given": "Xiang"
            },
            {
                "family": "Shimodaira",
                "given": "Hidetoshi"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/my-fault-self-diagnosis-as-credit-assignment-in-self-evolving-agentic-reinforcement-learning",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Agentic reinforcement learning (RL) has emerged as a powerful approach for training large language model agents on multi-step tasks, yet reliance on terminal outcome rewards creates two credit-assignment problems, particularly in long-horizon tasks. First, same-outcome rollout groups provide no learning signal from terminal rewards. Second, terminal rewards provide only trajectory-wide feedback, making it difficult to identify which decisions caused a failure. Recent work supplements terminal rewards with finer-grained information from trajectory analysis, such as natural-language reflections on intermediate decisions and errors. However, natural-language diagnoses are difficult to use directly for credit assignment: their error claims may be unreliable, and they do not quantify how much each error should affect learning. We propose Self-Diagnosis-guided Terminal Credit Redistribution (FAULT), which turns diagnosed errors into explicit step-level credit anchored by terminal outcomes. FAULT checks diagnostic evidence and learns relative error costs from task outcomes. During training, the policy and self-diagnoser co-evolve, while error costs are updated online from recent outcomes. On ALFWorld, FAULT recovers learning signals from same-outcome groups, reaching 95% signal coverage versus 41% for GRPO and 72% for GiGPO, while better localizing credit to specific error steps. Across two model scales, FAULT delivers strong. improvements on the long-horizon ALFWorld and WebShop tasks while remaining competitive on short-horizon Search-based QA."
    }
]