[
    {
        "id": "osp-15348",
        "type": "article-journal",
        "title": "World Potential Model: Pretrained World Knowledge as Progress Potentials",
        "author": [
            {
                "family": "Zhao",
                "given": "Jun"
            },
            {
                "family": "Tang",
                "given": "Jixin"
            },
            {
                "family": "Shu",
                "given": "Yang"
            },
            {
                "family": "Wu",
                "given": "Jinyang"
            },
            {
                "family": "Lu",
                "given": "Yuyang"
            },
            {
                "family": "Tong",
                "given": "Jingqi"
            },
            {
                "family": "Xu",
                "given": "Hao"
            },
            {
                "family": "Ge",
                "given": "Weifeng"
            },
            {
                "family": "Zhang",
                "given": "Qi"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/world-potential-model-pretrained-world-knowledge-as-progress-potentials",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Long-horizon language agents often receive supervision only from terminal task outcomes, leaving little signal for distinguishing productive intermediate behavior from stagnation or even regression. Rather than learning a separate value function or process reward model for every task, we ask whether pretrained models can recognize task progress from their existing world knowledge. We formalize this capability with a World Potential Model (WPM), a goal-conditioned evaluator of task-relative realized progress in agent contexts. In ALFWorld and ScienceWorld, off-the-shelf pretrained models substantially outperform chance at recovering realized-progress structure without task-specific evaluator fine-tuning. We further anchor these progress judgments to task-specific milestones to obtain scalar world potentials, whose temporal differences provide process-sensitive step-level credit for policy optimization. Under matched comparisons, WPM-guided optimization improves success over outcome-only GRPO across all evaluated configurations. Together, these results provide initial evidence that pretrained world knowledge can support reusable realized-progress evaluation and provide useful supervision for long-horizon agents."
    }
]