[
    {
        "id": "osp-15325",
        "type": "article-journal",
        "title": "From Pareto to Preference: Personalized Test-Time Scaling via Amortized Agentic Policy Discovery",
        "author": [
            {
                "family": "Wang",
                "given": "Xinglin"
            },
            {
                "family": "Liu",
                "given": "Zishen"
            },
            {
                "family": "Zheng",
                "given": "Tong"
            },
            {
                "family": "Feng",
                "given": "Shaoxiong"
            },
            {
                "family": "Yuan",
                "given": "Peiwen"
            },
            {
                "family": "Li",
                "given": "Yiwei"
            },
            {
                "family": "Shi",
                "given": "Jiayi"
            },
            {
                "family": "Zhang",
                "given": "Yueqi"
            },
            {
                "family": "Tan",
                "given": "Chuyi"
            },
            {
                "family": "Zhang",
                "given": "Ji"
            },
            {
                "family": "Pan",
                "given": "Boyuan"
            },
            {
                "family": "Li",
                "given": "Kan"
            }
        ],
        "URL": "https://omanscience.com/en/articles/from-pareto-to-preference-personalized-test-time-scaling-via-amortized-agentic-policy-discovery",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Test-time scaling (TTS) improves the reasoning capabilities of large language models by allocating additional inference computation. Existing approaches to improving TTS efficiency largely optimize accuracy against one resource dimension at a time, advancing either the accuracy--cost or accuracy--latency Pareto frontier. Yet user requirements are multidimensional: users may specify accuracy, latency, and inference-cost requirements jointly, and different requirements can favor different controllers. We formulate Personalized Test-Time Scaling as discovering executable controllers that maximize the joint satisfaction rate of user-specific requirements. To reduce the overhead of repeated policy discovery for new user profiles, we propose PersonTTS, an amortized agentic policy-discovery framework that reuses prior search experience through requirement-matched controller initialization and source-distilled procedural guidance, while retaining target-profile evaluation for every candidate. Experiments on AIME and HMMT show that PersonTTS substantially outperforms strong TTS baselines in joint requirement satisfaction on unseen user profiles and held-out problems. Under the same candidate-evaluation budget, cross-user experience reuse further improves policy quality while substantially reducing discovery-agent time and cost."
    }
]