[
    {
        "id": "osp-20209",
        "type": "article-journal",
        "title": "RefCon: Iterative Refinement and Contrastive Memory Extraction for Context-Evolving Agent",
        "author": [
            {
                "family": "Prathama",
                "given": "Ubaidillah Ariq"
            },
            {
                "family": "Liu",
                "given": "Bo"
            },
            {
                "family": "Hong",
                "given": "Yeo Boon"
            },
            {
                "family": "Huang",
                "given": "Yu-Xuan"
            },
            {
                "family": "Ding",
                "given": "Yangkai"
            },
            {
                "family": "Yu",
                "given": "Tao"
            }
        ],
        "URL": "https://omanscience.com/en/articles/refcon-iterative-refinement-and-contrastive-memory-extraction-for-context-evolving-agent",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Long-horizon agent interactions generate useful but noisy experience, and retraining models to absorb it is expensive. Context-evolving agents therefore need memory extraction methods that improve with more test-time compute without relying on gold labels. We propose RefCon, which combines sequential self-refinement with parallel self-contrast to extract higher-quality memories without gold labels. Evaluated on AppWorld and BFCL-V3 across multiple context-evolving agent frameworks, RefCon delivers strong and consistent gains, including relative improvements of 21.6% on ACE and 16.6% on ReMe over no-scaling baselines, while a diversity-focused variant (DivCon) achieves a 35.5% gain on ReasoningBank. RefCon consistently outperforms existing baselines without ground-truth labels, and generalizes across model scales and to software engineering tasks, where it surpasses even ground-truth baselines. We further analyze the accuracy-token trade-off and scaling behavior, showing RefCon maintains favorable efficiency and continues to improve as more trajectories are used, unlike diversity-only scaling which saturates earlier."
    }
]