[
    {
        "id": "osp-15690",
        "type": "article-journal",
        "title": "Linear Fitness Subspace in Protein Language Models Enables Sample-Efficient Directed Evolution",
        "author": [
            {
                "family": "Ma",
                "given": "Siyuan"
            },
            {
                "family": "Xiao",
                "given": "Canran"
            },
            {
                "family": "Xiao",
                "given": "Zikai"
            },
            {
                "family": "Gao",
                "given": "Albert"
            },
            {
                "family": "He",
                "given": "Liang"
            },
            {
                "family": "Wang",
                "given": "Xuan-Yu"
            },
            {
                "family": "Cao",
                "given": "Shuying"
            },
            {
                "family": "Jia",
                "given": "Xiaojun"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/linear-fitness-subspace-in-protein-language-models-enables-sample-efficient-directed-evolution",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Model-guided directed evolution seeks to identify high-fitness protein variants under limited oracle budgets. Protein language models (PLMs) provide rich representations for this task, but task-agnostic zero-shot scores can be misaligned with a target assay, while supervised search in high-dimensional embedding spaces can make surrogate modeling and uncertainty estimation sample-inefficient. We propose the Linear Fitness Subspace (LFS) hypothesis: within mutation-induced residue-level representation changes, a compact, assay-specific set of directions makes fitness variation linearly accessible from few labeled variants. This is a local, supervision-recoverable statement rather than a claim that protein fitness landscapes or global PLM geometry are universally linear. Building on this observation, we introduce Subspace-Guided Evolutionary Search (SGES), which estimates an LFS from a small initial sample and performs surrogate modeling, uncertainty estimation, and acquisition in the learned subspace. Across 10 core ProteinGym assays, 87 extended static-validation assays, and an 18-assay budgeted-search evaluation, SGES improves fitness prediction and search efficiency over zero-shot PLMs and recent ML-guided protein optimization baselines. Controlled comparisons with PCA, random projections, label-shuffled PLS, classical mutation features, and acquisition ablations further isolate the benefit of a fitness-aligned site-delta coordinate."
    }
]