[
    {
        "id": "osp-17070",
        "type": "article-journal",
        "title": "dIon: Fragmentation-Based Invariance for Self-Supervised Learning of Tandem Mass Spectra",
        "author": [
            {
                "family": "Nilsson",
                "given": "Alfred"
            },
            {
                "family": "Lapin",
                "given": "Joel"
            },
            {
                "family": "Payne",
                "given": "Samuel H."
            },
            {
                "family": "Wilhelm",
                "given": "Mathias"
            },
            {
                "family": "Käll",
                "given": "Lukas"
            }
        ],
        "URL": "https://omanscience.com/en/articles/dion-fragmentation-based-invariance-for-self-supervised-learning-of-tandem-mass-spectra",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "We introduce a novel invariance for peptide tandem mass spectrometry data, unlocking self-supervised representation learning that improves de novo sequencing of peptides. This invariance exploits the physical relationship between precursor properties (mass and charge) and fragment-ion evidence, without requiring peptide sequence labels. We introduce dIon, which adapts the DINO framework with two latent prediction tasks, both recovering a clean teacher representation: one from a spectrum mixture, using the precursor as a selection query, and one from a partial spectrum with the precursor withheld. The first associates precursor information with fragment-ion evidence; the second prevents representational collapse onto that information alone. Mechanistic probes support both effects, and ablations show that the full objective performs best. Under identical end-to-end training, dIon initialization improves de novo peptide precision over training from scratch by 5.5 and 8.4 percentage points on the held-out MassIVE-KB and Kingdoms test sets, and by 2.3 and 4.8 percentage points with a larger supervised training corpus. The resulting models surpass fully supervised state-of-the-art de novo sequencing models on the diverse, multi-species Kingdoms corpus under the same greedy-decoding protocol. Without peptide labels, dIon learns strong native peptide-similarity geometry compared with other learned models; with limited peptide-supervised adaptation, it achieves the best retrieval and pair-discrimination performance across all representation benchmarks."
    }
]