[
    {
        "id": "osp-21760",
        "type": "article-journal",
        "title": "Procedural Core: A Compact Recurrent Initialization for Vision Transformers",
        "author": [
            {
                "family": "Shinnick",
                "given": "Zachary"
            },
            {
                "family": "Internò",
                "given": "Christian"
            },
            {
                "family": "Saratchandran",
                "given": "Hemanth"
            },
            {
                "family": "Hengel",
                "given": "Anton van den"
            },
            {
                "family": "Teney",
                "given": "Damien"
            }
        ],
        "URL": "https://omanscience.com/en/articles/procedural-core-a-compact-recurrent-initialization-for-vision-transformers",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Transformers are typically trained from random initialization, requiring all their capabilities to emerge from large-scale optimization. Recent work showed that a small amount of abstract procedurally generated data can help acquire generic inductive structure at low cost. However, this adds a pretraining stage that must be repeated for every target model. We propose Procedural Core, an initialization strategy that captures this generic structure into a compact set of weights that can be reused across models. We train a minimal recurrent transformer on procedural data, then expand its weights to initialize transformers of arbitrary width and depth. The resulting initialization improves performance on image classification, self-supervised visual learning (DINO), and modeling natural language (FineWeb-Edu) and code (CodeParrot). For image classification, expanding a 1M-parameter core to initialize an 85M-parameter ViT-Base improves ImageNet top-1 accuracy by 2.2 pp over standard random initialization. Our analysis identifies recurrence as essential for learning compact weights that transfer across models. In ViTs, we localize a key benefit in the suppression of high-norm tokens that produces substantial improvements in zero-shot segmentation (ImageNet-S mAP 32.3 to 42.9), object localization (VOC07 CorLoc 9.9 to 18.4), and depth estimation (NYUv2 RMSE 1.104 to 0.998). This demonstrates that transformers need not start from a blank slate, and can be initialized with generic capabilities at low cost with no domain- or task-specific data."
    }
]