[
    {
        "id": "osp-24154",
        "type": "article-journal",
        "title": "FAST: Flow Any Scene Transformer",
        "author": [
            {
                "family": "Zhang",
                "given": "Yongjian"
            },
            {
                "family": "Wang",
                "given": "Longguang"
            },
            {
                "family": "Song",
                "given": "Zhuo"
            },
            {
                "family": "Fu",
                "given": "Zhiheng"
            },
            {
                "family": "Lin",
                "given": "Liang"
            },
            {
                "family": "Guo",
                "given": "Yulan"
            }
        ],
        "URL": "https://omanscience.com/en/articles/fast-flow-any-scene-transformer",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Scaling has become a primary driver of progress in language and vision foundation models, yet its role in precise correspondence matching remains underexplored. In this work, we present Flow Any Scene Transformer (FAST), a scalable correspondence model driven by two key insights. First, we reveal that the query-key projections inside single-view vision foundation models encode a coarse yet reusable prior for cross-view matching. Second, reusing these pretrained projections in cross-attention form yields a highly effective initialization for a ViT-based matcher built from a single-view encoder. Guided by these insights, we build FAST upon a vanilla single-view foundation model, utilizing a zero-parameter rewiring strategy to convert selected self-attention layers into cross-attention for cross-view interaction. This design allows ViT-based matchers to scale with advances in single-view foundation models, bypassing the need for a dedicated pair-centric pretraining stage. To fully unlock the scaling potential of this formulation, we assemble a 6-million-pair training corpus for general-purpose dense 2D displacement estimation across diverse co-visible image pairs. Extensive experiments demonstrate that FAST achieves state-of-the-art performance across a wide range of benchmarks, while scaling favorably with both backbone size and training data."
    }
]