[
    {
        "id": "osp-23925",
        "type": "article-journal",
        "title": "Feedforward Novel View Synthesis for Heterogeneous Cameras",
        "author": [
            {
                "family": "Wei",
                "given": "Meng"
            },
            {
                "family": "Zhang",
                "given": "Cheng"
            },
            {
                "family": "Li",
                "given": "Boying"
            },
            {
                "family": "Chen",
                "given": "Yihang"
            },
            {
                "family": "Zheng",
                "given": "Jianmin"
            },
            {
                "family": "Rezatofighi",
                "given": "Hamid"
            },
            {
                "family": "Cai",
                "given": "Jianfei"
            }
        ],
        "URL": "https://omanscience.com/en/articles/feedforward-novel-view-synthesis-for-heterogeneous-cameras",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Feed-forward novel view synthesis has recently shown promising results from sparse posed images, but most existing methods assume that context and target views share a fixed camera family. This homogeneous-camera assumption breaks in practical multi-sensor systems, where perspective, fisheye, and panoramic cameras may coexist and where the target projection may be unseen during training. We study feed-forward NVS across heterogeneous central cameras and identify a key ambiguity introduced by tokenization: a visual token aggregates a projection-dependent bundle of pixel rays, while existing camera encodings mainly expose absolute rays or token-center relations. To address this, we combine token-center relative Camera Positional Encodings and proposed local raymaps, a token-level representation that explicitly describes the intra-patch ray distribution summarized by each token. We further propose projection-aware 2D RoPE, which replaces raw image-grid coordinates with ray-induced angular coordinates so that relative positional reasoning is aligned across camera projections. Together, these components treat diverse cameras as calibrated samplings of a shared ray space rather than separate visual domains. On ScanNet++ with heterogeneous-camera system, our method improves over camera-conditioned baselines under mixed-camera evaluation and demonstrates zero-shot generalization to panoramic views."
    }
]