[
    {
        "id": "osp-24267",
        "type": "article-journal",
        "title": "TrackFish3D: Self-Supervised 3D Tracking of Schooling Fish from Multi-view Videos",
        "author": [
            {
                "family": "Phurtivilai",
                "given": "Patt"
            },
            {
                "family": "Dou",
                "given": "Zhiyang"
            },
            {
                "family": "Wu",
                "given": "Yifan"
            },
            {
                "family": "Chu",
                "given": "Kinfung"
            },
            {
                "family": "Liu",
                "given": "Yuan"
            },
            {
                "family": "Yang",
                "given": "Lei"
            },
            {
                "family": "Wang",
                "given": "Wenping"
            },
            {
                "family": "Komura",
                "given": "Taku"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/trackfish3d-self-supervised-3d-tracking-of-schooling-fish-from-multi-view-videos",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Quantifying collective fish behavior requires accurate trajectories, yet multi-view 3D tracking remains challenging due to frequent occlusions, visually similar individuals, and the long-standing scarcity of identity annotations. We present TrackFish3D, a geometry-driven self-supervised framework for dense multi-camera 3D tracking of schooling fish. Instead of relying on appearance-based re-identification or manually annotated identities, TrackFish3D turns calibrated multi-view geometry into supervision: triangulation and reprojection consistency provide pseudo-associations, while a geometric encoder and global association transformer learn all-to-all cross-view correspondence within each frame. To make these associations identity-aware, TrackFish3D introduces a self-supervised contrastive objective that separates co-visible individuals in the embedding space, together with a temporal predictor that preserves identities and bridges short occlusions across frames. The resulting model is trained once on unlabeled footage and applied directly to unseen test videos, requiring no cross-view identity labels, temporal annotations, 3D ground truth, appearance features, or test-time optimization. On our benchmark, TrackFish3D improves 3D Multi-Object Tracking Accuracy from 87.7% for the strongest baseline to 95.8%. On the 3D-ZeF zebrafish benchmark, it achieves 81.1% MOTA, compared with 77.4% for the best geometric baseline. TrackFish3D also generalizes beyond fish, achieving strong results on real-world bird tracking."
    }
]