[
    {
        "id": "osp-18039",
        "type": "article-journal",
        "title": "Robust Surgical Robotic Instrument Tracking via Sequential Multi-Cue Fusion and Sim-to-Real Self-Training",
        "author": [
            {
                "family": "Hu",
                "given": "Hanyang"
            },
            {
                "family": "Liang",
                "given": "Zekai"
            },
            {
                "family": "Richter",
                "given": "Florian"
            },
            {
                "family": "Yip",
                "given": "Michael C."
            }
        ],
        "URL": "https://omanscience.com/en/articles/robust-surgical-robotic-instrument-tracking-via-sequential-multi-cue-fusion-and-sim-to-real-self-training",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Efficient and robust tracking of surgical robotic instruments is important for robot-assisted minimally invasive surgery, yet remains challenging due to the complexity of surgical scenes and the unconventional geometry of surgical instruments. Keypoint-based approaches are efficient, but their performance depends on reliable feature detection. Improving these detectors with real-world supervision is difficult because accurate real-world annotations are costly to obtain at scale. To address this limitation, we introduce a tracker-guided self-training framework that adapts a model pretrained on synthetic images to unlabeled real-world videos. Given measured robot joint states, an uncertainty-aware EKF recursively corrects the instrument pose and the observable joint angles by comparing projected model features with detected keypoints, shaft boundaries, and mask-derived cues. An RTS smoother subsequently refines the resulting trajectory, which is projected into pseudo-labels for fine-tuning the feature detector without laborious pose annotations. Experiments on real-world videos demonstrate consistent improvements from self-training across all evaluated keypoint metrics, and the resulting model outperforms prior approaches in both accuracy and runtime. The code and data will be released upon publication."
    }
]