[
    {
        "id": "osp-18065",
        "type": "article-journal",
        "title": "SemCam: Semantic Camera Motion Control for Video Generation",
        "author": [
            {
                "family": "Bruner",
                "given": "Janna"
            },
            {
                "family": "Talmi",
                "given": "Omer"
            },
            {
                "family": "Ideses",
                "given": "Ianir"
            },
            {
                "family": "Fritz",
                "given": "Lior"
            },
            {
                "family": "Wolf",
                "given": "Lior"
            },
            {
                "family": "Benaim",
                "given": "Sagie"
            }
        ],
        "URL": "https://omanscience.com/en/articles/semcam-semantic-camera-motion-control-for-video-generation",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Controlling the camera relative to a moving subject in an existing video is challenging: behaviors such as maintaining a frontal view require the camera to adapt to the subject's changing position and orientation, making the desired trajectory difficult to specify in advance. Existing camera-controlled video-to-video methods typically rely on explicit trajectories or reference motions, which do not directly express these dynamic camera--subject relationships. We introduce semantic camera motion control, a novel video-to-video task in which a reference video and a target motion label specify the desired subject-relative camera behavior without an explicit target trajectory. Our method, SemCam, learns to realize this behavior while preserving source content. It combines shared-basis low-rank adaptation with motion-conditioned modulation, while a background-consistency loss encourages fidelity in regions visible in both reference and target videos. We construct 661 paired videos covering eight semantic camera behaviors and evaluate on a separate 109-scene benchmark using subject-relative motion metrics, appearance measures, and a user study. SemCam achieves a semantic-motion success rate of 68.6%, compared with 45.3% for Vista4D, the strongest evaluated baseline, while maintaining comparable subject identity preservation."
    }
]