[
    {
        "id": "osp-16265",
        "type": "article-journal",
        "title": "XTurnix: Large-Scale Self-Supervised Turn Control through Two-State Binary Decisions",
        "author": [
            {
                "family": "Liu",
                "given": "Zhanxun"
            },
            {
                "family": "Duan",
                "given": "Yifan"
            },
            {
                "family": "Wu",
                "given": "Hengtao"
            },
            {
                "family": "Yang",
                "given": "Chen"
            },
            {
                "family": "Cheng",
                "given": "Qinyuan"
            },
            {
                "family": "Wang",
                "given": "Kun"
            },
            {
                "family": "Zeng",
                "given": "Xingyu"
            },
            {
                "family": "Qiu",
                "given": "Xipeng"
            },
            {
                "family": "Lu",
                "given": "Chaochao"
            },
            {
                "family": "Chen",
                "given": "Xie"
            }
        ],
        "URL": "https://omanscience.com/en/articles/xturnix-large-scale-self-supervised-turn-control-through-two-state-binary-decisions",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "General turn-taking behavior in real-time dialogue systems requires deciding whether to keep listening or start responding while listening, and whether to continue or stop while speaking. Existing turn detectors use heterogeneous, task-specific label spaces and are often trained on limited annotations or evaluated on isolated utterances, making them difficult to use as a unified causal controller with comprehensive context. We propose XTurnix, a compact text-based model that formulates turn control as two binary decisions conditioned on the AI's current listening or speaking state and predicts a single control token from the complete dialogue history. XTurnix is pretrained on 5.5 million causal action examples automatically derived from timestamped two-speaker transcripts, then fine-tuned on synthetic multi-turn examples with a flatter distribution across the four state-action labels. We evaluate XTurnix on four public benchmarks and a balanced self-curated benchmark. Across the public benchmarks, XTurnix achieves the best results on all SemanticVAD and LiveKit splits, ties the native Smart-Turn model on Smart-Turn Bench, and achieves the highest incomplete-turn accuracy on Easy-Turn. On the self-curated benchmark, it reaches 89.06% accuracy, more than 20 percentage points above the strongest third-party baseline at 68.75%, while maintaining F1 scores between 84.21% and 90.63% across all four categories. These results demonstrate unified listening- and speaking-state turn control in a single compact model. Code is available at https://github.com/xcc-zach/xturnix, with an interactive demo at https://huggingface.co/spaces/xcczach/xturnix-demo."
    }
]