[
    {
        "id": "osp-17422",
        "type": "article-journal",
        "title": "Beyond Speech Captions: Speech-Rewarded Style Planning for Conversational Text-to-Speech",
        "author": [
            {
                "family": "Zhu",
                "given": "Shiao"
            },
            {
                "family": "Liu",
                "given": "Lianbo"
            },
            {
                "family": "Lyu",
                "given": "Sizhen"
            },
            {
                "family": "Wang",
                "given": "Yuzhe"
            },
            {
                "family": "Li",
                "given": "Sheng"
            },
            {
                "family": "Shinozaki",
                "given": "Takahiro"
            }
        ],
        "URL": "https://omanscience.com/en/articles/beyond-speech-captions-speech-rewarded-style-planning-for-conversational-text-to-speech",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Natural-language style descriptions provide an interpretable interface between large language models (LLMs) and controllable text-to-speech (TTS). However, using descriptions as pseudo-labels compresses target acoustics into text, and descriptive fidelity need not imply effective control of a particular synthesizer. We empirically show that speech-text alignment only weakly predicts downstream acoustic similarity among candidate instructions for the same utterance. We therefore propose Speech-Rewarded Style Planning (SRSP), which trains a text-based style planner through a frozen downstream TTS model. Given dialogue history and response text, the planner generates candidate instructions and is optimized with group-relative policy optimization (GRPO), using the teacher-forced likelihood of target speech tokens as the reward. On an English subset of the ISCSLP 2026 CoT-TTS corpus, SRSP achieves higher speech-style and emotion similarity to target speech and lower mel-cepstral distortion than the Base LLM and target-audio-informed captioning baselines. LLM-based expressive speech evaluation further shows gains over all baselines in contextual appropriateness and reference consistency."
    }
]