[
    {
        "id": "osp-24187",
        "type": "article-journal",
        "title": "MegaAvatar: Controllable Talking Avatar Generation",
        "author": [
            {
                "family": "Gao",
                "given": "Junyao"
            },
            {
                "family": "Liu",
                "given": "Sibo"
            },
            {
                "family": "Zhang",
                "given": "Weidong"
            },
            {
                "family": "Zhao",
                "given": "Cairong"
            },
            {
                "family": "Zhang",
                "given": "Jun"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/megaavatar-controllable-talking-avatar-generation",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "This report presents \\textbf{MegaAvatar}, a controllable talking avatar generation framework built on top of the Wan2.2-TI2V-5B model. Compared with previous talking-avatar methods that mainly rely on audio or reference-image conditioning, we introduce additional SMPL-X-derived 3D guidance, enabling global control over body pose and head motion. Specifically, we render the driving SMPL-X sequence into dense mesh frames and encode them with a lightweight 3D convolutional encoder, whose outputs are injected into the latent tokens to provide overall motion control. Furthermore, we extend Wan2.2-TI2V-5B with additional audio and face cross-attention modules to enable fine-grained expression control and preserve the input identity, respectively. In addition, we implement an audio-to-SMPL-X model to predict an SMPL-X sequence conditioned on the reference image and input audio, allowing MegaAvatar to support audio-driven inference without user-provided SMPL-X frames. Experiments show that MegaAvatar achieves high-quality talking avatar generation with controllable body and head motion, speech-synchronized facial expressions, and consistent identity preservation. MegaAvatar also supports inference with flexible resolutions and video lengths. Codes, dataset, models will be avaliable in https://github.com/Jeoyal/MegaAvatar"
    }
]