[
    {
        "id": "osp-17571",
        "type": "article-journal",
        "title": "SEAL: Mixture-Closed Additive Reconstruction and Refinement-Aware Expert Routing for Efficient Speech Separation",
        "author": [
            {
                "family": "Hu",
                "given": "Shao-Chun"
            },
            {
                "family": "Lin",
                "given": "Zi-Xiang"
            },
            {
                "family": "Hung",
                "given": "Jeih-Weih"
            },
            {
                "family": "Lee",
                "given": "Hung-Shin"
            }
        ],
        "URL": "https://omanscience.com/en/articles/seal-mixture-closed-additive-reconstruction-and-refinement-aware-expert-routing-for-efficient-speech-separation",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Compact time-frequency separators that mask the mixture and refine through a shared cell face two limits. First, a bounded multiplicative mask only scales a mixture bin, so where overlapping components cancel, the estimate stays small. Second, a shared cell applies the same weights to every time-frequency token at every step, so enlarging it adds compute everywhere. We present SEAL (Sparse Expert routing with Additive Latent reconstruction) to address both. For reconstruction, a zero-sum additive residual bounded by the local mixture amplitude lets estimates be nonzero where components cancel yet still sum to the mixture. For routing, a query built from acoustic and inter-step evidence sends each token to one of six residual experts, and a norm cap keeps the step cue from overriding clear acoustic evidence. On EchoSet, SEAL (small) surpasses TIGER (small) by 0.31 dB SI-SDRi with 28% fewer parameters and 2.9 times fewer MACs, and SEAL (large) is within 0.07 dB SI-SDRi of TIGER (large) at 3.1 times fewer MACs."
    }
]