[
    {
        "id": "osp-16895",
        "type": "article-journal",
        "title": "MASKerade: Token-Routed Mask Experts for Dense-to-MoE Upcycling",
        "author": [
            {
                "family": "Zhang",
                "given": "Mingyuan"
            },
            {
                "family": "Bai",
                "given": "Yue"
            },
            {
                "family": "Wang",
                "given": "Zhongruo"
            },
            {
                "family": "Huang",
                "given": "Yupin"
            },
            {
                "family": "Huang",
                "given": "Yiyang"
            },
            {
                "family": "Wang",
                "given": "Hailing"
            },
            {
                "family": "Zeng",
                "given": "Huimin"
            },
            {
                "family": "Fu",
                "given": "Yun"
            }
        ],
        "URL": "https://omanscience.com/en/articles/maskerade-token-routed-mask-experts-for-dense-to-moe-upcycling",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Sparsely activated Mixture-of-Experts (MoE) models increase model capacity without a proportional increase in per-token computation. Dense-to-MoE upcycling reuses pretrained dense models to construct such systems, commonly by copying feed-forward networks (FFNs) into independently trained experts. We introduce MASKerade, a dense-to-MoE training method that instead learns experts as sparse subnetworks of a frozen pretrained FFN. Each expert is defined by a learned binary mask, and a token-level router selects which masked FFNs to execute and combine. The router and mask scores are optimized jointly, while the underlying FFN weight values remain unchanged. This formulation supports neuron-structured, semi-structured, and unstructured experts within the same routing architecture. Our main configuration uses four 2:4 experts with top-2 routing, where two half-dense expert passes have the nominal FFN arithmetic of one dense pass, without requiring independent expert weight matrices. On five vision-language benchmarks with Qwen and Gemma backbones, this configuration achieves the highest performance among the compared baselines. Comparisons across mask granularities, routing interventions, and compute-matched controls distinguish the effects of learned connectivity from expert activation count. These results establish mask learning over frozen weights as a practical alternative for constructing token-routed MoE experts."
    }
]