[
    {
        "id": "osp-15389",
        "type": "article-journal",
        "title": "Shared Low-rank Basis Factorization for Data-free Mixture-of-Experts Compression",
        "author": [
            {
                "family": "Cao",
                "given": "Tianxiao"
            },
            {
                "family": "Shao",
                "given": "Jiahe"
            },
            {
                "family": "Qiu",
                "given": "Yuning"
            },
            {
                "family": "Atarashi",
                "given": "Kyohei"
            },
            {
                "family": "Kashima",
                "given": "Hisashi"
            },
            {
                "family": "Zhao",
                "given": "Qibin"
            }
        ],
        "URL": "https://omanscience.com/en/articles/shared-low-rank-basis-factorization-for-data-free-mixture-of-experts-compression",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Mixture-of-Experts (MoE) large language models decouple capacity from compute through sparse routing, but their large parameter count creates storage and serving challenges. We analyze three MoE compression families: expert pruning, expert merging, and weight reconstruction, and derive structural error bounds showing that pruning and merging can incur non-vanishing errors tied to routing and expert heterogeneity. In contrast, weight reconstruction avoids these structural costs by preserving expert structure and routing. Motivated by the analysis, we propose Shared Low-rank Basis Factorization (SLBF), a data-free weight reconstruction method that uses rank-$k$ bases shared among experts, enabling richer cross-expert sharing, faster convergence, and lower reconstruction error. A post-hoc gauge fixing removes redundant parameters at no representational cost. Across five MoE architectures spanning 16B to 122B parameters, SLBF consistently outperforms methods from all three compression families."
    }
]