[
    {
        "id": "osp-19517",
        "type": "article-journal",
        "title": "SPEAR: A Spectral-Disentangled MoE Neural Operator with Knowledge-Guided Expert Aggregation for Large-Scale PDE Pretraining",
        "author": [
            {
                "family": "Sun",
                "given": "Dengdi"
            },
            {
                "family": "Zhou",
                "given": "Xiaoya"
            },
            {
                "family": "Wang",
                "given": "Xiao"
            },
            {
                "family": "Lyu",
                "given": "Wanli"
            },
            {
                "family": "Tang",
                "given": "Jin"
            },
            {
                "family": "Luo",
                "given": "Bin"
            }
        ],
        "URL": "https://omanscience.com/en/articles/spear-a-spectral-disentangled-moe-neural-operator-with-knowledge-guided-expert-aggregation-for-large-scale-pde-pretraining",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Large-scale pre-training has improved the generalization of neural operators across diverse PDEs. However, existing PDE foundation models still struggle with heterogeneous dynamics, where shared representations may cause knowledge interference, while mixture-of-experts (MoE) architectures suffer from increasing expert redundancy. We propose SPEAR, a spectral-disentangled MoE neural operator with knowledge-guided expert aggregation for large-scale PDE pre-training. SPEAR decouples latent features into low- and high-frequency components, enabling shared modeling of transferable dynamics and specialized learning of PDE-specific patterns. To address expert redundancy, we design a knowledge-guided expert aggregation strategy that measures expert similarity from dataset-specific learned knowledge and routing preferences, enabling the identification and consolidation of similar experts. Experiments on twelve PDE datasets and multiple downstream benchmarks demonstrate superior performance in pre-training, fine-tuning, and transfer learning. Furthermore, our aggregation strategy reduces the number of experts by 50\\% while maintaining or improving prediction accuracy, achieving a balance between model efficiency and generalization for PDE foundation models."
    }
]