[
    {
        "id": "osp-17648",
        "type": "article-journal",
        "title": "Is In-Domain Training Enough for Fine-Grained Industrial Anomaly Understanding?",
        "author": [
            {
                "family": "Zhang",
                "given": "Xingwu"
            },
            {
                "family": "Du",
                "given": "Duanyang"
            },
            {
                "family": "Zhu",
                "given": "Huiling"
            },
            {
                "family": "Dai",
                "given": "Jiayue"
            },
            {
                "family": "Liu",
                "given": "Yixiao"
            },
            {
                "family": "Liu",
                "given": "Guozhi"
            },
            {
                "family": "Zhang",
                "given": "Zhihan"
            },
            {
                "family": "Long",
                "given": "Zijun"
            }
        ],
        "URL": "https://omanscience.com/en/articles/is-in-domain-training-enough-for-fine-grained-industrial-anomaly-understanding",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "A single multimodal large language model (MLLM) struggles to excel simultaneously at detection, localization, description, and reasoning in multimodal industrial anomaly understanding (MM-IAU). We show that in-domain training does not close this gap. On MMAD, a widely adopted MM-IAU benchmark, trained specialists reach at most 75.5% accuracy in defect localization, against 92.3% for human experts, and even detect anomalies less accurately than their untrained base model. Meanwhile, different MLLMs offer complementary strengths but share this weakness in fine-grained perception, so combining them alone cannot remove it. We therefore propose SiGMA, a spatially grounded multi-agent framework that divides labor between heterogeneous MLLM agents and a dedicated visual defect expert. A multimodal searcher supplies industrial knowledge and normal references, the defect expert turns query-reference comparison into calibrated anomaly evidence, and a label-free reliability controller weighs each source by task-wise competence and query-level evidence quality. SiGMA reaches 85.2% average accuracy on MMAD, 4.0% above the strongest trained specialist and Gemini-2.5-Pro and within 1.5% of human experts. Even with three agents of at most 9B parameters, it reaches 84.4%, and new MLLMs join without retraining."
    }
]