[
    {
        "id": "osp-22438",
        "type": "article-journal",
        "title": "A Guideline-Augmented Multi-Agent Framework for Schema-as-Code Biomedical Named Entity Recognition",
        "author": [
            {
                "family": "Li",
                "given": "Songtao"
            },
            {
                "family": "Zhang",
                "given": "Yijia"
            },
            {
                "family": "Zhang",
                "given": "Shidi"
            },
            {
                "family": "Yuan",
                "given": "Jianyuan"
            },
            {
                "family": "Zhang",
                "given": "Fengyu"
            },
            {
                "family": "Lin",
                "given": "Hongfei"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/a-guideline-augmented-multi-agent-framework-for-schema-as-code-biomedical-named-entity-recognition",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Large language models (LLMs) have shown promising potential for biomedical named entity recognition (BioNER) through instruction following and in-context learning. However, existing LLM-based BioNER methods still face two key limitations. First, retrieved demonstrations and external biomedical knowledge provide limited support for dataset-specific annotation semantics, leaving entity boundaries, type scopes, and annotation conventions ambiguous. Second, free-form generation lacks sufficient structural control, often leading to invalid formats, hallucinated mentions, duplicated entities, and boundary errors. To address these limitations, we propose GAMA, a guideline-augmented multi-agent framework for schema-as-code BioNER. GAMA first induces candidate annotation rules from labeled training instances and verifies them against annotated data to construct reliable dataset-specific guideline memory. Guided by these verified rules, a planning component generates ranked span-type hypotheses with rationales, and a coding component converts them into schema-constrained entity objects. A verification module then checks span grounding, type validity, and structural compliance, and performs dual-loop refinement to correct invalid or low-confidence predictions. Experiments on five widely used BioNER datasets with multiple LLM backbones show that GAMA consistently outperforms strong LLM-based baselines. Ablation and parameter analyses further verify the effectiveness of the proposed components."
    }
]