[
    {
        "id": "osp-20760",
        "type": "article-journal",
        "title": "How Medical VLMs Underutilize Their Vision Encoders: A Dermatology Perspective",
        "author": [
            {
                "family": "Wang",
                "given": "Janet"
            },
            {
                "family": "Zhang",
                "given": "Yunbei"
            },
            {
                "family": "Wang",
                "given": "Xiao"
            },
            {
                "family": "Hamm",
                "given": "Jihun"
            }
        ],
        "URL": "https://omanscience.com/en/articles/how-medical-vlms-underutilize-their-vision-encoders-a-dermatology-perspective",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Medical Vision-Language Models (VLMs) show significant promise for clinical image understanding, offering accurate diagnosis with interpretable reasoning. However, a critical performance gap exists between their strong vision encoders and the full multimodal model: in dermatology, the MedSigLIP encoder outperforms MedGemma by an average of 10.26 percentage points even when both use zero target-task labels; few-shot linear probing provides further evidence of strong visual representations. This gap motivates an investigation of how visual information is used in end-to-end diagnosis and why plausible-sounding predictions can lack grounding in image evidence. Using dermatology as our primary testbed, we systematically investigate three hypotheses for this phenomenon. We further provide a mechanistic analysis of the model's internal attention patterns, showing that a simple describe-then-decide prompting strategy increases vision attention by 30-40% during generation. Task-specific fine-tuning improves dermatology classification but reduces cross-domain medical question-answering performance in our evaluation. To address these challenges, we combine label-free prompting with low-label encoder-assisted reranking while keeping the VLM frozen. We validate the interventions across five VLM backbones in dermatology and provide supporting representation and attention analyses across additional medical modalities."
    }
]