[
    {
        "id": "osp-22908",
        "type": "article-journal",
        "title": "Gradient-Guided Decoupled Adaptation for Geospatial Vision-Language Models",
        "author": [
            {
                "family": "Wang",
                "given": "Dongdong"
            },
            {
                "family": "Balakrishnan",
                "given": "Deepak"
            },
            {
                "family": "Srinivasan",
                "given": "Ravi"
            },
            {
                "family": "Wang",
                "given": "Shenhao"
            }
        ],
        "URL": "https://omanscience.com/en/articles/gradient-guided-decoupled-adaptation-for-geospatial-vision-language-models",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Existing geospatial vision-language models (Geo-VLMs) typically optimize diverse geospatial tasks through a unified multi-task adaptation paradigm without explicitly accounting for the heterogeneous optimization characteristics. Our empirical observations reveal heterogeneous gradient characteristics across tasks, including vision-language differences, intra-branch gradient relationships, and task interference, which hinder effective multi-task optimization. Motivated by these observations, we propose Gradient-Guided Decoupled Adaptation (G2DA), a gradient-aware optimization framework for multi-task Geo-VLM learning. G2DA first partitions tasks into vision- and language-centric groups through gradient-guided cross-modal decoupling. It then constructs modality-specific curricula based on task gradient similarity and employs bidirectional rehearsal to mitigate the recency effects introduced by sequential optimization. We evaluate G2DA on three Geo-VLM benchmarks using six InternVL3 and Qwen3.5-VL variants, along with GeoChat and GeoLLaVA. Across all 24 benchmark-model combinations, G2DA consistently outperforms representative baselines, improving over the strongest competitor by 3.08, 4.30, and 2.81 percentage points on UrBench-MCQ, XLRS-Bench-Lite, and VRS-Bench-VQA, respectively. These results demonstrate the effectiveness of gradient-guided task organization for Geo-VLM adaptation."
    }
]