[
    {
        "id": "osp-23175",
        "type": "article-journal",
        "title": "Small yet Assistive: Spatially-Aware Post-Training for Low Vision",
        "author": [
            {
                "family": "Choudhary",
                "given": "Rishabh"
            },
            {
                "family": "Raj",
                "given": "Shreyansh"
            },
            {
                "family": "Goyal",
                "given": "Umesh"
            },
            {
                "family": "Kashyap",
                "given": "Shubh"
            },
            {
                "family": "Kumar",
                "given": "Shrestha"
            },
            {
                "family": "Jena",
                "given": "Sushovan"
            },
            {
                "family": "Kumar",
                "given": "Komal"
            },
            {
                "family": "Cholakkal",
                "given": "Hisham"
            },
            {
                "family": "Nigam",
                "given": "Aditya"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/small-yet-assistive-spatially-aware-post-training-for-low-vision",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "An estimated 1 billion people worldwide live with vision impairment, yet current vision-language models (VLMs) produce descriptions too vague for safe navigation by blind and low-vision (BLV) users. Large VLMs can generate high-quality audio-description-compliant narrations but cannot run on mobile devices; small VLMs offer competitive latency but lack spatial detail, directional cues, and hazard awareness for navigational assistance. We present Smol-VL-BLV, a compact VLM for blind and low-vision users that closes this gap using a 500M decoder transformer model and two post-training mechanisms: (1) teacher-student distillation and (2) Group Relative Policy Optimization (GRPO) with a composite BLV reward targeting directional language, metric distances, and hazard detection. Because multi-stage post-training can induce catastrophic forgetting, we add a lightweight finetuning stage after the last stage GRPO finetuning to recover general descriptive quality while preserving BLV-specific spatial grounding. Our best model substantially outperforms the baseline across various benchmarks, including tasks: VQA, BLV captioning, OCR, and latency. Compared with the baseline for relative improvement, it improves the Spatial score gain of 19.3%, and the Social score gain of 14.8%. It also increases OCR-Bench by 101.5%, and raises TextVQA accuracy by 44.2%. These results show that BLV-focused post-training improves both accessibility-specific spatial grounding and general visual-text reasoning. Deployed on a mid-range Android smartphone via Mixed-Precision Quantization, the model remains approx. 450 MB and runs entirely on-device, offline and without network dependency, generating descriptions with latency dependent on host hardware capabilities. Our model, dataset, and code is publicly released at https://smol-vl-blv.github.io/Smol-VL-BLV-website/"
    }
]