[
    {
        "id": "osp-25200",
        "type": "article-journal",
        "title": "VisionPsy-Nano: Improving Accuracy, Efficiency, and Reliability in On-Device Vision-Language Models",
        "author": [
            {
                "family": "Hashmi",
                "given": "Khurram Azeem"
            },
            {
                "family": "Zolfaghari",
                "given": "Mohammadreza"
            },
            {
                "family": "Park",
                "given": "Changdae"
            },
            {
                "family": "Jain",
                "given": "Rishabh"
            },
            {
                "family": "Moratelli",
                "given": "Nicholas"
            },
            {
                "family": "Wei",
                "given": "Pengfei"
            },
            {
                "family": "Lu",
                "given": "Louis"
            },
            {
                "family": "Liu",
                "given": "Tianchi"
            },
            {
                "family": "Nazir",
                "given": "Amril"
            }
        ],
        "URL": "https://omanscience.com/en/articles/visionpsy-nano-improving-accuracy-efficiency-and-reliability-in-on-device-vision-language-models",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Sub-billion-parameter Vision-Language Models are increasingly viable for on-device deployment, yet compact model size alone does not guarantee usability. On a phone, such a model can still require more than two minutes to produce its first token. On-device usability depends on three axes: accuracy, efficiency, and behavioral reliability; standard benchmarks miss the third, with answers too short to expose doom loops and prompts too benign to probe adversarial safety. We introduce a diagnosis-driven post-training recipe in which a teacher VLM stress-tests the student, uncovers failure modes beyond human priors, and converts them into targeted supervision and preference alignment, supplementing generic data scaling with failure-driven optimization. Coupled with two visual-token policies, the recipe yields two accuracy-efficiency variants with improved behavioral reliability. \\textbf{\\NanoFull} attains a 62.3 normalized average over 17 benchmarks, the highest among openly released $\\sim$0.5B models, +7.4 over its base at identical architecture and token budget, with doom-loop rates at or below the strongest baseline's. \\textbf{\\FlashFull} retains 61.4 while cutting warm time-to-first-token on a Pixel 9 from 138\\,s to 6.1\\,s (23$\\times$). By jointly addressing all three axes, we move compact VLMs toward practical on-device usability."
    }
]