[
    {
        "id": "osp-24859",
        "type": "article-journal",
        "title": "Presence Is Not Faithfulness: Figurative Vehicle Intrusion in Text-to-Image Generation",
        "author": [
            {
                "family": "Ma",
                "given": "Xiaoyu"
            },
            {
                "family": "Yang",
                "given": "Chen"
            },
            {
                "family": "Chen",
                "given": "Hao"
            }
        ],
        "URL": "https://omanscience.com/en/articles/presence-is-not-faithfulness-figurative-vehicle-intrusion-in-text-to-image-generation",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Text-to-image (TTI) models increasingly generate high-quality images from natural-language prompts, yet figurative language exposes a failure: a vehicle that should guide the depiction of a tenor may instead be rendered as a visible object. We call this failure Figurative Vehicle Intrusion: the intruding content is textually licensed, but it is assigned the wrong visual role, showing that visual presence is not always faithfulness and that presence-oriented evaluation can miss such errors. To study it systematically, we introduce Vehicle Intrusion and Semantic Tenor Assessment (VISTA), a multilingual benchmark of figurative prompts organized by Figurative Form and Mapping Mechanism. We further propose V-Score, a diagnostic question-answering metric that evaluates role-aware figurative faithfulness in generated images. Evaluations on recent high-performing TTI models show that vehicle intrusion persists across languages and figurative categories. As a lightweight mitigation, we introduce VISTA-Guard, which partially reduces vehicle intrusion and suggests a practical path toward more figuratively faithful TTI generation. All resources will be released publicly."
    }
]