[
    {
        "id": "osp-15796",
        "type": "article-journal",
        "title": "PlaySuite: A Large-Scale Benchmark for Interactive Visual Intelligence",
        "author": [
            {
                "family": "Varghese",
                "given": "Dheeraj"
            },
            {
                "family": "Vettoruzzo",
                "given": "Anna"
            },
            {
                "family": "Simoncini",
                "given": "Walter"
            },
            {
                "family": "Callejas",
                "given": "Michelle Lorena Acevedo"
            },
            {
                "family": "Derakhshani",
                "given": "Mohammad Mahdi"
            },
            {
                "family": "Meding",
                "given": "Kristof"
            },
            {
                "family": "Vanschoren",
                "given": "Joaquin"
            },
            {
                "family": "Snoek",
                "given": "Cees G. M."
            }
        ],
        "URL": "https://omanscience.com/en/articles/playsuite-a-large-scale-benchmark-for-interactive-visual-intelligence",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Recent advances in multimodal foundation models yield strong performance on static perception and reasoning benchmarks, yet such evaluations largely overlook a central aspect of intelligence: acting competently in dynamic environments over extended time horizons. We introduce PlaySuite, a large-scale benchmark for evaluating interactive visual intelligence across more than 5K open-source video games curated from PyWeek and itch.io. Spanning diverse genres and engines, including Pygame, HTML5, Godot, and Unity, these independent games are largely out-of-distribution for current models, reducing the likelihood that success can be achieved by retrieving memorized walkthroughs or web-scale training artifacts. To enable scalable evaluation across heterogeneous titles, we develop a unified closed-loop interaction framework optimized for HPC clusters alongside a Video-LLM-as-a-judge protocol that maps observable gameplay milestones to standardized progress levels. We evaluate fourteen recent open models spanning vision-language models, computer-use agents, and vision-language-action models. Our results yield strong evidence of a perception-action gap: despite strong reasoning capabilities, current models struggle to make sustained progress and exhibit recurring failures in spatial grounding, action execution, and self-correction. PlaySuite provides a reproducible and extensible testbed for measuring progress from visual perception to goal-directed interaction, and a foundation for developing models that can act, adapt, and generalize in dynamic visual environments."
    }
]