[
    {
        "id": "osp-22687",
        "type": "article-journal",
        "title": "AwarenessBench: Assessing Cognitive Capabilities of Language Models",
        "author": [
            {
                "family": "Li",
                "given": "Xiaojian"
            },
            {
                "family": "Xu",
                "given": "Rongwu"
            },
            {
                "family": "Zhang",
                "given": "Tianyun"
            },
            {
                "family": "Wang",
                "given": "Yue"
            },
            {
                "family": "Chen",
                "given": "Shuo"
            },
            {
                "family": "Lyu",
                "given": "Qiner"
            },
            {
                "family": "Zhang",
                "given": "Briana"
            },
            {
                "family": "Yang",
                "given": "Peiran"
            },
            {
                "family": "Chen",
                "given": "Kyle Xue"
            },
            {
                "family": "Shi",
                "given": "Haoyuan"
            },
            {
                "family": "Wang",
                "given": "Yu"
            },
            {
                "family": "Xu",
                "given": "Wei"
            }
        ],
        "URL": "https://omanscience.com/ar/articles/awarenessbench-assessing-cognitive-capabilities-of-language-models",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "As language models (LMs) exhibit increasingly consciousness-like behaviors, evaluating their cognitive abilities becomes essential. We introduce AwarenessBench, the first comprehensive benchmark for assessing the cognitive abilities of LMs in four dimensions: metacognition, self-awareness, social awareness, and situational awareness, covering 15 cognitive functions and 14,381 samples. Evaluating 18 state-of-the-art LMs, we find that all consistently surpass random baselines, with more advanced models performing better. We further compare LMs with human performance across three demographic groups, where the best-performing model surpasses human averages overall, but most still fall markedly short in metacognition and self-awareness. Finally, we show that awareness is a distinct capability: progress in language modeling or reasoning does not necessarily translate into improved cognition."
    }
]