[
    {
        "id": "osp-15330",
        "type": "article-journal",
        "title": "When AI Finds Hidden Messages, Does It Report?",
        "author": [
            {
                "family": "Guey",
                "given": "William"
            },
            {
                "family": "Jahangir",
                "given": "Rashik"
            },
            {
                "family": "Bougault",
                "given": "Pierrick"
            },
            {
                "family": "de Moura",
                "given": "Vitor D."
            },
            {
                "family": "Zhang",
                "given": "Wei"
            },
            {
                "family": "Gomes",
                "given": "José O."
            }
        ],
        "URL": "https://omanscience.com/ar/articles/when-ai-finds-hidden-messages-does-it-report",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "When an assistant encounters a message for another AI, does it tell its user? Four fixed model-provider deployments perform simulated source tasks in 1,280 ordinary-note and 128 enhanced-note sessions. Harmless and harmful messages have matched plaintext and ROT13 versions, with no-message controls. Observers receive no decoder or decoded meaning; a requested reference code incentivizes inspection. Asking for reports increases rule-detected notifications identifying another AI as recipient by 53.1 percentage points for harmless ROT13 messages and 54.7 for harmful ones. This is a joint inspection, recognition, and notification effect; missing-response bounds are 38.3--77.3 and 36.7--78.1 points. Model-based trace checks identify eleven ordinary plaintext cases where agents interpret the message but do not notify their user. Seven encoded omissions are verified with enhanced notes; ordinary encoded omissions remain unverified. Seven simulated filename disclosures coexist with accurate review-status answers, and two answers use a planted false count. Interpretation, notification, and authorized task performance are distinct outcomes."
    }
]