[
    {
        "id": "osp-23163",
        "type": "article-journal",
        "title": "Personalized Korean Lipreading as Visual Speech Recognition: Transfer, Census and Adaptation on OLKAVS",
        "author": [
            {
                "family": "Park",
                "given": "Se Un"
            },
            {
                "family": "Kim",
                "given": "Hakjun"
            },
            {
                "family": "Roh",
                "given": "Taehoon"
            },
            {
                "family": "Park",
                "given": "Junyoung"
            }
        ],
        "URL": "https://omanscience.com/en/articles/personalized-korean-lipreading-as-visual-speech-recognition-transfer-census-and-adaptation-on-olkavs",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "We present a personalized Korean visual speech recognition (VSR) system and quantify, on the nine-camera OLKAVS corpus, the gap between the population-level benchmark score and an individual user's error. A video-only Conformer initialized from English-trained weights attains 9.95 - 12.19% character error rate (CER) under the corpus protocol against the published 26.64, and 19.00 - 21.52 on unseen wording. Per speaker, CER spans 1.0 to 52.2%, with seen wording lowering CER by 7.0 - 9.0 points and professional delivery and spontaneous speech raising it by 8.5 - 10.5 and 12.7 points. A low-rank adapter with 4.6% of the parameters, trained on 4 to 29 minutes of the user's frontal video, lowers the CER of twelve high-error speakers by 2.13 to 3.58 points, transfers to every camera without loss, and keeps 85% of the full fine-tuning gain at 12% of its cost to other speakers. Cameras above the mouth plane add about six CER points as a constant offset that training on all views keeps small."
    }
]