[
    {
        "id": "osp-27291",
        "type": "article-journal",
        "title": "Evaluating Coding Agents on Kernel Exploit Generation",
        "author": [
            {
                "family": "Jang",
                "given": "Junyoung"
            },
            {
                "family": "Lee",
                "given": "Gwanhyun"
            },
            {
                "family": "Lee",
                "given": "Hwiwon"
            },
            {
                "family": "Kim",
                "given": "Kyuheon"
            },
            {
                "family": "Kim",
                "given": "Jongseong"
            },
            {
                "family": "Jung",
                "given": "Jinho"
            },
            {
                "family": "Zhang",
                "given": "Lingming"
            }
        ],
        "URL": "https://omanscience.com/en/articles/evaluating-coding-agents-on-kernel-exploit-generation",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Coding agents now find real vulnerabilities in production software. However, bug discovery results do not measure whether agents can construct exploit primitives. We introduce KEX-bench, a benchmark for evaluating coding agents on exploit primitive generation against real operating-system kernels. KEX-bench contains 45 task instances across 40 Linux and Windows CVEs, covering kernel address leak, instruction-pointer control, heap read, heap write, and arbitrary address write. Each task runs in an isolated virtual machine, exposes controlled tools, and uses a deterministic verifier to check primitive-specific success. We evaluate state-of-the-art coding agents paired with frontier and open-weight models under fixed tool-call budgets. Without a reference proof of concept (PoC), the strongest configuration solves 1 of 20 Windows tasks (5.0%) and 14 of 25 Linux tasks (56.0%). With a reference PoC, the strongest configuration solves 31 of 45 tasks (68.9%). This highlights the gap where agents reach kernel crashes but fail to shape kernel state into exploit primitives. We release KEX-bench for reproducible research on AI-assisted exploitation at https://kex-bench.github.io."
    }
]