[
    {
        "id": "osp-23185",
        "type": "article-journal",
        "title": "Can LLMs Reason About Runtime Behavior? A Repository-Level Dynamic Benchmark",
        "author": [
            {
                "family": "Taherkhani",
                "given": "Hamed"
            },
            {
                "family": "Abdollahi",
                "given": "Mohammad"
            },
            {
                "family": "Sepidband",
                "given": "Melika"
            },
            {
                "family": "Dhulipala",
                "given": "Hridya"
            },
            {
                "family": "Nguyen",
                "given": "Tien N."
            },
            {
                "family": "Hemmati",
                "given": "Hadi"
            }
        ],
        "URL": "https://omanscience.com/en/articles/can-llms-reason-about-runtime-behavior-a-repository-level-dynamic-benchmark",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Large language models (LLMs) are increasingly used in coding tasks, but their ability to reason about code execution remains unclear. Existing repository-level QA benchmarks mainly evaluate static code understanding and often rely on LLM-based evaluation, while execution-reasoning benchmarks are mostly limited to snippets or functions. We introduce SWE-Flux, a repository-level benchmark for dynamic execution reasoning containing 480 execution-grounded instances across 12 real Python repositories, with gold answers automatically harvested from instrumented test executions rather than written manually or judged by LLMs. The benchmark covers singletest and multi-test questions over control flow, loops, program state, dataflow, exceptions, and program invariants. Evaluating five LLMs shows that this task remains challenging. The best model achieves only 37% accuracy. Models perform better on localized behavior such as invariants, intra-procedural control flow, exceptions, and simple loops, but struggle with dataflow, inter-procedural execution, precise state reasoning, and suite-level aggregation. Finally, we show that the oracle-harvesting pipeline can generate fresh benchmark variants using input perturbation. It successfully harvests valid variants for almost 90% of the selected instances, and the resulting variants are substantially more challenging for the evaluated models."
    }
]