[
    {
        "id": "osp-26928",
        "type": "article-journal",
        "title": "System-Level Optimization Beyond Cryptographic Kernels: An ML-KEM Case Study on Arm Cortex-M7",
        "author": [
            {
                "family": "Sayed",
                "given": "Mahmoud Abdelhafeez"
            },
            {
                "family": "Taha",
                "given": "Mostafa"
            },
            {
                "family": "Nijjer",
                "given": "Gurp"
            }
        ],
        "URL": "https://omanscience.com/en/articles/system-level-optimization-beyond-cryptographic-kernels-an-ml-kem-case-study-on-arm-cortex-m7",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Recent work on embedded post-quantum cryptography has focused primarily on instruction-level optimization, including arithmetic-kernel improvements, assembly tuning, register allocation, and instruction scheduling. Using the Module-Lattice-Based Key-Encapsulation Mechanism (ML-KEM) on an Arm Cortex-M7 as a case study, we examine the additional gains available from memory-hierarchy utilization, tightly coupled memory placement, peripheral integration, clock configuration, and deterministic public-data reuse. The evaluation starts from a state-of-the-art SLOTHY-optimized implementation and covers all three ML-KEM parameter sets. Without modifying the cryptographic algorithm or standardized wire formats, the evaluated profiles without auxiliary public state reduce cycles by up to 2.5%. A selected public-data-reuse profile reduces encapsulation and decapsulation cycles by up to 74.6% and 58.8%, respectively. These results demonstrate that substantial deployment gains remain after arithmetic-kernel optimization and motivate a two-stage methodology that also examines the surrounding execution system."
    }
]