[
    {
        "id": "osp-20989",
        "type": "article-journal",
        "title": "16-bit Precision of Convolutional Neural Networks on Microcontroller Units for 8-bit Costs",
        "author": [
            {
                "family": "Liu",
                "given": "Rui"
            },
            {
                "family": "Paaßen",
                "given": "Benjamin"
            }
        ],
        "URL": "https://omanscience.com/en/articles/16-bit-precision-of-convolutional-neural-networks-on-microcontroller-units-for-8-bit-costs",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "To deploy deep neural networks on edge hardware, highly efficient inference schemes are necessary that retain high accuracy. This work presents W16A16, a high precision (16-bit), fast speed, low energy quantization method. On a widely applied microcontroller architecture Armv7E-M, our proposed approach achieves faster speed and lower energy consumption on layer- and model-level compared to alternative quantization schemes. We analyze the architecture of Armv7E-M, explain the underlying principles behind the performance advantages of 16-bit approaches, and evaluate the empiric quantization errors for regression and classification tasks, as well as empiric time- and energy consumption in MCU deployment. We observe ca.\\ 10 times lower quantization errors compared to 8-bit quantization schemes while achieving similar or better inference times and energy consumption."
    }
]