@article{2026_benchmarking_npus, title = {Benchmarking {Ultra}-{Low}-{Power} -{NPUs}}, volume = {30}, issn = {2375-0529, 2375-0537}, url = {https://dl.acm.org/doi/10.1145/3833428.3833430}, doi = {10.1145/3833428.3833430}, abstract = {Microcontrollers (MCUs) are widely used in resource-constrained environments due to their form factor and low cost and power consumption. Performing model inference on MCU-scale hardware improves privacy by running locally, lowers operating costs for model vendors, and eliminates dependence on network connectivity [1, 2]. However, deployments remain constrained by limited memory, throughput, and compute. The growing computational demands of modern models have catalyzed specialized accelerators across the computing spectrum, from data centers to embedded systems. At this resource-constrained end, MCU-scale neural processing units, or ?NPUs, have emerged to provide real-time or near-real-time inference within milliwatt-scale power budgets.}, language = {en}, number = {2}, urldate = {2026-08-05}, journal = {GetMobile: Mobile Computing and Communications}, author = {Millar, Josh and Huang, Yushan and Sethi, Sarab and Haddadi, Hamed and Madhavapeddy, Anil}, month = jul, year = {2026}, pages = {11--15}, }