From 92106d2a4d6fa47094fe030209358b8a14c3b3b6 Mon Sep 17 00:00:00 2001 From: "J. Nick Koston" Date: Sun, 1 Mar 2026 22:53:27 -1000 Subject: [PATCH 1/2] [core] Eliminate __udivdi3 in millis() on ESP32 and RP2040 On 32-bit targets, GCC does not optimize 64-bit constant division into a multiply-by-reciprocal and instead calls __udivdi3 (software 64-bit divide). This replaces the division with Euclidean decomposition using a single 32-bit divide that GCC optimizes into a multiply-by-reciprocal. Benchmarked on ESP32 classic (Xtensa LX6, 240 MHz): __udivdi3: 628-705 ns fast_div1000_32: 269-279 ns (2.3-2.6x faster) millis() is called ~21 times per main loop iteration. At ~350 ns savings per call, this saves ~7 us per loop iteration. --- esphome/components/esp32/core.cpp | 2 +- esphome/components/rp2040/core.cpp | 2 +- esphome/core/helpers.h | 33 ++++++++++++++++++++++++++++++ 3 files changed, 35 insertions(+), 2 deletions(-) diff --git a/esphome/components/esp32/core.cpp b/esphome/components/esp32/core.cpp index b9ae871abf..0d7ab52281 100644 --- a/esphome/components/esp32/core.cpp +++ b/esphome/components/esp32/core.cpp @@ -22,7 +22,7 @@ extern "C" __attribute__((weak)) void initArduino() {} namespace esphome { void HOT yield() { vPortYield(); } -uint32_t IRAM_ATTR HOT millis() { return (uint32_t) (esp_timer_get_time() / 1000ULL); } +uint32_t IRAM_ATTR HOT millis() { return fast_div1000_32(static_cast(esp_timer_get_time())); } uint64_t HOT millis_64() { return static_cast(esp_timer_get_time()) / 1000ULL; } void HOT delay(uint32_t ms) { vTaskDelay(ms / portTICK_PERIOD_MS); } uint32_t IRAM_ATTR HOT micros() { return (uint32_t) esp_timer_get_time(); } diff --git a/esphome/components/rp2040/core.cpp b/esphome/components/rp2040/core.cpp index 8b86de4be1..2958d93a02 100644 --- a/esphome/components/rp2040/core.cpp +++ b/esphome/components/rp2040/core.cpp @@ -12,7 +12,7 @@ namespace esphome { void HOT yield() { ::yield(); } uint64_t millis_64() { return time_us_64() / 1000ULL; } -uint32_t HOT millis() { return static_cast(millis_64()); } +uint32_t HOT millis() { return fast_div1000_32(time_us_64()); } void HOT delay(uint32_t ms) { ::delay(ms); } uint32_t HOT micros() { return ::micros(); } void HOT delayMicroseconds(uint32_t us) { delay_microseconds_safe(us); } diff --git a/esphome/core/helpers.h b/esphome/core/helpers.h index b606e68df3..2218c17a49 100644 --- a/esphome/core/helpers.h +++ b/esphome/core/helpers.h @@ -527,6 +527,39 @@ template constexpr uint32_t fnv1a_hash_extend(uint32_t hash, T constexpr uint32_t fnv1a_hash(const char *str) { return fnv1a_hash_extend(FNV1_OFFSET_BASIS, str); } inline uint32_t fnv1a_hash(const std::string &str) { return fnv1a_hash(str.c_str()); } +/// Divide a uint64_t by 1000 and return the low 32 bits of the quotient. +/// +/// On 32-bit targets, GCC does not optimize 64-bit constant division into a +/// multiply-by-reciprocal and instead calls __udivdi3 (software 64-bit divide, +/// ~1200 ns on Xtensa @ 240 MHz). This uses the Euclidean division identity to +/// decompose the 64-bit division into a single 32-bit division: +/// +/// 2^32 = 4294967 * 1000 + 296 (Q * D + R) +/// +/// Therefore: (hi * 2^32 + lo) / 1000 = hi * Q + (hi * R + lo) / 1000 +/// +/// GCC optimizes the remaining 32-bit "/ 1000U" into a multiply-by-reciprocal +/// (mulhu + shift), so no division instruction is emitted. +/// +/// Only the low 32 bits of the quotient are returned (same 49.7-day wrap as millis()). +/// +/// See: https://en.wikipedia.org/wiki/Euclidean_division +/// See: https://ridiculousfish.com/blog/posts/labor-of-division-episode-iii.html +inline ESPHOME_ALWAYS_INLINE uint32_t fast_div1000_32(uint64_t us) { + static constexpr uint32_t D = 1000U; // divisor (microseconds per millisecond) + static constexpr uint32_t Q = 4294967U; // 2^32 / 1000 + static constexpr uint32_t R = 296U; // 2^32 % 1000 + uint32_t lo = static_cast(us); + uint32_t hi = static_cast(us >> 32); + // Combine remainder term: hi * (2^32 % 1000) + lo + uint32_t adj = hi * R + lo; + if (adj < lo) { + // Overflow: the true value is 2^32 + adj, so apply the identity again + return hi * Q + (adj + R) / D + Q; + } + return hi * Q + adj / D; +} + /// Return a random 32-bit unsigned integer. uint32_t random_uint32(); /// Return a random float between 0 and 1. From 4dbc139ede2a711410c4e23b19c7896361126bd2 Mon Sep 17 00:00:00 2001 From: "J. Nick Koston" Date: Sun, 1 Mar 2026 22:57:23 -1000 Subject: [PATCH 2/2] [core] Use ternary in fast_div1000_32 for clarity --- esphome/core/helpers.h | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/esphome/core/helpers.h b/esphome/core/helpers.h index 2218c17a49..8a55359840 100644 --- a/esphome/core/helpers.h +++ b/esphome/core/helpers.h @@ -553,11 +553,8 @@ inline ESPHOME_ALWAYS_INLINE uint32_t fast_div1000_32(uint64_t us) { uint32_t hi = static_cast(us >> 32); // Combine remainder term: hi * (2^32 % 1000) + lo uint32_t adj = hi * R + lo; - if (adj < lo) { - // Overflow: the true value is 2^32 + adj, so apply the identity again - return hi * Q + (adj + R) / D + Q; - } - return hi * Q + adj / D; + // If adj overflowed, the true value is 2^32 + adj; apply the identity again + return hi * Q + (adj < lo ? (adj + R) / D + Q : adj / D); } /// Return a random 32-bit unsigned integer.