diff --git a/esphome/components/esp32/core.cpp b/esphome/components/esp32/core.cpp index a8f585aaf8..f61160ae8b 100644 --- a/esphome/components/esp32/core.cpp +++ b/esphome/components/esp32/core.cpp @@ -22,7 +22,7 @@ extern "C" __attribute__((weak)) void initArduino() {} namespace esphome { void HOT yield() { vPortYield(); } -uint32_t IRAM_ATTR HOT millis() { return (uint32_t) (esp_timer_get_time() / 1000ULL); } +uint32_t IRAM_ATTR HOT millis() { return fast_div1000_32(static_cast(esp_timer_get_time())); } uint64_t HOT millis_64() { return static_cast(esp_timer_get_time()) / 1000ULL; } void HOT delay(uint32_t ms) { vTaskDelay(ms / portTICK_PERIOD_MS); } uint32_t IRAM_ATTR HOT micros() { return (uint32_t) esp_timer_get_time(); } diff --git a/esphome/components/rp2040/core.cpp b/esphome/components/rp2040/core.cpp index 7e2894d532..3237b2f21c 100644 --- a/esphome/components/rp2040/core.cpp +++ b/esphome/components/rp2040/core.cpp @@ -12,7 +12,7 @@ namespace esphome { void HOT yield() { ::yield(); } uint64_t millis_64() { return time_us_64() / 1000ULL; } -uint32_t HOT millis() { return static_cast(millis_64()); } +uint32_t HOT millis() { return fast_div1000_32(time_us_64()); } void HOT delay(uint32_t ms) { ::delay(ms); } uint32_t HOT micros() { return ::micros(); } void HOT delayMicroseconds(uint32_t us) { delay_microseconds_safe(us); } diff --git a/esphome/core/helpers.h b/esphome/core/helpers.h index 64dc108d4b..a5929760a5 100644 --- a/esphome/core/helpers.h +++ b/esphome/core/helpers.h @@ -602,6 +602,36 @@ template constexpr uint32_t fnv1a_hash_extend(uint32_t hash, T constexpr uint32_t fnv1a_hash(const char *str) { return fnv1a_hash_extend(FNV1_OFFSET_BASIS, str); } inline uint32_t fnv1a_hash(const std::string &str) { return fnv1a_hash(str.c_str()); } +/// Divide a uint64_t by 1000 and return the low 32 bits of the quotient. +/// +/// On 32-bit targets, GCC does not optimize 64-bit constant division into a +/// multiply-by-reciprocal and instead calls __udivdi3 (software 64-bit divide, +/// ~1200 ns on Xtensa @ 240 MHz). This uses the Euclidean division identity to +/// decompose the 64-bit division into a single 32-bit division: +/// +/// 2^32 = 4294967 * 1000 + 296 (Q * D + R) +/// +/// Therefore: (hi * 2^32 + lo) / 1000 = hi * Q + (hi * R + lo) / 1000 +/// +/// GCC optimizes the remaining 32-bit "/ 1000U" into a multiply-by-reciprocal +/// (mulhu + shift), so no division instruction is emitted. +/// +/// Only the low 32 bits of the quotient are returned (same 49.7-day wrap as millis()). +/// +/// See: https://en.wikipedia.org/wiki/Euclidean_division +/// See: https://ridiculousfish.com/blog/posts/labor-of-division-episode-iii.html +inline ESPHOME_ALWAYS_INLINE uint32_t fast_div1000_32(uint64_t us) { + static constexpr uint32_t D = 1000U; // divisor (microseconds per millisecond) + static constexpr uint32_t Q = 4294967U; // 2^32 / 1000 + static constexpr uint32_t R = 296U; // 2^32 % 1000 + uint32_t lo = static_cast(us); + uint32_t hi = static_cast(us >> 32); + // Combine remainder term: hi * (2^32 % 1000) + lo + uint32_t adj = hi * R + lo; + // If adj overflowed, the true value is 2^32 + adj; apply the identity again + return hi * Q + (adj < lo ? (adj + R) / D + Q : adj / D); +} + /// Return a random 32-bit unsigned integer. uint32_t random_uint32(); /// Return a random float between 0 and 1.