From afba43160095d97b812770818138e65487029fef Mon Sep 17 00:00:00 2001 From: Remco van Essen Date: Thu, 1 Oct 2026 20:15:13 +0200 Subject: [PATCH] [resampler] Add resampler microphone platform (#19953) --- esphome/components/resampler/__init__.py | 15 ++ .../resampler/microphone/__init__.py | 77 ++++++++ .../microphone/resampler_microphone.cpp | 174 ++++++++++++++++++ .../microphone/resampler_microphone.h | 62 +++++++ .../components/resampler/speaker/__init__.py | 16 +- .../resampler/common-microphone.yaml | 29 +++ .../resampler/test-microphone.esp32-idf.yaml | 6 + .../test-microphone.esp32-s3-idf.yaml | 6 + 8 files changed, 372 insertions(+), 13 deletions(-) create mode 100644 esphome/components/resampler/microphone/__init__.py create mode 100644 esphome/components/resampler/microphone/resampler_microphone.cpp create mode 100644 esphome/components/resampler/microphone/resampler_microphone.h create mode 100644 tests/components/resampler/common-microphone.yaml create mode 100644 tests/components/resampler/test-microphone.esp32-idf.yaml create mode 100644 tests/components/resampler/test-microphone.esp32-s3-idf.yaml diff --git a/esphome/components/resampler/__init__.py b/esphome/components/resampler/__init__.py index e69de29bb2..b9b070e015 100644 --- a/esphome/components/resampler/__init__.py +++ b/esphome/components/resampler/__init__.py @@ -0,0 +1,15 @@ +from typing import Any + +import esphome.codegen as cg +import esphome.config_validation as cv + +resampler_ns = cg.esphome_ns.namespace("resampler") + +CONF_TAPS = "taps" + + +def validate_taps(taps: Any) -> int: + value = cv.int_range(min=16, max=128)(taps) + if value % 4 != 0: + raise cv.Invalid("Number of taps must be divisible by 4") + return value diff --git a/esphome/components/resampler/microphone/__init__.py b/esphome/components/resampler/microphone/__init__.py new file mode 100644 index 0000000000..0a8d0c1ca3 --- /dev/null +++ b/esphome/components/resampler/microphone/__init__.py @@ -0,0 +1,77 @@ +import esphome.codegen as cg +from esphome.components import audio, microphone +import esphome.config_validation as cv +from esphome.const import ( + CONF_BITS_PER_SAMPLE, + CONF_CHANNELS, + CONF_FILTERS, + CONF_ID, + CONF_MICROPHONE, + CONF_SAMPLE_RATE, + PLATFORM_ESP32, +) +from esphome.types import ConfigType + +from .. import CONF_TAPS, resampler_ns, validate_taps + +AUTO_LOAD = ["audio"] +DEPENDENCIES = ["microphone"] + +ResamplerMicrophone = resampler_ns.class_( + "ResamplerMicrophone", cg.Component, microphone.Microphone +) + + +def _set_stream_limits(config: ConfigType) -> ConfigType: + # Only the sample rate changes; the bits and channels are those selected from the source microphone + source = config[CONF_MICROPHONE] + audio.set_stream_limits( + min_bits_per_sample=source[CONF_BITS_PER_SAMPLE], + max_bits_per_sample=source[CONF_BITS_PER_SAMPLE], + min_channels=len(source[CONF_CHANNELS]), + max_channels=len(source[CONF_CHANNELS]), + min_sample_rate=config[CONF_SAMPLE_RATE], + max_sample_rate=config[CONF_SAMPLE_RATE], + )(config) + return config + + +CONFIG_SCHEMA = cv.All( + microphone.MICROPHONE_SCHEMA.extend( + { + cv.GenerateID(): cv.declare_id(ResamplerMicrophone), + cv.Required(CONF_MICROPHONE): microphone.microphone_source_schema( + min_bits_per_sample=16, + max_bits_per_sample=32, + min_channels=1, + max_channels=2, + ), + cv.Optional(CONF_SAMPLE_RATE, default=16000): cv.int_range(8000, 48000), + cv.Optional(CONF_FILTERS, default=16): cv.int_range(min=2, max=1024), + cv.Optional(CONF_TAPS, default=16): validate_taps, + } + ).extend(cv.COMPONENT_SCHEMA), + cv.only_on([PLATFORM_ESP32]), + _set_stream_limits, +) + + +FINAL_VALIDATE_SCHEMA = cv.Schema( + { + cv.Required( + CONF_MICROPHONE + ): microphone.final_validate_microphone_source_schema("resampler"), + }, + extra=cv.ALLOW_EXTRA, +) + + +async def to_code(config: ConfigType) -> None: + mic_source = await microphone.microphone_source_to_code(config[CONF_MICROPHONE]) + var = cg.new_Pvariable(config[CONF_ID], mic_source) + await cg.register_component(var, config) + await microphone.register_microphone(var, config) + + cg.add(var.set_target_sample_rate(config[CONF_SAMPLE_RATE])) + cg.add(var.set_filters(config[CONF_FILTERS])) + cg.add(var.set_taps(config[CONF_TAPS])) diff --git a/esphome/components/resampler/microphone/resampler_microphone.cpp b/esphome/components/resampler/microphone/resampler_microphone.cpp new file mode 100644 index 0000000000..0dcd523380 --- /dev/null +++ b/esphome/components/resampler/microphone/resampler_microphone.cpp @@ -0,0 +1,174 @@ +#include "resampler_microphone.h" + +#ifdef USE_ESP32 + +#include "esphome/core/helpers.h" +#include "esphome/core/log.h" + +#include + +namespace esphome::resampler { + +static const char *const TAG = "resampler.microphone"; + +// Duration of audio the resampler converts per step; longer source chunks are processed in several steps +static constexpr uint32_t BUFFER_DURATION_MS = 16; + +void ResamplerMicrophone::setup() { + const audio::AudioStreamInfo input_stream_info = this->source_->get_audio_stream_info(); + this->audio_stream_info_ = audio::AudioStreamInfo(input_stream_info.get_bits_per_sample(), + input_stream_info.get_channels(), this->target_sample_rate_); + + // Allocate now for the expected source format; process_audio_ only sets up again if that format changes + if (!this->init_resampler_(input_stream_info)) { + this->mark_failed(); + return; + } + + this->source_->add_data_callback([this](const std::vector &data) { this->process_audio_(data); }); + + this->disable_loop(); +} + +void ResamplerMicrophone::dump_config() { + ESP_LOGCONFIG(TAG, + "Resampler Microphone:\n" + " Target Sample Rate: %" PRIu32 " Hz\n" + " Taps: %u\n" + " Filters: %u", + this->target_sample_rate_, this->taps_, this->filters_); +} + +void ResamplerMicrophone::start() { + if (this->is_failed() || this->active_listeners_ == UINT8_MAX) + return; + ++this->active_listeners_; + this->enable_loop(); +} + +void ResamplerMicrophone::stop() { + if (this->active_listeners_ == 0) + return; + --this->active_listeners_; + this->enable_loop(); +} + +void ResamplerMicrophone::loop() { + if (this->active_listeners_ == 0) { + if (this->state_ != microphone::STATE_STOPPED) { + this->source_->stop(); + this->state_ = microphone::STATE_STOPPED; + } + this->disable_loop(); + return; + } + + switch (this->state_) { + case microphone::STATE_STOPPED: + this->source_->start(); + this->state_ = microphone::STATE_STARTING; + break; + case microphone::STATE_STARTING: + if (this->source_->is_running()) { + this->state_ = microphone::STATE_RUNNING; + } + break; + case microphone::STATE_RUNNING: + // Follow the source if it restarts, e.g. after a driver error + if (!this->source_->is_running()) { + this->state_ = microphone::STATE_STARTING; + } + break; + case microphone::STATE_STOPPING: + break; + } +} + +bool ResamplerMicrophone::init_resampler_(const audio::AudioStreamInfo &input_stream_info) { + this->resampler_.reset(); + this->resampler_ready_ = false; + + if (input_stream_info.get_sample_rate() == this->target_sample_rate_) { + // The source already delivers the target sample rate, so its audio is passed through unchanged + this->input_stream_info_ = input_stream_info; + this->resampler_ready_ = true; + return true; + } + + const audio::AudioStreamInfo output_stream_info(input_stream_info.get_bits_per_sample(), + input_stream_info.get_channels(), this->target_sample_rate_); + + auto resampler = make_unique( + input_stream_info.ms_to_samples(BUFFER_DURATION_MS), output_stream_info.ms_to_samples(BUFFER_DURATION_MS)); + + esp_audio_libs::resampler::ResamplerConfiguration resample_config = { + .source_sample_rate = static_cast(input_stream_info.get_sample_rate()), + .target_sample_rate = static_cast(this->target_sample_rate_), + .source_bits_per_sample = input_stream_info.get_bits_per_sample(), + .target_bits_per_sample = input_stream_info.get_bits_per_sample(), + .channels = input_stream_info.get_channels(), + // Filters out frequencies above the new Nyquist limit when downsampling, to avoid aliasing + .use_pre_or_post_filter = this->target_sample_rate_ < input_stream_info.get_sample_rate(), + .subsample_interpolate = false, // Doubles the CPU load; more filters is a better alternative + .number_of_taps = this->taps_, + .number_of_filters = this->filters_, + }; + + if (!resampler->initialize(resample_config)) { + ESP_LOGE(TAG, "Not enough memory to resample"); + return false; + } + + this->output_buffer_.reserve(output_stream_info.ms_to_bytes(BUFFER_DURATION_MS)); + this->resampler_ = std::move(resampler); + // Only set on success, so a failed set up is retried with the next chunk + this->input_stream_info_ = input_stream_info; + this->resampler_ready_ = true; + return true; +} + +void ResamplerMicrophone::process_audio_(const std::vector &data) { + const audio::AudioStreamInfo input_stream_info = this->source_->get_audio_stream_info(); + if (input_stream_info != this->input_stream_info_) { + this->init_resampler_(input_stream_info); + } + if (!this->resampler_ready_) { + return; + } + + if (this->resampler_ == nullptr) { + this->data_callbacks_.call(data); + return; + } + + const size_t input_bytes_per_frame = input_stream_info.frames_to_bytes(1); + const uint32_t max_input_frames = input_stream_info.ms_to_frames(BUFFER_DURATION_MS); + // Both limits match the sizes the resampler's internal buffers were allocated with in init_resampler_ + const uint32_t max_output_frames = this->audio_stream_info_.ms_to_frames(BUFFER_DURATION_MS); + + const uint8_t *input = data.data(); + uint32_t input_frames = input_stream_info.bytes_to_frames(data.size()); + while (input_frames > 0) { + // Stays within the reserved capacity, so this never reallocates + this->output_buffer_.resize(this->audio_stream_info_.frames_to_bytes(max_output_frames)); + + // The resampler's internal buffers hold at most BUFFER_DURATION_MS of audio, so feed it in steps of that size. + // 0 dB keeps the microphone level that downstream detectors are tuned for; overshoot saturates instead of wrapping. + esp_audio_libs::resampler::ResamplerResults results = this->resampler_->resample( + input, this->output_buffer_.data(), std::min(input_frames, max_input_frames), max_output_frames, 0.0f); + + input += results.frames_used * input_bytes_per_frame; + input_frames -= results.frames_used; + + if (results.frames_generated > 0) { + this->output_buffer_.resize(this->audio_stream_info_.frames_to_bytes(results.frames_generated)); + this->data_callbacks_.call(this->output_buffer_); + } else if (results.frames_used == 0) { + break; // No progress; drop the rest of the chunk instead of spinning + } + } +} + +} // namespace esphome::resampler + +#endif // USE_ESP32 diff --git a/esphome/components/resampler/microphone/resampler_microphone.h b/esphome/components/resampler/microphone/resampler_microphone.h new file mode 100644 index 0000000000..780f22e429 --- /dev/null +++ b/esphome/components/resampler/microphone/resampler_microphone.h @@ -0,0 +1,62 @@ +#pragma once + +#ifdef USE_ESP32 + +#include "esphome/components/audio/audio.h" +#include "esphome/components/microphone/microphone.h" +#include "esphome/components/microphone/microphone_source.h" + +#include "esphome/core/component.h" + +#include // esp-audio-libs + +#include +#include + +namespace esphome::resampler { + +/// @brief Microphone that converts the audio of a source microphone to a different sample rate. +/// The bits per sample and channels are selected by the source's ``MicrophoneSource``; only the sample rate changes. +/// Resampling runs in the source microphone's data callback, so it needs no task or ring buffer of its own. +class ResamplerMicrophone final : public Component, public microphone::Microphone { + public: + explicit ResamplerMicrophone(microphone::MicrophoneSource *source) : source_(source) {} + + void setup() override; + void loop() override; + void dump_config() override; + + void start() override; + void stop() override; + + void set_target_sample_rate(uint32_t target_sample_rate) { this->target_sample_rate_ = target_sample_rate; } + void set_filters(uint16_t filters) { this->filters_ = filters; } + void set_taps(uint16_t taps) { this->taps_ = taps; } + + protected: + /// @brief Sets up the resampler for the given input format. No resampler is needed if the sample rates match. + /// @return false if the resampler failed to allocate; the audio is then dropped + bool init_resampler_(const audio::AudioStreamInfo &input_stream_info); + + /// @brief Resamples a chunk of source audio and passes it to the data callbacks. Source microphone task only. + void process_audio_(const std::vector &data); + + microphone::MicrophoneSource *source_; + std::unique_ptr resampler_; + // Reused for every chunk so resampling does not allocate + std::vector output_buffer_; + + // Format the resampler is set up for + audio::AudioStreamInfo input_stream_info_; + + uint32_t target_sample_rate_; + uint16_t taps_; + uint16_t filters_; + + uint8_t active_listeners_{0}; + bool resampler_ready_{false}; +}; + +} // namespace esphome::resampler + +#endif // USE_ESP32 diff --git a/esphome/components/resampler/speaker/__init__.py b/esphome/components/resampler/speaker/__init__.py index 7de468cb50..8fa8aeb61c 100644 --- a/esphome/components/resampler/speaker/__init__.py +++ b/esphome/components/resampler/speaker/__init__.py @@ -1,5 +1,3 @@ -from typing import Any - import esphome.codegen as cg from esphome.components import audio, psram, speaker import esphome.config_validation as cv @@ -17,16 +15,15 @@ from esphome.const import ( from esphome.core.entity_helpers import inherit_property_from from esphome.types import ConfigType +from .. import CONF_TAPS, resampler_ns, validate_taps + AUTO_LOAD = ["audio"] CODEOWNERS = ["@kahrendt"] -resampler_ns = cg.esphome_ns.namespace("resampler") ResamplerSpeaker = resampler_ns.class_( "ResamplerSpeaker", cg.Component, speaker.Speaker ) -CONF_TAPS = "taps" - PASSTHROUGH = "passthrough" @@ -60,13 +57,6 @@ def _validate_audio_compatibility(config: ConfigType) -> None: )(config) -def _validate_taps(taps: Any) -> int: - value = cv.int_range(min=16, max=128)(taps) - if value % 4 != 0: - raise cv.Invalid("Number of taps must be divisible by 4") - return value - - CONFIG_SCHEMA = cv.All( speaker.SPEAKER_SCHEMA.extend( { @@ -80,7 +70,7 @@ CONFIG_SCHEMA = cv.All( ): cv.positive_time_period_milliseconds, cv.Optional(CONF_TASK_STACK_IN_PSRAM): psram.validate_task_stack_in_psram, cv.Optional(CONF_FILTERS, default=16): cv.int_range(min=2, max=1024), - cv.Optional(CONF_TAPS, default=16): _validate_taps, + cv.Optional(CONF_TAPS, default=16): validate_taps, } ).extend(cv.COMPONENT_SCHEMA), cv.only_on([PLATFORM_ESP32]), diff --git a/tests/components/resampler/common-microphone.yaml b/tests/components/resampler/common-microphone.yaml new file mode 100644 index 0000000000..123ba87edc --- /dev/null +++ b/tests/components/resampler/common-microphone.yaml @@ -0,0 +1,29 @@ +microphone: + - platform: i2s_audio + id: resampler_i2s_mic_id + i2s_audio_id: i2s_audio_bus + adc_type: external + i2s_din_pin: ${din_pin} + sample_rate: 48000 + bits_per_sample: 32bit + channel: stereo + - platform: resampler + id: resampler_mic_id + microphone: + microphone: resampler_i2s_mic_id + channels: 0 + bits_per_sample: 16 + sample_rate: 16000 + on_data: + - logger.log: + format: "Received %u bytes" + args: [x.size()] + - platform: resampler + id: resampler_mic_stereo_id + microphone: + microphone: resampler_i2s_mic_id + channels: [0, 1] + bits_per_sample: 32 + sample_rate: 44100 + filters: 32 + taps: 32 diff --git a/tests/components/resampler/test-microphone.esp32-idf.yaml b/tests/components/resampler/test-microphone.esp32-idf.yaml new file mode 100644 index 0000000000..9ca0ce7f05 --- /dev/null +++ b/tests/components/resampler/test-microphone.esp32-idf.yaml @@ -0,0 +1,6 @@ +substitutions: + din_pin: GPIO21 + +packages: + i2s_audio: !include ../../test_build_components/common/i2s_audio/esp32-idf.yaml + resampler: !include common-microphone.yaml diff --git a/tests/components/resampler/test-microphone.esp32-s3-idf.yaml b/tests/components/resampler/test-microphone.esp32-s3-idf.yaml new file mode 100644 index 0000000000..5ff29672a0 --- /dev/null +++ b/tests/components/resampler/test-microphone.esp32-s3-idf.yaml @@ -0,0 +1,6 @@ +substitutions: + din_pin: GPIO16 + +packages: + i2s_audio: !include ../../test_build_components/common/i2s_audio/esp32-s3-idf.yaml + resampler: !include common-microphone.yaml