[resampler] Add resampler microphone platform (#19953)

This commit is contained in:
Remco van Essen
2026-10-01 13:15:13 -05:00
committed by GitHub
parent 5e67258818
commit afba431600
8 changed files with 372 additions and 13 deletions
+15
View File
@@ -0,0 +1,15 @@
from typing import Any
import esphome.codegen as cg
import esphome.config_validation as cv
resampler_ns = cg.esphome_ns.namespace("resampler")
CONF_TAPS = "taps"
def validate_taps(taps: Any) -> int:
value = cv.int_range(min=16, max=128)(taps)
if value % 4 != 0:
raise cv.Invalid("Number of taps must be divisible by 4")
return value
@@ -0,0 +1,77 @@
import esphome.codegen as cg
from esphome.components import audio, microphone
import esphome.config_validation as cv
from esphome.const import (
CONF_BITS_PER_SAMPLE,
CONF_CHANNELS,
CONF_FILTERS,
CONF_ID,
CONF_MICROPHONE,
CONF_SAMPLE_RATE,
PLATFORM_ESP32,
)
from esphome.types import ConfigType
from .. import CONF_TAPS, resampler_ns, validate_taps
AUTO_LOAD = ["audio"]
DEPENDENCIES = ["microphone"]
ResamplerMicrophone = resampler_ns.class_(
"ResamplerMicrophone", cg.Component, microphone.Microphone
)
def _set_stream_limits(config: ConfigType) -> ConfigType:
# Only the sample rate changes; the bits and channels are those selected from the source microphone
source = config[CONF_MICROPHONE]
audio.set_stream_limits(
min_bits_per_sample=source[CONF_BITS_PER_SAMPLE],
max_bits_per_sample=source[CONF_BITS_PER_SAMPLE],
min_channels=len(source[CONF_CHANNELS]),
max_channels=len(source[CONF_CHANNELS]),
min_sample_rate=config[CONF_SAMPLE_RATE],
max_sample_rate=config[CONF_SAMPLE_RATE],
)(config)
return config
CONFIG_SCHEMA = cv.All(
microphone.MICROPHONE_SCHEMA.extend(
{
cv.GenerateID(): cv.declare_id(ResamplerMicrophone),
cv.Required(CONF_MICROPHONE): microphone.microphone_source_schema(
min_bits_per_sample=16,
max_bits_per_sample=32,
min_channels=1,
max_channels=2,
),
cv.Optional(CONF_SAMPLE_RATE, default=16000): cv.int_range(8000, 48000),
cv.Optional(CONF_FILTERS, default=16): cv.int_range(min=2, max=1024),
cv.Optional(CONF_TAPS, default=16): validate_taps,
}
).extend(cv.COMPONENT_SCHEMA),
cv.only_on([PLATFORM_ESP32]),
_set_stream_limits,
)
FINAL_VALIDATE_SCHEMA = cv.Schema(
{
cv.Required(
CONF_MICROPHONE
): microphone.final_validate_microphone_source_schema("resampler"),
},
extra=cv.ALLOW_EXTRA,
)
async def to_code(config: ConfigType) -> None:
mic_source = await microphone.microphone_source_to_code(config[CONF_MICROPHONE])
var = cg.new_Pvariable(config[CONF_ID], mic_source)
await cg.register_component(var, config)
await microphone.register_microphone(var, config)
cg.add(var.set_target_sample_rate(config[CONF_SAMPLE_RATE]))
cg.add(var.set_filters(config[CONF_FILTERS]))
cg.add(var.set_taps(config[CONF_TAPS]))
@@ -0,0 +1,174 @@
#include "resampler_microphone.h"
#ifdef USE_ESP32
#include "esphome/core/helpers.h"
#include "esphome/core/log.h"
#include <algorithm>
namespace esphome::resampler {
static const char *const TAG = "resampler.microphone";
// Duration of audio the resampler converts per step; longer source chunks are processed in several steps
static constexpr uint32_t BUFFER_DURATION_MS = 16;
void ResamplerMicrophone::setup() {
const audio::AudioStreamInfo input_stream_info = this->source_->get_audio_stream_info();
this->audio_stream_info_ = audio::AudioStreamInfo(input_stream_info.get_bits_per_sample(),
input_stream_info.get_channels(), this->target_sample_rate_);
// Allocate now for the expected source format; process_audio_ only sets up again if that format changes
if (!this->init_resampler_(input_stream_info)) {
this->mark_failed();
return;
}
this->source_->add_data_callback([this](const std::vector<uint8_t> &data) { this->process_audio_(data); });
this->disable_loop();
}
void ResamplerMicrophone::dump_config() {
ESP_LOGCONFIG(TAG,
"Resampler Microphone:\n"
" Target Sample Rate: %" PRIu32 " Hz\n"
" Taps: %u\n"
" Filters: %u",
this->target_sample_rate_, this->taps_, this->filters_);
}
void ResamplerMicrophone::start() {
if (this->is_failed() || this->active_listeners_ == UINT8_MAX)
return;
++this->active_listeners_;
this->enable_loop();
}
void ResamplerMicrophone::stop() {
if (this->active_listeners_ == 0)
return;
--this->active_listeners_;
this->enable_loop();
}
void ResamplerMicrophone::loop() {
if (this->active_listeners_ == 0) {
if (this->state_ != microphone::STATE_STOPPED) {
this->source_->stop();
this->state_ = microphone::STATE_STOPPED;
}
this->disable_loop();
return;
}
switch (this->state_) {
case microphone::STATE_STOPPED:
this->source_->start();
this->state_ = microphone::STATE_STARTING;
break;
case microphone::STATE_STARTING:
if (this->source_->is_running()) {
this->state_ = microphone::STATE_RUNNING;
}
break;
case microphone::STATE_RUNNING:
// Follow the source if it restarts, e.g. after a driver error
if (!this->source_->is_running()) {
this->state_ = microphone::STATE_STARTING;
}
break;
case microphone::STATE_STOPPING:
break;
}
}
bool ResamplerMicrophone::init_resampler_(const audio::AudioStreamInfo &input_stream_info) {
this->resampler_.reset();
this->resampler_ready_ = false;
if (input_stream_info.get_sample_rate() == this->target_sample_rate_) {
// The source already delivers the target sample rate, so its audio is passed through unchanged
this->input_stream_info_ = input_stream_info;
this->resampler_ready_ = true;
return true;
}
const audio::AudioStreamInfo output_stream_info(input_stream_info.get_bits_per_sample(),
input_stream_info.get_channels(), this->target_sample_rate_);
auto resampler = make_unique<esp_audio_libs::resampler::Resampler>(
input_stream_info.ms_to_samples(BUFFER_DURATION_MS), output_stream_info.ms_to_samples(BUFFER_DURATION_MS));
esp_audio_libs::resampler::ResamplerConfiguration resample_config = {
.source_sample_rate = static_cast<float>(input_stream_info.get_sample_rate()),
.target_sample_rate = static_cast<float>(this->target_sample_rate_),
.source_bits_per_sample = input_stream_info.get_bits_per_sample(),
.target_bits_per_sample = input_stream_info.get_bits_per_sample(),
.channels = input_stream_info.get_channels(),
// Filters out frequencies above the new Nyquist limit when downsampling, to avoid aliasing
.use_pre_or_post_filter = this->target_sample_rate_ < input_stream_info.get_sample_rate(),
.subsample_interpolate = false, // Doubles the CPU load; more filters is a better alternative
.number_of_taps = this->taps_,
.number_of_filters = this->filters_,
};
if (!resampler->initialize(resample_config)) {
ESP_LOGE(TAG, "Not enough memory to resample");
return false;
}
this->output_buffer_.reserve(output_stream_info.ms_to_bytes(BUFFER_DURATION_MS));
this->resampler_ = std::move(resampler);
// Only set on success, so a failed set up is retried with the next chunk
this->input_stream_info_ = input_stream_info;
this->resampler_ready_ = true;
return true;
}
void ResamplerMicrophone::process_audio_(const std::vector<uint8_t> &data) {
const audio::AudioStreamInfo input_stream_info = this->source_->get_audio_stream_info();
if (input_stream_info != this->input_stream_info_) {
this->init_resampler_(input_stream_info);
}
if (!this->resampler_ready_) {
return;
}
if (this->resampler_ == nullptr) {
this->data_callbacks_.call(data);
return;
}
const size_t input_bytes_per_frame = input_stream_info.frames_to_bytes(1);
const uint32_t max_input_frames = input_stream_info.ms_to_frames(BUFFER_DURATION_MS);
// Both limits match the sizes the resampler's internal buffers were allocated with in init_resampler_
const uint32_t max_output_frames = this->audio_stream_info_.ms_to_frames(BUFFER_DURATION_MS);
const uint8_t *input = data.data();
uint32_t input_frames = input_stream_info.bytes_to_frames(data.size());
while (input_frames > 0) {
// Stays within the reserved capacity, so this never reallocates
this->output_buffer_.resize(this->audio_stream_info_.frames_to_bytes(max_output_frames));
// The resampler's internal buffers hold at most BUFFER_DURATION_MS of audio, so feed it in steps of that size.
// 0 dB keeps the microphone level that downstream detectors are tuned for; overshoot saturates instead of wrapping.
esp_audio_libs::resampler::ResamplerResults results = this->resampler_->resample(
input, this->output_buffer_.data(), std::min(input_frames, max_input_frames), max_output_frames, 0.0f);
input += results.frames_used * input_bytes_per_frame;
input_frames -= results.frames_used;
if (results.frames_generated > 0) {
this->output_buffer_.resize(this->audio_stream_info_.frames_to_bytes(results.frames_generated));
this->data_callbacks_.call(this->output_buffer_);
} else if (results.frames_used == 0) {
break; // No progress; drop the rest of the chunk instead of spinning
}
}
}
} // namespace esphome::resampler
#endif // USE_ESP32
@@ -0,0 +1,62 @@
#pragma once
#ifdef USE_ESP32
#include "esphome/components/audio/audio.h"
#include "esphome/components/microphone/microphone.h"
#include "esphome/components/microphone/microphone_source.h"
#include "esphome/core/component.h"
#include <resampler.h> // esp-audio-libs
#include <memory>
#include <vector>
namespace esphome::resampler {
/// @brief Microphone that converts the audio of a source microphone to a different sample rate.
/// The bits per sample and channels are selected by the source's ``MicrophoneSource``; only the sample rate changes.
/// Resampling runs in the source microphone's data callback, so it needs no task or ring buffer of its own.
class ResamplerMicrophone final : public Component, public microphone::Microphone {
public:
explicit ResamplerMicrophone(microphone::MicrophoneSource *source) : source_(source) {}
void setup() override;
void loop() override;
void dump_config() override;
void start() override;
void stop() override;
void set_target_sample_rate(uint32_t target_sample_rate) { this->target_sample_rate_ = target_sample_rate; }
void set_filters(uint16_t filters) { this->filters_ = filters; }
void set_taps(uint16_t taps) { this->taps_ = taps; }
protected:
/// @brief Sets up the resampler for the given input format. No resampler is needed if the sample rates match.
/// @return false if the resampler failed to allocate; the audio is then dropped
bool init_resampler_(const audio::AudioStreamInfo &input_stream_info);
/// @brief Resamples a chunk of source audio and passes it to the data callbacks. Source microphone task only.
void process_audio_(const std::vector<uint8_t> &data);
microphone::MicrophoneSource *source_;
std::unique_ptr<esp_audio_libs::resampler::Resampler> resampler_;
// Reused for every chunk so resampling does not allocate
std::vector<uint8_t> output_buffer_;
// Format the resampler is set up for
audio::AudioStreamInfo input_stream_info_;
uint32_t target_sample_rate_;
uint16_t taps_;
uint16_t filters_;
uint8_t active_listeners_{0};
bool resampler_ready_{false};
};
} // namespace esphome::resampler
#endif // USE_ESP32
@@ -1,5 +1,3 @@
from typing import Any
import esphome.codegen as cg
from esphome.components import audio, psram, speaker
import esphome.config_validation as cv
@@ -17,16 +15,15 @@ from esphome.const import (
from esphome.core.entity_helpers import inherit_property_from
from esphome.types import ConfigType
from .. import CONF_TAPS, resampler_ns, validate_taps
AUTO_LOAD = ["audio"]
CODEOWNERS = ["@kahrendt"]
resampler_ns = cg.esphome_ns.namespace("resampler")
ResamplerSpeaker = resampler_ns.class_(
"ResamplerSpeaker", cg.Component, speaker.Speaker
)
CONF_TAPS = "taps"
PASSTHROUGH = "passthrough"
@@ -60,13 +57,6 @@ def _validate_audio_compatibility(config: ConfigType) -> None:
)(config)
def _validate_taps(taps: Any) -> int:
value = cv.int_range(min=16, max=128)(taps)
if value % 4 != 0:
raise cv.Invalid("Number of taps must be divisible by 4")
return value
CONFIG_SCHEMA = cv.All(
speaker.SPEAKER_SCHEMA.extend(
{
@@ -80,7 +70,7 @@ CONFIG_SCHEMA = cv.All(
): cv.positive_time_period_milliseconds,
cv.Optional(CONF_TASK_STACK_IN_PSRAM): psram.validate_task_stack_in_psram,
cv.Optional(CONF_FILTERS, default=16): cv.int_range(min=2, max=1024),
cv.Optional(CONF_TAPS, default=16): _validate_taps,
cv.Optional(CONF_TAPS, default=16): validate_taps,
}
).extend(cv.COMPONENT_SCHEMA),
cv.only_on([PLATFORM_ESP32]),
@@ -0,0 +1,29 @@
microphone:
- platform: i2s_audio
id: resampler_i2s_mic_id
i2s_audio_id: i2s_audio_bus
adc_type: external
i2s_din_pin: ${din_pin}
sample_rate: 48000
bits_per_sample: 32bit
channel: stereo
- platform: resampler
id: resampler_mic_id
microphone:
microphone: resampler_i2s_mic_id
channels: 0
bits_per_sample: 16
sample_rate: 16000
on_data:
- logger.log:
format: "Received %u bytes"
args: [x.size()]
- platform: resampler
id: resampler_mic_stereo_id
microphone:
microphone: resampler_i2s_mic_id
channels: [0, 1]
bits_per_sample: 32
sample_rate: 44100
filters: 32
taps: 32
@@ -0,0 +1,6 @@
substitutions:
din_pin: GPIO21
packages:
i2s_audio: !include ../../test_build_components/common/i2s_audio/esp32-idf.yaml
resampler: !include common-microphone.yaml
@@ -0,0 +1,6 @@
substitutions:
din_pin: GPIO16
packages:
i2s_audio: !include ../../test_build_components/common/i2s_audio/esp32-s3-idf.yaml
resampler: !include common-microphone.yaml