mirror of
https://github.com/esphome/esphome.git
synced 2026-10-04 10:09:13 +00:00
[resampler] Add resampler microphone platform (#19953)
This commit is contained in:
@@ -0,0 +1,15 @@
|
||||
from typing import Any
|
||||
|
||||
import esphome.codegen as cg
|
||||
import esphome.config_validation as cv
|
||||
|
||||
resampler_ns = cg.esphome_ns.namespace("resampler")
|
||||
|
||||
CONF_TAPS = "taps"
|
||||
|
||||
|
||||
def validate_taps(taps: Any) -> int:
|
||||
value = cv.int_range(min=16, max=128)(taps)
|
||||
if value % 4 != 0:
|
||||
raise cv.Invalid("Number of taps must be divisible by 4")
|
||||
return value
|
||||
|
||||
@@ -0,0 +1,77 @@
|
||||
import esphome.codegen as cg
|
||||
from esphome.components import audio, microphone
|
||||
import esphome.config_validation as cv
|
||||
from esphome.const import (
|
||||
CONF_BITS_PER_SAMPLE,
|
||||
CONF_CHANNELS,
|
||||
CONF_FILTERS,
|
||||
CONF_ID,
|
||||
CONF_MICROPHONE,
|
||||
CONF_SAMPLE_RATE,
|
||||
PLATFORM_ESP32,
|
||||
)
|
||||
from esphome.types import ConfigType
|
||||
|
||||
from .. import CONF_TAPS, resampler_ns, validate_taps
|
||||
|
||||
AUTO_LOAD = ["audio"]
|
||||
DEPENDENCIES = ["microphone"]
|
||||
|
||||
ResamplerMicrophone = resampler_ns.class_(
|
||||
"ResamplerMicrophone", cg.Component, microphone.Microphone
|
||||
)
|
||||
|
||||
|
||||
def _set_stream_limits(config: ConfigType) -> ConfigType:
|
||||
# Only the sample rate changes; the bits and channels are those selected from the source microphone
|
||||
source = config[CONF_MICROPHONE]
|
||||
audio.set_stream_limits(
|
||||
min_bits_per_sample=source[CONF_BITS_PER_SAMPLE],
|
||||
max_bits_per_sample=source[CONF_BITS_PER_SAMPLE],
|
||||
min_channels=len(source[CONF_CHANNELS]),
|
||||
max_channels=len(source[CONF_CHANNELS]),
|
||||
min_sample_rate=config[CONF_SAMPLE_RATE],
|
||||
max_sample_rate=config[CONF_SAMPLE_RATE],
|
||||
)(config)
|
||||
return config
|
||||
|
||||
|
||||
CONFIG_SCHEMA = cv.All(
|
||||
microphone.MICROPHONE_SCHEMA.extend(
|
||||
{
|
||||
cv.GenerateID(): cv.declare_id(ResamplerMicrophone),
|
||||
cv.Required(CONF_MICROPHONE): microphone.microphone_source_schema(
|
||||
min_bits_per_sample=16,
|
||||
max_bits_per_sample=32,
|
||||
min_channels=1,
|
||||
max_channels=2,
|
||||
),
|
||||
cv.Optional(CONF_SAMPLE_RATE, default=16000): cv.int_range(8000, 48000),
|
||||
cv.Optional(CONF_FILTERS, default=16): cv.int_range(min=2, max=1024),
|
||||
cv.Optional(CONF_TAPS, default=16): validate_taps,
|
||||
}
|
||||
).extend(cv.COMPONENT_SCHEMA),
|
||||
cv.only_on([PLATFORM_ESP32]),
|
||||
_set_stream_limits,
|
||||
)
|
||||
|
||||
|
||||
FINAL_VALIDATE_SCHEMA = cv.Schema(
|
||||
{
|
||||
cv.Required(
|
||||
CONF_MICROPHONE
|
||||
): microphone.final_validate_microphone_source_schema("resampler"),
|
||||
},
|
||||
extra=cv.ALLOW_EXTRA,
|
||||
)
|
||||
|
||||
|
||||
async def to_code(config: ConfigType) -> None:
|
||||
mic_source = await microphone.microphone_source_to_code(config[CONF_MICROPHONE])
|
||||
var = cg.new_Pvariable(config[CONF_ID], mic_source)
|
||||
await cg.register_component(var, config)
|
||||
await microphone.register_microphone(var, config)
|
||||
|
||||
cg.add(var.set_target_sample_rate(config[CONF_SAMPLE_RATE]))
|
||||
cg.add(var.set_filters(config[CONF_FILTERS]))
|
||||
cg.add(var.set_taps(config[CONF_TAPS]))
|
||||
@@ -0,0 +1,174 @@
|
||||
#include "resampler_microphone.h"
|
||||
|
||||
#ifdef USE_ESP32
|
||||
|
||||
#include "esphome/core/helpers.h"
|
||||
#include "esphome/core/log.h"
|
||||
|
||||
#include <algorithm>
|
||||
|
||||
namespace esphome::resampler {
|
||||
|
||||
static const char *const TAG = "resampler.microphone";
|
||||
|
||||
// Duration of audio the resampler converts per step; longer source chunks are processed in several steps
|
||||
static constexpr uint32_t BUFFER_DURATION_MS = 16;
|
||||
|
||||
void ResamplerMicrophone::setup() {
|
||||
const audio::AudioStreamInfo input_stream_info = this->source_->get_audio_stream_info();
|
||||
this->audio_stream_info_ = audio::AudioStreamInfo(input_stream_info.get_bits_per_sample(),
|
||||
input_stream_info.get_channels(), this->target_sample_rate_);
|
||||
|
||||
// Allocate now for the expected source format; process_audio_ only sets up again if that format changes
|
||||
if (!this->init_resampler_(input_stream_info)) {
|
||||
this->mark_failed();
|
||||
return;
|
||||
}
|
||||
|
||||
this->source_->add_data_callback([this](const std::vector<uint8_t> &data) { this->process_audio_(data); });
|
||||
|
||||
this->disable_loop();
|
||||
}
|
||||
|
||||
void ResamplerMicrophone::dump_config() {
|
||||
ESP_LOGCONFIG(TAG,
|
||||
"Resampler Microphone:\n"
|
||||
" Target Sample Rate: %" PRIu32 " Hz\n"
|
||||
" Taps: %u\n"
|
||||
" Filters: %u",
|
||||
this->target_sample_rate_, this->taps_, this->filters_);
|
||||
}
|
||||
|
||||
void ResamplerMicrophone::start() {
|
||||
if (this->is_failed() || this->active_listeners_ == UINT8_MAX)
|
||||
return;
|
||||
++this->active_listeners_;
|
||||
this->enable_loop();
|
||||
}
|
||||
|
||||
void ResamplerMicrophone::stop() {
|
||||
if (this->active_listeners_ == 0)
|
||||
return;
|
||||
--this->active_listeners_;
|
||||
this->enable_loop();
|
||||
}
|
||||
|
||||
void ResamplerMicrophone::loop() {
|
||||
if (this->active_listeners_ == 0) {
|
||||
if (this->state_ != microphone::STATE_STOPPED) {
|
||||
this->source_->stop();
|
||||
this->state_ = microphone::STATE_STOPPED;
|
||||
}
|
||||
this->disable_loop();
|
||||
return;
|
||||
}
|
||||
|
||||
switch (this->state_) {
|
||||
case microphone::STATE_STOPPED:
|
||||
this->source_->start();
|
||||
this->state_ = microphone::STATE_STARTING;
|
||||
break;
|
||||
case microphone::STATE_STARTING:
|
||||
if (this->source_->is_running()) {
|
||||
this->state_ = microphone::STATE_RUNNING;
|
||||
}
|
||||
break;
|
||||
case microphone::STATE_RUNNING:
|
||||
// Follow the source if it restarts, e.g. after a driver error
|
||||
if (!this->source_->is_running()) {
|
||||
this->state_ = microphone::STATE_STARTING;
|
||||
}
|
||||
break;
|
||||
case microphone::STATE_STOPPING:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
bool ResamplerMicrophone::init_resampler_(const audio::AudioStreamInfo &input_stream_info) {
|
||||
this->resampler_.reset();
|
||||
this->resampler_ready_ = false;
|
||||
|
||||
if (input_stream_info.get_sample_rate() == this->target_sample_rate_) {
|
||||
// The source already delivers the target sample rate, so its audio is passed through unchanged
|
||||
this->input_stream_info_ = input_stream_info;
|
||||
this->resampler_ready_ = true;
|
||||
return true;
|
||||
}
|
||||
|
||||
const audio::AudioStreamInfo output_stream_info(input_stream_info.get_bits_per_sample(),
|
||||
input_stream_info.get_channels(), this->target_sample_rate_);
|
||||
|
||||
auto resampler = make_unique<esp_audio_libs::resampler::Resampler>(
|
||||
input_stream_info.ms_to_samples(BUFFER_DURATION_MS), output_stream_info.ms_to_samples(BUFFER_DURATION_MS));
|
||||
|
||||
esp_audio_libs::resampler::ResamplerConfiguration resample_config = {
|
||||
.source_sample_rate = static_cast<float>(input_stream_info.get_sample_rate()),
|
||||
.target_sample_rate = static_cast<float>(this->target_sample_rate_),
|
||||
.source_bits_per_sample = input_stream_info.get_bits_per_sample(),
|
||||
.target_bits_per_sample = input_stream_info.get_bits_per_sample(),
|
||||
.channels = input_stream_info.get_channels(),
|
||||
// Filters out frequencies above the new Nyquist limit when downsampling, to avoid aliasing
|
||||
.use_pre_or_post_filter = this->target_sample_rate_ < input_stream_info.get_sample_rate(),
|
||||
.subsample_interpolate = false, // Doubles the CPU load; more filters is a better alternative
|
||||
.number_of_taps = this->taps_,
|
||||
.number_of_filters = this->filters_,
|
||||
};
|
||||
|
||||
if (!resampler->initialize(resample_config)) {
|
||||
ESP_LOGE(TAG, "Not enough memory to resample");
|
||||
return false;
|
||||
}
|
||||
|
||||
this->output_buffer_.reserve(output_stream_info.ms_to_bytes(BUFFER_DURATION_MS));
|
||||
this->resampler_ = std::move(resampler);
|
||||
// Only set on success, so a failed set up is retried with the next chunk
|
||||
this->input_stream_info_ = input_stream_info;
|
||||
this->resampler_ready_ = true;
|
||||
return true;
|
||||
}
|
||||
|
||||
void ResamplerMicrophone::process_audio_(const std::vector<uint8_t> &data) {
|
||||
const audio::AudioStreamInfo input_stream_info = this->source_->get_audio_stream_info();
|
||||
if (input_stream_info != this->input_stream_info_) {
|
||||
this->init_resampler_(input_stream_info);
|
||||
}
|
||||
if (!this->resampler_ready_) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (this->resampler_ == nullptr) {
|
||||
this->data_callbacks_.call(data);
|
||||
return;
|
||||
}
|
||||
|
||||
const size_t input_bytes_per_frame = input_stream_info.frames_to_bytes(1);
|
||||
const uint32_t max_input_frames = input_stream_info.ms_to_frames(BUFFER_DURATION_MS);
|
||||
// Both limits match the sizes the resampler's internal buffers were allocated with in init_resampler_
|
||||
const uint32_t max_output_frames = this->audio_stream_info_.ms_to_frames(BUFFER_DURATION_MS);
|
||||
|
||||
const uint8_t *input = data.data();
|
||||
uint32_t input_frames = input_stream_info.bytes_to_frames(data.size());
|
||||
while (input_frames > 0) {
|
||||
// Stays within the reserved capacity, so this never reallocates
|
||||
this->output_buffer_.resize(this->audio_stream_info_.frames_to_bytes(max_output_frames));
|
||||
|
||||
// The resampler's internal buffers hold at most BUFFER_DURATION_MS of audio, so feed it in steps of that size.
|
||||
// 0 dB keeps the microphone level that downstream detectors are tuned for; overshoot saturates instead of wrapping.
|
||||
esp_audio_libs::resampler::ResamplerResults results = this->resampler_->resample(
|
||||
input, this->output_buffer_.data(), std::min(input_frames, max_input_frames), max_output_frames, 0.0f);
|
||||
|
||||
input += results.frames_used * input_bytes_per_frame;
|
||||
input_frames -= results.frames_used;
|
||||
|
||||
if (results.frames_generated > 0) {
|
||||
this->output_buffer_.resize(this->audio_stream_info_.frames_to_bytes(results.frames_generated));
|
||||
this->data_callbacks_.call(this->output_buffer_);
|
||||
} else if (results.frames_used == 0) {
|
||||
break; // No progress; drop the rest of the chunk instead of spinning
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace esphome::resampler
|
||||
|
||||
#endif // USE_ESP32
|
||||
@@ -0,0 +1,62 @@
|
||||
#pragma once
|
||||
|
||||
#ifdef USE_ESP32
|
||||
|
||||
#include "esphome/components/audio/audio.h"
|
||||
#include "esphome/components/microphone/microphone.h"
|
||||
#include "esphome/components/microphone/microphone_source.h"
|
||||
|
||||
#include "esphome/core/component.h"
|
||||
|
||||
#include <resampler.h> // esp-audio-libs
|
||||
|
||||
#include <memory>
|
||||
#include <vector>
|
||||
|
||||
namespace esphome::resampler {
|
||||
|
||||
/// @brief Microphone that converts the audio of a source microphone to a different sample rate.
|
||||
/// The bits per sample and channels are selected by the source's ``MicrophoneSource``; only the sample rate changes.
|
||||
/// Resampling runs in the source microphone's data callback, so it needs no task or ring buffer of its own.
|
||||
class ResamplerMicrophone final : public Component, public microphone::Microphone {
|
||||
public:
|
||||
explicit ResamplerMicrophone(microphone::MicrophoneSource *source) : source_(source) {}
|
||||
|
||||
void setup() override;
|
||||
void loop() override;
|
||||
void dump_config() override;
|
||||
|
||||
void start() override;
|
||||
void stop() override;
|
||||
|
||||
void set_target_sample_rate(uint32_t target_sample_rate) { this->target_sample_rate_ = target_sample_rate; }
|
||||
void set_filters(uint16_t filters) { this->filters_ = filters; }
|
||||
void set_taps(uint16_t taps) { this->taps_ = taps; }
|
||||
|
||||
protected:
|
||||
/// @brief Sets up the resampler for the given input format. No resampler is needed if the sample rates match.
|
||||
/// @return false if the resampler failed to allocate; the audio is then dropped
|
||||
bool init_resampler_(const audio::AudioStreamInfo &input_stream_info);
|
||||
|
||||
/// @brief Resamples a chunk of source audio and passes it to the data callbacks. Source microphone task only.
|
||||
void process_audio_(const std::vector<uint8_t> &data);
|
||||
|
||||
microphone::MicrophoneSource *source_;
|
||||
std::unique_ptr<esp_audio_libs::resampler::Resampler> resampler_;
|
||||
// Reused for every chunk so resampling does not allocate
|
||||
std::vector<uint8_t> output_buffer_;
|
||||
|
||||
// Format the resampler is set up for
|
||||
audio::AudioStreamInfo input_stream_info_;
|
||||
|
||||
uint32_t target_sample_rate_;
|
||||
uint16_t taps_;
|
||||
uint16_t filters_;
|
||||
|
||||
uint8_t active_listeners_{0};
|
||||
bool resampler_ready_{false};
|
||||
};
|
||||
|
||||
} // namespace esphome::resampler
|
||||
|
||||
#endif // USE_ESP32
|
||||
@@ -1,5 +1,3 @@
|
||||
from typing import Any
|
||||
|
||||
import esphome.codegen as cg
|
||||
from esphome.components import audio, psram, speaker
|
||||
import esphome.config_validation as cv
|
||||
@@ -17,16 +15,15 @@ from esphome.const import (
|
||||
from esphome.core.entity_helpers import inherit_property_from
|
||||
from esphome.types import ConfigType
|
||||
|
||||
from .. import CONF_TAPS, resampler_ns, validate_taps
|
||||
|
||||
AUTO_LOAD = ["audio"]
|
||||
CODEOWNERS = ["@kahrendt"]
|
||||
|
||||
resampler_ns = cg.esphome_ns.namespace("resampler")
|
||||
ResamplerSpeaker = resampler_ns.class_(
|
||||
"ResamplerSpeaker", cg.Component, speaker.Speaker
|
||||
)
|
||||
|
||||
CONF_TAPS = "taps"
|
||||
|
||||
PASSTHROUGH = "passthrough"
|
||||
|
||||
|
||||
@@ -60,13 +57,6 @@ def _validate_audio_compatibility(config: ConfigType) -> None:
|
||||
)(config)
|
||||
|
||||
|
||||
def _validate_taps(taps: Any) -> int:
|
||||
value = cv.int_range(min=16, max=128)(taps)
|
||||
if value % 4 != 0:
|
||||
raise cv.Invalid("Number of taps must be divisible by 4")
|
||||
return value
|
||||
|
||||
|
||||
CONFIG_SCHEMA = cv.All(
|
||||
speaker.SPEAKER_SCHEMA.extend(
|
||||
{
|
||||
@@ -80,7 +70,7 @@ CONFIG_SCHEMA = cv.All(
|
||||
): cv.positive_time_period_milliseconds,
|
||||
cv.Optional(CONF_TASK_STACK_IN_PSRAM): psram.validate_task_stack_in_psram,
|
||||
cv.Optional(CONF_FILTERS, default=16): cv.int_range(min=2, max=1024),
|
||||
cv.Optional(CONF_TAPS, default=16): _validate_taps,
|
||||
cv.Optional(CONF_TAPS, default=16): validate_taps,
|
||||
}
|
||||
).extend(cv.COMPONENT_SCHEMA),
|
||||
cv.only_on([PLATFORM_ESP32]),
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
microphone:
|
||||
- platform: i2s_audio
|
||||
id: resampler_i2s_mic_id
|
||||
i2s_audio_id: i2s_audio_bus
|
||||
adc_type: external
|
||||
i2s_din_pin: ${din_pin}
|
||||
sample_rate: 48000
|
||||
bits_per_sample: 32bit
|
||||
channel: stereo
|
||||
- platform: resampler
|
||||
id: resampler_mic_id
|
||||
microphone:
|
||||
microphone: resampler_i2s_mic_id
|
||||
channels: 0
|
||||
bits_per_sample: 16
|
||||
sample_rate: 16000
|
||||
on_data:
|
||||
- logger.log:
|
||||
format: "Received %u bytes"
|
||||
args: [x.size()]
|
||||
- platform: resampler
|
||||
id: resampler_mic_stereo_id
|
||||
microphone:
|
||||
microphone: resampler_i2s_mic_id
|
||||
channels: [0, 1]
|
||||
bits_per_sample: 32
|
||||
sample_rate: 44100
|
||||
filters: 32
|
||||
taps: 32
|
||||
@@ -0,0 +1,6 @@
|
||||
substitutions:
|
||||
din_pin: GPIO21
|
||||
|
||||
packages:
|
||||
i2s_audio: !include ../../test_build_components/common/i2s_audio/esp32-idf.yaml
|
||||
resampler: !include common-microphone.yaml
|
||||
@@ -0,0 +1,6 @@
|
||||
substitutions:
|
||||
din_pin: GPIO16
|
||||
|
||||
packages:
|
||||
i2s_audio: !include ../../test_build_components/common/i2s_audio/esp32-s3-idf.yaml
|
||||
resampler: !include common-microphone.yaml
|
||||
Reference in New Issue
Block a user