Step 7.5b3a: per-source PCM converter (audio_out), shared by HLS and Spotify

- main/audio_out: one converter instance per source (own rate converter
  state): s16 any rate/channels -> 48 kHz stereo int32 -> ring.
- player_write(src, ...) only writes for the active source, drops the
  rest; player_src_t is public in player.h.
- HLS decoder uses audio_out (no behaviour change).
- Verified: HLS still sample exact (441344 -> 480375 frames per
  10.008 s segment), buffer ~4 s, 0 underruns.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-25 18:34:15 +10:00
parent c240ac9a3c
commit 5b0d7c9ba4
6 changed files with 131 additions and 73 deletions
+6 -55
View File
@@ -6,19 +6,15 @@
#include "esp_audio_dec_default.h"
#include "esp_audio_simple_dec.h"
#include "esp_audio_simple_dec_default.h"
#include "esp_ae_rate_cvt.h"
#include "player.h"
#include "audio_out.h"
#include "esp_heap_caps.h"
#include "esp_log.h"
static const char *TAG = "decoder";
#define OUT_RATE 48000
#define CONV_FRAMES 4096 // per rate converter call
static esp_ae_rate_cvt_handle_t s_cvt; // NULL: no conversion (source is 48 kHz)
static uint32_t s_cvt_rate; // input rate the converter was opened for
static int32_t *s_in32, *s_out32; // stereo int32 work buffers (PSRAM)
static audio_conv_t *s_conv; // HLS: s16 -> 48 kHz -> ring
static uint64_t s_seg_out; // 48 kHz frames written this segment
static esp_audio_simple_dec_handle_t s_dec;
@@ -45,51 +41,6 @@ static void report_segment(void)
s_seg_count++;
}
// Decoded PCM (s16, any channel count) -> stereo int32 -> 48 kHz -> ring.
static void output_pcm(const int16_t *pcm, uint32_t frames)
{
int ch = s_info.channel;
if (s_info.sample_rate != s_cvt_rate) {
if (s_cvt) {
esp_ae_rate_cvt_close(s_cvt);
s_cvt = NULL;
}
s_cvt_rate = s_info.sample_rate;
if (s_cvt_rate != OUT_RATE) {
esp_ae_rate_cvt_cfg_t cfg = {
.src_rate = s_cvt_rate, .dest_rate = OUT_RATE, .channel = 2, .bits_per_sample = 32,
.complexity = 3, .perf_type = ESP_AE_RATE_CVT_PERF_TYPE_SPEED,
};
if (esp_ae_rate_cvt_open(&cfg, &s_cvt) != ESP_AE_ERR_OK) {
ESP_LOGE(TAG, "rate converter %lu -> %d Hz: open failed", (unsigned long)s_cvt_rate, OUT_RATE);
} else {
ESP_LOGI(TAG, "rate converter %lu -> %d Hz", (unsigned long)s_cvt_rate, OUT_RATE);
}
}
}
while (frames) {
uint32_t n = frames < CONV_FRAMES ? frames : CONV_FRAMES;
for (uint32_t i = 0; i < n; i++) { // to stereo int32 (full scale = INT32_MAX)
int32_t l = (int32_t)pcm[i * ch] << 16;
s_in32[i * 2] = l;
s_in32[i * 2 + 1] = ch > 1 ? (int32_t)pcm[i * ch + 1] << 16 : l;
}
const int32_t *out = s_in32;
uint32_t out_n = n;
if (s_cvt) {
out_n = CONV_FRAMES * 2;
if (esp_ae_rate_cvt_process(s_cvt, s_in32, n, s_out32, &out_n) != ESP_AE_ERR_OK) {
ESP_LOGW(TAG, "rate conversion failed");
return;
}
out = s_out32;
}
s_seg_out += player_write(out, out_n);
pcm += n * ch;
frames -= n;
}
}
// Feed ADTS-AAC elementary stream bytes to the decoder.
static void decode_es(const uint8_t *data, size_t len)
{
@@ -121,7 +72,8 @@ static void decode_es(const uint8_t *data, size_t len)
uint32_t n = out.decoded_size / (s_info.channel * s_info.bits_per_sample / 8);
s_seg_frames += n;
if (s_info.bits_per_sample == 16) {
output_pcm((const int16_t *)out.buffer, n);
s_seg_out += audio_conv_write(s_conv, (const int16_t *)out.buffer, n, s_info.channel,
s_info.sample_rate);
}
}
raw.buffer += raw.consumed;
@@ -228,7 +180,6 @@ esp_err_t decoder_init(void)
return ESP_FAIL;
}
s_pcm = heap_caps_malloc(s_pcm_size, MALLOC_CAP_SPIRAM);
s_in32 = heap_caps_malloc(CONV_FRAMES * 2 * sizeof(int32_t), MALLOC_CAP_SPIRAM);
s_out32 = heap_caps_malloc(CONV_FRAMES * 2 * 2 * sizeof(int32_t), MALLOC_CAP_SPIRAM);
return s_pcm && s_in32 && s_out32 ? ESP_OK : ESP_ERR_NO_MEM;
s_conv = audio_conv_create(PLAYER_SRC_HLS, "hls");
return s_pcm && s_conv ? ESP_OK : ESP_ERR_NO_MEM;
}