Step 7.5b3a: per-source PCM converter (audio_out), shared by HLS and Spotify

- main/audio_out: one converter instance per source (own rate converter
  state): s16 any rate/channels -> 48 kHz stereo int32 -> ring.
- player_write(src, ...) only writes for the active source, drops the
  rest; player_src_t is public in player.h.
- HLS decoder uses audio_out (no behaviour change).
- Verified: HLS still sample exact (441344 -> 480375 frames per
  10.008 s segment), buffer ~4 s, 0 underruns.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
2026-09-25 18:34:15 +10:00
parent c240ac9a3c
commit 5b0d7c9ba4
6 changed files with 131 additions and 73 deletions
+1 -1
View File
@@ -1,4 +1,4 @@
idf_component_register(SRCS "main.c" "project_cfg.c" "player.c" "audio_ring.c" "hls.c" "decoder.c"
idf_component_register(SRCS "main.c" "project_cfg.c" "player.c" "audio_ring.c" "hls.c" "decoder.c" "audio_out.c"
INCLUDE_DIRS "."
REQUIRES esp_app_format esp_hw_support heap esp_http_client mbedtls esp_timer
aes67_board aes67_health aes67_net aes67_ota aes67_ptp aes67_sdp_sap
+90
View File
@@ -0,0 +1,90 @@
#include "audio_out.h"
#include <stdlib.h>
#include "esp_ae_rate_cvt.h"
#include "esp_heap_caps.h"
#include "esp_log.h"
#define OUT_RATE 48000
#define CONV_FRAMES 4096 // per rate converter call
static const char *TAG = "audio_out";
struct audio_conv {
player_src_t src;
const char *name;
esp_ae_rate_cvt_handle_t cvt; // NULL: no conversion (source is 48 kHz)
uint32_t rate; // input rate the converter was opened for
int32_t *in32, *out32; // stereo int32 work buffers (PSRAM)
};
audio_conv_t *audio_conv_create(player_src_t src, const char *name)
{
audio_conv_t *c = calloc(1, sizeof(*c));
if (!c) {
return NULL;
}
c->src = src;
c->name = name;
c->in32 = heap_caps_malloc(CONV_FRAMES * 2 * sizeof(int32_t), MALLOC_CAP_SPIRAM);
c->out32 = heap_caps_malloc(CONV_FRAMES * 2 * 2 * sizeof(int32_t), MALLOC_CAP_SPIRAM);
if (!c->in32 || !c->out32) {
free(c->in32);
free(c->out32);
free(c);
return NULL;
}
return c;
}
static void set_rate(audio_conv_t *c, uint32_t rate)
{
if (c->cvt) {
esp_ae_rate_cvt_close(c->cvt);
c->cvt = NULL;
}
c->rate = rate;
if (rate == OUT_RATE) {
return;
}
esp_ae_rate_cvt_cfg_t cfg = {
.src_rate = rate, .dest_rate = OUT_RATE, .channel = 2, .bits_per_sample = 32,
.complexity = 3, .perf_type = ESP_AE_RATE_CVT_PERF_TYPE_SPEED,
};
if (esp_ae_rate_cvt_open(&cfg, &c->cvt) != ESP_AE_ERR_OK) {
ESP_LOGE(TAG, "%s: rate converter %lu -> %d Hz: open failed", c->name, (unsigned long)rate, OUT_RATE);
} else {
ESP_LOGI(TAG, "%s: rate converter %lu -> %d Hz", c->name, (unsigned long)rate, OUT_RATE);
}
}
size_t audio_conv_write(audio_conv_t *c, const int16_t *pcm, uint32_t frames, int ch, uint32_t rate)
{
if (rate != c->rate) {
set_rate(c, rate);
}
size_t produced = 0;
while (frames) {
uint32_t n = frames < CONV_FRAMES ? frames : CONV_FRAMES;
for (uint32_t i = 0; i < n; i++) { // to stereo int32 (full scale = INT32_MAX)
int32_t l = (int32_t)pcm[i * ch] << 16;
c->in32[i * 2] = l;
c->in32[i * 2 + 1] = ch > 1 ? (int32_t)pcm[i * ch + 1] << 16 : l;
}
const int32_t *out = c->in32;
uint32_t out_n = n;
if (c->cvt) {
out_n = CONV_FRAMES * 2;
if (esp_ae_rate_cvt_process(c->cvt, c->in32, n, c->out32, &out_n) != ESP_AE_ERR_OK) {
ESP_LOGW(TAG, "%s: rate conversion failed", c->name);
return produced;
}
out = c->out32;
}
produced += player_write(c->src, out, out_n);
pcm += n * ch;
frames -= n;
}
return produced;
}
+15
View File
@@ -0,0 +1,15 @@
// PCM from a source (s16, any rate/channels) -> 48 kHz stereo int32 -> player ring.
// One converter per source: each keeps its own resampler state.
#pragma once
#include <stddef.h>
#include <stdint.h>
#include "player.h"
typedef struct audio_conv audio_conv_t;
audio_conv_t *audio_conv_create(player_src_t src, const char *name);
// Blocks while the ring is full; dropped when src is not the active source.
// Returns the 48 kHz frames produced.
size_t audio_conv_write(audio_conv_t *c, const int16_t *pcm, uint32_t frames, int channels, uint32_t rate);
+6 -55
View File
@@ -6,19 +6,15 @@
#include "esp_audio_dec_default.h"
#include "esp_audio_simple_dec.h"
#include "esp_audio_simple_dec_default.h"
#include "esp_ae_rate_cvt.h"
#include "player.h"
#include "audio_out.h"
#include "esp_heap_caps.h"
#include "esp_log.h"
static const char *TAG = "decoder";
#define OUT_RATE 48000
#define CONV_FRAMES 4096 // per rate converter call
static esp_ae_rate_cvt_handle_t s_cvt; // NULL: no conversion (source is 48 kHz)
static uint32_t s_cvt_rate; // input rate the converter was opened for
static int32_t *s_in32, *s_out32; // stereo int32 work buffers (PSRAM)
static audio_conv_t *s_conv; // HLS: s16 -> 48 kHz -> ring
static uint64_t s_seg_out; // 48 kHz frames written this segment
static esp_audio_simple_dec_handle_t s_dec;
@@ -45,51 +41,6 @@ static void report_segment(void)
s_seg_count++;
}
// Decoded PCM (s16, any channel count) -> stereo int32 -> 48 kHz -> ring.
static void output_pcm(const int16_t *pcm, uint32_t frames)
{
int ch = s_info.channel;
if (s_info.sample_rate != s_cvt_rate) {
if (s_cvt) {
esp_ae_rate_cvt_close(s_cvt);
s_cvt = NULL;
}
s_cvt_rate = s_info.sample_rate;
if (s_cvt_rate != OUT_RATE) {
esp_ae_rate_cvt_cfg_t cfg = {
.src_rate = s_cvt_rate, .dest_rate = OUT_RATE, .channel = 2, .bits_per_sample = 32,
.complexity = 3, .perf_type = ESP_AE_RATE_CVT_PERF_TYPE_SPEED,
};
if (esp_ae_rate_cvt_open(&cfg, &s_cvt) != ESP_AE_ERR_OK) {
ESP_LOGE(TAG, "rate converter %lu -> %d Hz: open failed", (unsigned long)s_cvt_rate, OUT_RATE);
} else {
ESP_LOGI(TAG, "rate converter %lu -> %d Hz", (unsigned long)s_cvt_rate, OUT_RATE);
}
}
}
while (frames) {
uint32_t n = frames < CONV_FRAMES ? frames : CONV_FRAMES;
for (uint32_t i = 0; i < n; i++) { // to stereo int32 (full scale = INT32_MAX)
int32_t l = (int32_t)pcm[i * ch] << 16;
s_in32[i * 2] = l;
s_in32[i * 2 + 1] = ch > 1 ? (int32_t)pcm[i * ch + 1] << 16 : l;
}
const int32_t *out = s_in32;
uint32_t out_n = n;
if (s_cvt) {
out_n = CONV_FRAMES * 2;
if (esp_ae_rate_cvt_process(s_cvt, s_in32, n, s_out32, &out_n) != ESP_AE_ERR_OK) {
ESP_LOGW(TAG, "rate conversion failed");
return;
}
out = s_out32;
}
s_seg_out += player_write(out, out_n);
pcm += n * ch;
frames -= n;
}
}
// Feed ADTS-AAC elementary stream bytes to the decoder.
static void decode_es(const uint8_t *data, size_t len)
{
@@ -121,7 +72,8 @@ static void decode_es(const uint8_t *data, size_t len)
uint32_t n = out.decoded_size / (s_info.channel * s_info.bits_per_sample / 8);
s_seg_frames += n;
if (s_info.bits_per_sample == 16) {
output_pcm((const int16_t *)out.buffer, n);
s_seg_out += audio_conv_write(s_conv, (const int16_t *)out.buffer, n, s_info.channel,
s_info.sample_rate);
}
}
raw.buffer += raw.consumed;
@@ -228,7 +180,6 @@ esp_err_t decoder_init(void)
return ESP_FAIL;
}
s_pcm = heap_caps_malloc(s_pcm_size, MALLOC_CAP_SPIRAM);
s_in32 = heap_caps_malloc(CONV_FRAMES * 2 * sizeof(int32_t), MALLOC_CAP_SPIRAM);
s_out32 = heap_caps_malloc(CONV_FRAMES * 2 * 2 * sizeof(int32_t), MALLOC_CAP_SPIRAM);
return s_pcm && s_in32 && s_out32 ? ESP_OK : ESP_ERR_NO_MEM;
s_conv = audio_conv_create(PLAYER_SRC_HLS, "hls");
return s_pcm && s_conv ? ESP_OK : ESP_ERR_NO_MEM;
}
+15 -16
View File
@@ -20,17 +20,16 @@
static const char *TAG = "player";
typedef enum { SRC_TONE, SRC_OFF, SRC_HLS, SRC_SPOTIFY } src_t;
static const char *const SRC_NAME[] = { "tone", "off", "hls", "spotify" };
static volatile src_t s_src = SRC_OFF;
static volatile player_src_t s_src = PLAYER_SRC_OFF;
static volatile bool s_playing; // ring output running (after prefill)
static volatile bool s_flush; // consumer drops buffered audio on the next read
static src_t mode_of(const char *m)
static player_src_t mode_of(const char *m)
{
return !strcmp(m, "tone") ? SRC_TONE : !strcmp(m, "hls") ? SRC_HLS :
!strcmp(m, "spotify") ? SRC_SPOTIFY : !strcmp(m, "auto") ? SRC_HLS /* failover: step 7 */ : SRC_OFF;
return !strcmp(m, "tone") ? PLAYER_SRC_TONE : !strcmp(m, "hls") ? PLAYER_SRC_HLS :
!strcmp(m, "spotify") ? PLAYER_SRC_SPOTIFY : !strcmp(m, "auto") ? PLAYER_SRC_HLS /* failover: step 7 */ : PLAYER_SRC_OFF;
}
// AES67 TX pull callback (TX task, must not block). Silence while buffering or off: not an underrun.
@@ -41,7 +40,7 @@ static size_t player_read(int32_t *buf, size_t frames)
s_flush = false;
s_playing = false;
}
if (s_src == SRC_OFF || s_src == SRC_SPOTIFY) { // Spotify: step 7 (cspot)
if (s_src == PLAYER_SRC_OFF || s_src == PLAYER_SRC_SPOTIFY) { // Spotify: step 7 (cspot)
memset(buf, 0, frames * CHANNELS * sizeof(int32_t));
return frames;
}
@@ -61,25 +60,25 @@ static size_t player_read(int32_t *buf, size_t frames)
void player_apply(const cJSON *source)
{
src_t src = mode_of(cJSON_GetObjectItemCaseSensitive(source, "mode")->valuestring);
player_src_t src = mode_of(cJSON_GetObjectItemCaseSensitive(source, "mode")->valuestring);
if (src == s_src) {
return;
}
s_src = src;
s_flush = true;
// The test tone is TX's own (phase-locked to PTP); everything else goes through player_read.
aes67_tx_set_source(src == SRC_TONE ? NULL : player_read);
aes67_tx_set_source(src == PLAYER_SRC_TONE ? NULL : player_read);
ESP_LOGI(TAG, "source: %s", SRC_NAME[src]);
}
// Source side (HLS task): write converted 48 kHz frames, waiting for space at playback speed.
// Gives up when the source is no longer HLS.
size_t player_write(const int32_t *frames, size_t n)
// Source side: write converted 48 kHz frames, waiting for space at playback speed.
// Frames of a source that is not active are dropped.
size_t player_write(player_src_t src, const int32_t *frames, size_t n)
{
size_t done = 0;
while (done < n) {
if (s_src != SRC_HLS) {
return done;
if (s_src != src) {
return n; // not the active source: drop
}
size_t w = audio_ring_write(frames + done * CHANNELS, n - done);
if (!w) {
@@ -92,8 +91,8 @@ size_t player_write(const int32_t *frames, size_t n)
static void player_status(cJSON *st)
{
const char *state = s_src == SRC_TONE ? "playing" : s_src == SRC_OFF ? "idle" :
s_src == SRC_SPOTIFY ? "not implemented" : s_playing ? "playing" : "buffering";
const char *state = s_src == PLAYER_SRC_TONE ? "playing" : s_src == PLAYER_SRC_OFF ? "idle" :
s_src == PLAYER_SRC_SPOTIFY ? "not implemented" : s_playing ? "playing" : "buffering";
cJSON_AddStringToObject(st, "active_source", SRC_NAME[s_src]);
cJSON_AddStringToObject(st, "source_state", state);
cJSON_AddStringToObject(st, "spotify_state", spotify_state());
@@ -108,7 +107,7 @@ esp_err_t player_init(void)
return err;
}
cJSON *src = cfg_get("source");
s_src = (src_t)-1;
s_src = (player_src_t)-1;
player_apply(src);
cJSON_Delete(src);
err = decoder_init();
+4 -1
View File
@@ -7,8 +7,11 @@
#include "cJSON.h"
#include "esp_err.h"
typedef enum { PLAYER_SRC_TONE, PLAYER_SRC_OFF, PLAYER_SRC_HLS, PLAYER_SRC_SPOTIFY } player_src_t;
esp_err_t player_init(void);
// Apply the "source" config group (mode etc.).
void player_apply(const cJSON *source);
// Source side: write 48 kHz stereo int32 frames, blocking while the ring is full.
size_t player_write(const int32_t *frames, size_t n);
// Frames from a source that is not the active one are dropped (returns n).
size_t player_write(player_src_t src, const int32_t *frames, size_t n);