Files
aes67-ESP32-P4/main/decoder.c
T
bsncubed 3ebb27e4bb Step 7.4: HLS audio on AES67 (44.1 -> 48 kHz into the ring)
- decoder: s16 PCM -> stereo int32 -> esp_ae_rate_cvt 44.1 -> 48 kHz
  (32-bit, complexity 3; bypassed at 48 kHz; mono duplicated) -> ring.
  esp_audio_effects pinned to ~1.3.0 (1.4+ needs P4 rev >= 3).
- hls: each segment is downloaded completely into PSRAM (max 4 MB), then
  decoded; the connection is not held open while the decoder waits for
  ring space at playback speed.
- player: player_write() blocks while the ring is full and gives up when
  the source changes; the temporary 440 Hz producer is removed.
- Verified with Triple J Hottest: 441344 -> 480375 frames (10.008 s) and
  440320 -> 479260 (9.985 s) per segment; RTP 15000 packets, 0 gaps,
  peak -10 dBFS / RMS -22 dBFS; ring ~4.0 s, 0 underruns over ~50 s;
  heap 422 KB, PSRAM 27.5 MB free.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-25 14:52:19 +10:00

235 lines
8.4 KiB
C

#include "decoder.h"
#include <stdlib.h>
#include <string.h>
#include "esp_audio_dec_default.h"
#include "esp_audio_simple_dec.h"
#include "esp_audio_simple_dec_default.h"
#include "esp_ae_rate_cvt.h"
#include "player.h"
#include "esp_heap_caps.h"
#include "esp_log.h"
static const char *TAG = "decoder";
#define OUT_RATE 48000
#define CONV_FRAMES 4096 // per rate converter call
static esp_ae_rate_cvt_handle_t s_cvt; // NULL: no conversion (source is 48 kHz)
static uint32_t s_cvt_rate; // input rate the converter was opened for
static int32_t *s_in32, *s_out32; // stereo int32 work buffers (PSRAM)
static uint64_t s_seg_out; // 48 kHz frames written this segment
static esp_audio_simple_dec_handle_t s_dec;
static uint8_t *s_pcm;
static uint32_t s_pcm_size = 8192;
static esp_audio_simple_dec_info_t s_info;
static uint64_t s_seg_frames, s_seg_count, s_seg_es;
// MPEG-TS demux state: packets reassembled across HTTP chunks, PAT -> PMT -> ADTS-AAC PID.
static uint8_t s_pkt[188];
static size_t s_pkt_len;
static int s_pmt_pid = -1, s_audio_pid = -1;
static void report_segment(void)
{
if (s_seg_count && s_info.sample_rate) {
ESP_LOGI(TAG, "segment: %llu ES bytes -> %llu frames at %lu Hz = %.3f s -> %llu frames at 48 kHz = %.3f s",
s_seg_es, s_seg_frames, (unsigned long)s_info.sample_rate,
(double)s_seg_frames / s_info.sample_rate, s_seg_out, (double)s_seg_out / OUT_RATE);
}
s_seg_frames = 0;
s_seg_es = 0;
s_seg_out = 0;
s_seg_count++;
}
// Decoded PCM (s16, any channel count) -> stereo int32 -> 48 kHz -> ring.
static void output_pcm(const int16_t *pcm, uint32_t frames)
{
int ch = s_info.channel;
if (s_info.sample_rate != s_cvt_rate) {
if (s_cvt) {
esp_ae_rate_cvt_close(s_cvt);
s_cvt = NULL;
}
s_cvt_rate = s_info.sample_rate;
if (s_cvt_rate != OUT_RATE) {
esp_ae_rate_cvt_cfg_t cfg = {
.src_rate = s_cvt_rate, .dest_rate = OUT_RATE, .channel = 2, .bits_per_sample = 32,
.complexity = 3, .perf_type = ESP_AE_RATE_CVT_PERF_TYPE_SPEED,
};
if (esp_ae_rate_cvt_open(&cfg, &s_cvt) != ESP_AE_ERR_OK) {
ESP_LOGE(TAG, "rate converter %lu -> %d Hz: open failed", (unsigned long)s_cvt_rate, OUT_RATE);
} else {
ESP_LOGI(TAG, "rate converter %lu -> %d Hz", (unsigned long)s_cvt_rate, OUT_RATE);
}
}
}
while (frames) {
uint32_t n = frames < CONV_FRAMES ? frames : CONV_FRAMES;
for (uint32_t i = 0; i < n; i++) { // to stereo int32 (full scale = INT32_MAX)
int32_t l = (int32_t)pcm[i * ch] << 16;
s_in32[i * 2] = l;
s_in32[i * 2 + 1] = ch > 1 ? (int32_t)pcm[i * ch + 1] << 16 : l;
}
const int32_t *out = s_in32;
uint32_t out_n = n;
if (s_cvt) {
out_n = CONV_FRAMES * 2;
if (esp_ae_rate_cvt_process(s_cvt, s_in32, n, s_out32, &out_n) != ESP_AE_ERR_OK) {
ESP_LOGW(TAG, "rate conversion failed");
return;
}
out = s_out32;
}
s_seg_out += player_write(out, out_n);
pcm += n * ch;
frames -= n;
}
}
// Feed ADTS-AAC elementary stream bytes to the decoder.
static void decode_es(const uint8_t *data, size_t len)
{
s_seg_es += len;
esp_audio_simple_dec_raw_t raw = { .buffer = (uint8_t *)data, .len = len };
while (raw.len) {
esp_audio_simple_dec_out_t out = { .buffer = s_pcm, .len = s_pcm_size };
esp_audio_err_t err = esp_audio_simple_dec_process(s_dec, &raw, &out);
if (err == ESP_AUDIO_ERR_BUFF_NOT_ENOUGH) {
uint8_t *p = heap_caps_realloc(s_pcm, out.needed_size, MALLOC_CAP_SPIRAM);
if (!p) {
return;
}
s_pcm = p;
s_pcm_size = out.needed_size;
continue;
}
if (err != ESP_AUDIO_ERR_OK) {
ESP_LOGW(TAG, "decode error %d, resetting decoder", err);
esp_audio_simple_dec_reset(s_dec);
return;
}
if (out.decoded_size) {
if (!s_info.sample_rate) {
esp_audio_simple_dec_get_info(s_dec, &s_info);
ESP_LOGI(TAG, "stream: %lu Hz, %u ch, %u bit, %lu bit/s", (unsigned long)s_info.sample_rate,
s_info.channel, s_info.bits_per_sample, (unsigned long)s_info.bitrate);
}
uint32_t n = out.decoded_size / (s_info.channel * s_info.bits_per_sample / 8);
s_seg_frames += n;
if (s_info.bits_per_sample == 16) {
output_pcm((const int16_t *)out.buffer, n);
}
}
raw.buffer += raw.consumed;
raw.len -= raw.consumed;
}
}
static void ts_packet(const uint8_t *p)
{
if (p[0] != 0x47) {
return;
}
bool pusi = p[1] & 0x40;
int pid = ((p[1] & 0x1f) << 8) | p[2];
int afc = (p[3] >> 4) & 3;
if (!(afc & 1)) {
return; // no payload
}
size_t off = 4 + ((afc & 2) ? 1 + p[4] : 0);
if (off >= 188) {
return;
}
const uint8_t *pl = p + off;
size_t len = 188 - off;
if (pid == 0 && pusi) { // PAT: first program's PMT PID
const uint8_t *sec = pl + 1 + pl[0];
s_pmt_pid = ((sec[10] & 0x1f) << 8) | sec[11];
} else if (pid == s_pmt_pid && pusi && s_audio_pid < 0) { // PMT: first ADTS-AAC stream
const uint8_t *sec = pl + 1 + pl[0];
int slen = ((sec[1] & 0x0f) << 8) | sec[2];
int i = 12 + (((sec[10] & 0x0f) << 8) | sec[11]);
while (i + 5 <= 3 + slen - 4 && sec + i + 5 <= p + 188) {
int type = sec[i], epid = ((sec[i + 1] & 0x1f) << 8) | sec[i + 2];
if (type == 0x0F) {
s_audio_pid = epid;
ESP_LOGI(TAG, "TS: PMT PID 0x%x, ADTS-AAC on PID 0x%x", s_pmt_pid, epid);
break;
}
i += 5 + (((sec[i + 3] & 0x0f) << 8) | sec[i + 4]);
}
} else if (pid == s_audio_pid) {
if (pusi) { // strip the PES header
if (len < 9 || pl[0] || pl[1] || pl[2] != 1 || 9u + pl[8] > len) {
return;
}
size_t h = 9 + pl[8];
pl += h;
len -= h;
}
decode_es(pl, len);
}
}
// hls_sink_t: MPEG-TS bytes in any chunking.
bool decoder_feed(const uint8_t *data, size_t len, bool segment_start)
{
if (segment_start) {
report_segment();
s_pkt_len = 0; // segments start on a packet boundary
}
while (len) {
if (!s_pkt_len) {
// resync on 0x47 if needed, then take whole packets straight from the input
while (len && data[0] != 0x47) {
data++;
len--;
}
while (len >= 188 && data[0] == 0x47) {
ts_packet(data);
data += 188;
len -= 188;
}
if (len && data[0] != 0x47) {
continue;
}
}
size_t n = 188 - s_pkt_len < len ? 188 - s_pkt_len : len;
memcpy(s_pkt + s_pkt_len, data, n);
s_pkt_len += n;
data += n;
len -= n;
if (s_pkt_len == 188) {
ts_packet(s_pkt);
s_pkt_len = 0;
}
}
return true;
}
esp_err_t decoder_init(void)
{
esp_audio_dec_register_default();
esp_audio_simple_dec_register_default();
// Own TS demux (the library's TS decoder lost ~8% of the frames); AAC with ADTS headers,
// AAC-Plus on so HE-AAC v1/v2 variants work too.
esp_aac_dec_cfg_t aac_cfg = ESP_AAC_DEC_CONFIG_DEFAULT();
aac_cfg.aac_plus_enable = true;
esp_audio_simple_dec_cfg_t cfg = {
.dec_type = ESP_AUDIO_SIMPLE_DEC_TYPE_AAC, .dec_cfg = &aac_cfg, .cfg_size = sizeof(aac_cfg),
};
if (esp_audio_simple_dec_open(&cfg, &s_dec) != ESP_AUDIO_ERR_OK) {
ESP_LOGE(TAG, "AAC decoder open failed");
return ESP_FAIL;
}
s_pcm = heap_caps_malloc(s_pcm_size, MALLOC_CAP_SPIRAM);
s_in32 = heap_caps_malloc(CONV_FRAMES * 2 * sizeof(int32_t), MALLOC_CAP_SPIRAM);
s_out32 = heap_caps_malloc(CONV_FRAMES * 2 * 2 * sizeof(int32_t), MALLOC_CAP_SPIRAM);
return s_pcm && s_in32 && s_out32 ? ESP_OK : ESP_ERR_NO_MEM;
}