Files
aes67-ESP32-P4/main/decoder.c
T
bsncubed 8a6cb53e3b Step 7.3: decode HLS TS segments to PCM (own TS demux + AAC decoder)
- main/decoder: MPEG-TS demux (packets reassembled across HTTP chunks,
  PAT -> PMT -> first ADTS-AAC PID, PES headers stripped) feeding the
  esp_audio_codec simple AAC decoder (ADTS, AAC-Plus enabled for HE-AAC
  v1/v2 variants). 7.3 only counts and logs PCM per segment.
- esp_audio_codec pinned to ~2.5.0: 2.6+ needs P4 rev >= 3 (this board
  is rev 1.3). Noted in CLAUDE.md, also for esp_audio_effects < 1.4.
- The library's combined TS decoder lost ~8% of the frames (segments
  decoded to 7.9-9.6 s, "decode error -1"); with the own demux every
  segment is sample exact: 441344 / 440320 frames = 10.008 / 9.985 s,
  matching EXTINF 10.0078 / 9.9846 (431 / 430 AAC frames), no errors.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-09-25 14:48:12 +10:00

173 lines
5.8 KiB
C

#include "decoder.h"
#include <stdlib.h>
#include <string.h>
#include "esp_audio_dec_default.h"
#include "esp_audio_simple_dec.h"
#include "esp_audio_simple_dec_default.h"
#include "esp_heap_caps.h"
#include "esp_log.h"
static const char *TAG = "decoder";
static esp_audio_simple_dec_handle_t s_dec;
static uint8_t *s_pcm;
static uint32_t s_pcm_size = 8192;
static esp_audio_simple_dec_info_t s_info;
static uint64_t s_seg_frames, s_seg_count, s_seg_es;
// MPEG-TS demux state: packets reassembled across HTTP chunks, PAT -> PMT -> ADTS-AAC PID.
static uint8_t s_pkt[188];
static size_t s_pkt_len;
static int s_pmt_pid = -1, s_audio_pid = -1;
static void report_segment(void)
{
if (s_seg_count && s_info.sample_rate) {
ESP_LOGI(TAG, "segment: %llu ES bytes -> %llu PCM frames = %.3f s (%lu Hz, %u ch, %u bit)",
s_seg_es, s_seg_frames, (double)s_seg_frames / s_info.sample_rate,
(unsigned long)s_info.sample_rate, s_info.channel, s_info.bits_per_sample);
}
s_seg_frames = 0;
s_seg_es = 0;
s_seg_count++;
}
// Feed ADTS-AAC elementary stream bytes to the decoder.
static void decode_es(const uint8_t *data, size_t len)
{
s_seg_es += len;
esp_audio_simple_dec_raw_t raw = { .buffer = (uint8_t *)data, .len = len };
while (raw.len) {
esp_audio_simple_dec_out_t out = { .buffer = s_pcm, .len = s_pcm_size };
esp_audio_err_t err = esp_audio_simple_dec_process(s_dec, &raw, &out);
if (err == ESP_AUDIO_ERR_BUFF_NOT_ENOUGH) {
uint8_t *p = heap_caps_realloc(s_pcm, out.needed_size, MALLOC_CAP_SPIRAM);
if (!p) {
return;
}
s_pcm = p;
s_pcm_size = out.needed_size;
continue;
}
if (err != ESP_AUDIO_ERR_OK) {
ESP_LOGW(TAG, "decode error %d, resetting decoder", err);
esp_audio_simple_dec_reset(s_dec);
return;
}
if (out.decoded_size) {
if (!s_info.sample_rate) {
esp_audio_simple_dec_get_info(s_dec, &s_info);
ESP_LOGI(TAG, "stream: %lu Hz, %u ch, %u bit, %lu bit/s", (unsigned long)s_info.sample_rate,
s_info.channel, s_info.bits_per_sample, (unsigned long)s_info.bitrate);
}
s_seg_frames += out.decoded_size / (s_info.channel * s_info.bits_per_sample / 8);
}
raw.buffer += raw.consumed;
raw.len -= raw.consumed;
}
}
static void ts_packet(const uint8_t *p)
{
if (p[0] != 0x47) {
return;
}
bool pusi = p[1] & 0x40;
int pid = ((p[1] & 0x1f) << 8) | p[2];
int afc = (p[3] >> 4) & 3;
if (!(afc & 1)) {
return; // no payload
}
size_t off = 4 + ((afc & 2) ? 1 + p[4] : 0);
if (off >= 188) {
return;
}
const uint8_t *pl = p + off;
size_t len = 188 - off;
if (pid == 0 && pusi) { // PAT: first program's PMT PID
const uint8_t *sec = pl + 1 + pl[0];
s_pmt_pid = ((sec[10] & 0x1f) << 8) | sec[11];
} else if (pid == s_pmt_pid && pusi && s_audio_pid < 0) { // PMT: first ADTS-AAC stream
const uint8_t *sec = pl + 1 + pl[0];
int slen = ((sec[1] & 0x0f) << 8) | sec[2];
int i = 12 + (((sec[10] & 0x0f) << 8) | sec[11]);
while (i + 5 <= 3 + slen - 4 && sec + i + 5 <= p + 188) {
int type = sec[i], epid = ((sec[i + 1] & 0x1f) << 8) | sec[i + 2];
if (type == 0x0F) {
s_audio_pid = epid;
ESP_LOGI(TAG, "TS: PMT PID 0x%x, ADTS-AAC on PID 0x%x", s_pmt_pid, epid);
break;
}
i += 5 + (((sec[i + 3] & 0x0f) << 8) | sec[i + 4]);
}
} else if (pid == s_audio_pid) {
if (pusi) { // strip the PES header
if (len < 9 || pl[0] || pl[1] || pl[2] != 1 || 9u + pl[8] > len) {
return;
}
size_t h = 9 + pl[8];
pl += h;
len -= h;
}
decode_es(pl, len);
}
}
// hls_sink_t: MPEG-TS bytes in any chunking.
bool decoder_feed(const uint8_t *data, size_t len, bool segment_start)
{
if (segment_start) {
report_segment();
s_pkt_len = 0; // segments start on a packet boundary
}
while (len) {
if (!s_pkt_len) {
// resync on 0x47 if needed, then take whole packets straight from the input
while (len && data[0] != 0x47) {
data++;
len--;
}
while (len >= 188 && data[0] == 0x47) {
ts_packet(data);
data += 188;
len -= 188;
}
if (len && data[0] != 0x47) {
continue;
}
}
size_t n = 188 - s_pkt_len < len ? 188 - s_pkt_len : len;
memcpy(s_pkt + s_pkt_len, data, n);
s_pkt_len += n;
data += n;
len -= n;
if (s_pkt_len == 188) {
ts_packet(s_pkt);
s_pkt_len = 0;
}
}
return true;
}
esp_err_t decoder_init(void)
{
esp_audio_dec_register_default();
esp_audio_simple_dec_register_default();
// Own TS demux (the library's TS decoder lost ~8% of the frames); AAC with ADTS headers,
// AAC-Plus on so HE-AAC v1/v2 variants work too.
esp_aac_dec_cfg_t aac_cfg = ESP_AAC_DEC_CONFIG_DEFAULT();
aac_cfg.aac_plus_enable = true;
esp_audio_simple_dec_cfg_t cfg = {
.dec_type = ESP_AUDIO_SIMPLE_DEC_TYPE_AAC, .dec_cfg = &aac_cfg, .cfg_size = sizeof(aac_cfg),
};
if (esp_audio_simple_dec_open(&cfg, &s_dec) != ESP_AUDIO_ERR_OK) {
ESP_LOGE(TAG, "AAC decoder open failed");
return ESP_FAIL;
}
s_pcm = heap_caps_malloc(s_pcm_size, MALLOC_CAP_SPIRAM);
return s_pcm ? ESP_OK : ESP_ERR_NO_MEM;
}