一、PCM → AAC 的转换
extern "C" {
#include <libavcodec/avcodec.h>
#include <libavutil/audio_fifo.h>
}
// 把 PCM 帧编码为 AAC packet
struct AudioEncoder {
AVCodecContext* ctx = nullptr;
bool init(int sample_rate, int channels) {
const AVCodec* codec = avcodec_find_encoder(AV_CODEC_ID_AAC);
if (!codec) return false;
ctx = avcodec_alloc_context3(codec);
ctx->sample_fmt = AV_SAMPLE_FLT_FLTP; // AAC 编码器要求 float planar
ctx->sample_rate = sample_rate;
ctx->channel_layout = channels == 1 ? AV_CH_LAYOUT_MONO : AV_CH_LAYOUT_STEREO;
ctx->channels = channels;
ctx->bit_rate = 128000; // 128 kbps
return avcodec_open2(ctx, codec, nullptr) >= 0;
}
// 编码一帧 PCM 数据
// 注意:AAC 一帧固定 1024 个样本
AVPacket* encode(const float* pcm_data, int num_samples) {
AVFrame* frame = av_frame_alloc();
frame->nb_samples = num_samples;
frame->format = ctx->sample_fmt;
frame->channel_layout = ctx->channel_layout;
av_frame_get_buffer(frame, 0);
// 把输入 PCM 拷贝到 frame(planar 格式:ch0 全在 data[0], ch1 全在 data[1])
memcpy(frame->data[0], pcm_data, num_samples * sizeof(float));
AVPacket* packet = av_packet_alloc();
if (avcodec_send_frame(ctx, frame) == 0) {
avcodec_receive_packet(ctx, packet);
}
av_frame_free(&frame);
return packet;
}
~AudioEncoder() { avcodec_free_context(&ctx); }
};
二、AAC 的 ADTS 头部
当你从编码器拿到 AAC 裸流时,每帧数据开头有一个 ADTS 头部(7 或 9 字节),包含了采样率、声道数、帧长度等信息。没有 ADTS 头部,解码器不知道数据怎么解析。
ADTS 头部结构(7 字节):
┌─┬─┬─┬─┬─┬─┬─┬─┐
│ syncword: 0xFFF (12bit) │ 同步头,标识一个 AAC 帧的开始
├─┼─┼─┼─┼─┼─┼─┼─┤
│ ID: 0 (1bit) │ MPEG 版本
│ layer: 0 (2bit) │ 总是 0
│ protection_absent: 1 (1bit) │ 是否有 CRC
├─┼─┼─┼─┼─┼─┼─┼─┤
│ profile: 1 (2bit) │ AAC-LC = 1
│ sampling_freq: 4 (4bit) │ 44100Hz = 4, 48000Hz = 3
│ private: 0 (1bit) │
├─┼─┼─┼─┼─┼─┼─┼─┤
│ channel_config: 2 (3bit) │ 1=Mono, 2=Stereo
│ original: 0 (1bit) │
│ home: 0 (1bit) │
│ copyrighted: 0 (1bit) │
│ copyright_start: 0 (1bit) │
├─┼─┼─┼─┼─┼─┼─┼─┤
│ frame_length: (13bit) │ 这一帧的总字节数(含 ADTS 头)
├─┼─┼─┼─┼─┼─┼─┼─┤
│ ... 剩余信息 │
└─┴─┴─┴─┴─┴─┴─┴─┘
手动解析 ADTS 头部的 C++ 代码:
struct ADTSHeader {
int profile; // 0=Main, 1=LC, 2=SSR
int sample_rate; // 实际采样率
int channels; // 声道数
int frame_length; // 这一帧总长度(字节)
};
ADTSHeader parse_adts(const uint8_t* data) {
ADTSHeader hdr = {};
// 前 12 bit 是 syncword 0xFFF
// uint16_t sync = (data[0] << 4) | (data[1] >> 4);
hdr.profile = ((data[2] >> 6) & 0x03) + 1; // profile 从 1 开始
static const int sample_rates[] = {
96000, 88200, 64000, 48000, 44100, 32000,
24000, 22050, 16000, 12000, 11025, 8000
};
int sr_idx = (data[2] >> 2) & 0x0F;
hdr.sample_rate = sample_rates[sr_idx];
hdr.channels = ((data[2] & 0x01) << 2) | ((data[3] >> 6) & 0x03);
// frame_length 跨 data[3]-data[4]-data[5] 共 13bit
hdr.frame_length = ((data[3] & 0x03) << 11) | (data[4] << 3) | ((data[5] >> 5) & 0x07);
return hdr;
}