文档目录
#include <vector>
#include <cstdint>
#include <cstdlib>

// 模拟一个非常简化的视频编码器
// 只编码 I 帧和 P 帧(无 B 帧),观察不同场景下的压缩效率

class SimpleEncoder {
    int width_, height_;
    std::vector<uint8_t> reference_;  // 参考帧
    int keyframe_interval_;           // 每隔几帧插一个 I 帧
    int frame_count_ = 0;

    // "运动估计":在参考帧中搜索最匹配的 8x8 块
    struct MotionVector { int dx, dy; };
    MotionVector search_block(const uint8_t* cur, const uint8_t* ref,
                               int bx, int by, int search_range = 16) {
        int best_cost = INT_MAX;
        MotionVector best = {0, 0};

        for (int dy = -search_range; dy <= search_range; dy++) {
            for (int dx = -search_range; dx <= search_range; dx++) {
                int cost = 0;
                for (int y = 0; y < 8; y++) {
                    for (int x = 0; x < 8; x++) {
                        int cur_px = cur[(by + y) * width_ + (bx + x)];
                        int ref_px = ref[(by + y + dy) * width_ + (bx + x + dx)];
                        cost += abs(cur_px - ref_px);
                    }
                }
                if (cost < best_cost) {
                    best_cost = cost;
                    best = {dx, dy};
                }
            }
        }
        return best;
    }

public:
    SimpleEncoder(int w, int h, int keyframe_interval)
        : width_(w), height_(h), reference_(w * h), keyframe_interval_(keyframe_interval) {
        std::fill(reference_.begin(), reference_.end(), 128);
    }

    struct EncodedFrame {
        bool is_keyframe;
        size_t data_size;  // 编码后的估计大小
    };

    EncodedFrame encode(const uint8_t* frame) {
        frame_count_++;
        EncodedFrame ef;

        if (frame_count_ % keyframe_interval_ == 1) {
            // I 帧:完整存储(模拟 JPEG 级压缩,约 1/3 原大小)
            ef.is_keyframe = true;
            ef.data_size = width_ * height_;  // 简化:算 1:1
            memcpy(reference_.data(), frame, width_ * height_);
        } else {
            // P 帧:只存运动向量 + 残差
            ef.is_keyframe = false;
            size_t motion_data = 0;
            size_t residual_data = 0;

            for (int y = 0; y < height_; y += 8) {
                for (int x = 0; x < width_; x += 8) {
                    auto mv = search_block(frame, reference_.data(), x, y);
                    // 运动向量:8 位够存 dx, dy
                    motion_data += 2;

                    // 残差:用绝对差和估计
                    for (int by = 0; by < 8; by++) {
                        for (int bx = 0; bx < 8; bx++) {
                            int cur_px = frame[(y + by) * width_ + (x + bx)];
                            int ref_px = reference_[(y + by + mv.dy) * width_ + (x + mv.dx + bx)];
                            residual_data += abs(cur_px - ref_px) < 8 ? 0 : 1;
                        }
                    }
                }
            }
            ef.data_size = motion_data + residual_data;
            memcpy(reference_.data(), frame, width_ * height_);
        }

        return ef;
    }
};

void test_encoder_scenarios() {
    const int W = 64, H = 48;
    SimpleEncoder encoder(W, H, 30);

    // 场景 1:静态画面(摄像头静止)
    std::vector<uint8_t> static_frame(W * H, 128);

    // 场景 2:快速运动(画面整体右移)
    std::vector<uint8_t> motion_frame(W * H);
    for (int y = 0; y < H; y++)
        for (int x = 0; x < W; x++)
            motion_frame[y * W + x] = (x + y * 3) % 256;

    printf("静态场景编码(30 帧):\n");
    for (int i = 0; i < 30; i++) {
        auto ef = encoder.encode(static_frame.data());
        printf("  帧 %2d: %s, size=%zu\n", i + 1,
               ef.is_keyframe ? "I" : "P", ef.data_size);
    }

    // 重置编码器
    SimpleEncoder motion_enc(W, H, 30);
    printf("\n动态场景编码(30 帧,每帧整体右移 1 像素):\n");
    for (int i = 0; i < 30; i++) {
        // 模拟画面平移
        for (int y = 0; y < H; y++)
            for (int x = 0; x < W; x++)
                motion_frame[y * W + x] = (x + i + y * 3) % 256;

        auto ef = motion_enc.encode(motion_frame.data());
        printf("  帧 %2d: %s, size=%zu\n", i + 1,
               ef.is_keyframe ? "I" : "P", ef.data_size);
    }
}

测试结果的意义:

  • 静态场景:I 帧之后全部 P 帧,数据量极低
  • 快速运动场景:运动估计找不到匹配块,残差变大,P 帧大小接近 I 帧甚至更大——这就是编码器面临真实挑战的地方