ffmpeg-api记录
·
ffmpeg-api记录
1. Frame
基于引用计数的frame
av_frame_ref: refav_frame_unref: unrefav_frame_move_ref:所有权转移av_frame_clone: copy一份,引用计数=1
2. Packet
3. AVAudioFifo
音频重采样时,用来缓存aac的帧(1024才能送去编码)
av_audio_fifo_allocav_audio_fifo_size: 大小查询av_audio_fifo_read: 消费fifoav_audio_fifo_write: 写入fifo
4. mux
目标: 根据配置, 给定图片、音频,转码(如果需要)并mux

graph LR
init_muxer_using_config[根据配置初始化mux]
init_muxer_using_origin_context[根据原视频初始化mux]
interact_input_image_audio_saple[交替输入图像/音频] --> convert_if_need[如果需要则进行转码] --> mux
convert_if_need --> video_convert[视频转码逻辑]
video_convert --> init_sws_getContext_using_format_width_height[使用src/dst像素格式、宽、高初始化]
init_sws_getContext_using_format_width_height --> call_sws_scale[调用sws_scale接口得到转码过后的frame]
call_sws_scale --> call_send_frame_and_mux[正常调用send_frame接口进行mux]
convert_if_need --> audio_convert[音频转码设置]
audio_convert --> init_swr_using_swr_alloc_set_opts2[使用src/dst布局、编码格式、采样率初始化 swr_alloc_set_opts2 转换器]
audio_convert --> init_audio_fifo[使用av_audio_fifo_alloc+dst编码格式、通道布局初始化音频fifo队列]
init_swr_using_swr_alloc_set_opts2 --> call_swr_convert[调用swr_convert得到转码过后的frame]
call_swr_convert --> push_fifo[将结果push到audio fifo中]
push_fifo --> consume_fifo[消费fifo如果samples大于1024]
consume_fifo --> call_send_frame_and_mux
4.1 mux.Stream
enum class stream_type {none, video, audio};
struct Stream {
stream_type type;
int index{-1};
AVCodecContext *codec_ctx{nullptr};
// for video
SwsContext *vid_sws_ctx{nullptr};
int64_t vid_next_pts{0};
// for audio
SwrContext *aud_swr_ctx{nullptr};
int64_t handled_samples{0};
AVAudioFifo *fifo{nullptr};
// common frame
AVFrame *frame{nullptr};
};
5. 硬件编解码
需要处理的问题
- 硬件上下文初始化
- 将AVFrame转换到硬件能够接受的输入
- host/device的数据传输
example
struct ret_valid_img {
uint64_t e_value_img{e_valid_image::none};
int score {0};
};
ret_valid_img is_valid_image(const uint8_t *input, uint64_t len) {
if (!g_log_callback_set) {
av_log_set_callback(log_callback);
g_log_callback_set = true;
}
g_ffmpeg_warnings.clear();
auto fmt_ctx = avformat_alloc_context();
if (fmt_ctx == nullptr)
return {};
constexpr uint64_t avio_buf_len = 32768ull;
struct mem_input {
const uint8_t *start{nullptr};
const uint8_t *end{nullptr};
const uint8_t *cur{nullptr};
};
static auto read = [](void *opaque, uint8_t *buf, int buf_size) -> int {
auto &ctx = *reinterpret_cast<mem_input *>(opaque);
auto residual = ctx.end - ctx.cur;
auto sz = std::min<uint64_t>(residual, buf_size);
if (sz > 0) {
// memcpy(buf, ctx.cur, sz);
std::copy(ctx.cur, ctx.cur + sz, buf);
ctx.cur += sz;
}
return sz == 0 ? AVERROR_EOF : sz;
};
static auto seek = [](void *opaque, int64_t offset, int whence) -> int64_t {
auto &ctx = *reinterpret_cast<mem_input *>(opaque);
auto len = ctx.end - ctx.start;
int64_t ret = AVERROR_EOF;
switch (whence) {
case AVSEEK_SIZE:
ret = static_cast<int64_t>(ctx.end - ctx.start);
break;
case SEEK_SET:
if (offset >= 0) {
if (offset >= len) {
ret = AVERROR_EOF;
} else {
ctx.cur = ctx.start + offset;
ret = 0;
}
}
break;
}
return ret;
};
mem_input opaque{input, input + len, input};
static auto free_avio_ctx = [](AVIOContext **avio_ctx){
if((*avio_ctx)->buffer)
av_freep(&((*avio_ctx)->buffer));
avio_context_free(avio_ctx);
};
auto avio_buffer = static_cast<uint8_t*>(av_malloc(avio_buf_len));
auto avio_ctx = avio_alloc_context(avio_buffer,
avio_buf_len, 0, &opaque, read, nullptr, seek);
scope_guard guard_avio_ctx(free_avio_ctx, &avio_ctx);
fmt_ctx->pb = avio_ctx;
if(avformat_open_input(&fmt_ctx, nullptr, nullptr, nullptr) < 0)
return {};
scope_guard guard_input_stream(avformat_close_input, &fmt_ctx);
(void) guard_input_stream;
// only video
if (avformat_find_stream_info(fmt_ctx, nullptr) < 0)
return {};
//
int vid_idx = -1;
for (int i = 0; i < fmt_ctx->nb_streams; ++i) {
if (fmt_ctx->streams[i]->codecpar->codec_type == AVMEDIA_TYPE_VIDEO) {
vid_idx = i;
break;
}
}
if (vid_idx < 0)
return {};
AVCodecContext *codec_ctx = nullptr;
scope_guard guard_codec_ctx(avcodec_free_context, &codec_ctx);
const auto codecpar = fmt_ctx->streams[vid_idx]->codecpar;
auto codec = avcodec_find_decoder(codecpar->codec_id);
if (codec != nullptr) {
codec_ctx = avcodec_alloc_context3(codec);
if (codec_ctx == nullptr)
return {};
if (avcodec_parameters_to_context(codec_ctx, codecpar) < 0)
return {};
if (avcodec_open2(codec_ctx, codec, nullptr) < 0)
return {};
}
int ret = 0;
auto pkt = av_packet_alloc();
scope_guard guard_pkt(av_packet_free, &pkt);
AVFrame *frame = av_frame_alloc();
scope_guard guard_frame(av_frame_free, &frame);
auto handle_packet = [](AVCodecContext *codec_ctx, AVFrame *frame,
auto &&fn) -> void {
int ret = 0;
while (ret >= 0) {
ret = avcodec_receive_frame(codec_ctx, frame);
if (ret == AVERROR(EAGAIN) || ret == AVERROR_EOF) {
av_frame_unref(frame);
break;
} else if (ret < 0) {
av_frame_unref(frame);
return;
}
fn(frame, codec_ctx->frame_number);
av_frame_unref(frame);
}
};
static auto pgm_save = [](unsigned char *buf, int wrap, int xsize, int ysize,
const char *filename) {
FILE *f;
int i;
f = fopen(filename, "wb");
fprintf(f, "P5\n%d %d\n%d\n", xsize, ysize, 255);
for (i = 0; i < ysize; i++)
fwrite(buf + i * wrap, 1, xsize, f);
fclose(f);
};
static auto calc_entropy = [](cv::Mat mat) -> double {
if (mat.channels() > 1) {
cv::cvtColor(mat, mat, cv::COLOR_BGR2GRAY);
}
assert(mat.channels() == 1);
cv::Mat grey = mat.clone();
std::vector<cv::Mat> histChannels;
cv::Mat hist;
int histSize = 256;
float range[] = {0, 256};
const float *histRange = {range};
cv::calcHist(&grey, 1, 0, cv::Mat(), hist, 1, &histSize, &histRange);
hist /= (grey.rows * grey.cols);
double entropy = 0.0;
for (int i = 0; i < histSize; i++) {
float p = hist.at<float>(i);
if (p > 0) {
entropy -= p * log(p);
}
}
return entropy;
};
static auto calc_edge_ratio = [](cv::Mat mat) {
assert(mat.channels() == 1);
auto auto_canny = [](const cv::Mat& gray, double sigma = 0.33) {
cv::Mat blurred = gray.clone();
// cv::GaussianBlur(gray, blurred, cv::Size(3, 3), 0);
cv::Scalar mean_scalar, stddev_scalar;
cv::meanStdDev(blurred, mean_scalar, stddev_scalar);
double mean = mean_scalar[0];
double stddev = stddev_scalar[0];
double lower = std::max<double>(15.0, mean - (1 - sigma) * stddev);
double upper = std::min<double>(255.0, mean + (1 + sigma) * stddev);
if (upper - lower < 10.0) upper = lower + 10.0;
cv::Mat result;
cv::Canny(blurred, result, lower, upper);
return result;
};
auto edges = auto_canny(mat);
return static_cast<double>(cv::countNonZero(edges)) /
static_cast<double>(edges.rows * edges.cols);
// cv::Mat gray = mat.clone();
// cv::Canny(gray, gray, 50, 150);
// return static_cast<double>(cv::countNonZero(gray)) /
// static_cast<double>(gray.rows * gray.cols);
};
static auto calc_std_dev = [](cv::Mat mat) -> double {
assert(mat.channels() == 1);
cv::Scalar mean, stddev;
cv::meanStdDev(mat, mean, stddev);
return stddev[0];
};
static auto calc_ratio = [](cv::Mat mat) -> double {
cv::Mat out;
cv::threshold(mat, out, 128.0, 255.0, cv::THRESH_BINARY);
return static_cast<double>(cv::countNonZero(out)) /
static_cast<double>(out.rows * out.cols);
};
static auto is_valid_image = [](cv::Mat mat) -> bool {
auto entropy = calc_entropy(mat);
auto edge_ratio = calc_edge_ratio(mat);
auto std_dev = calc_std_dev(mat);
return 0.5 <= entropy && entropy <= 7.5 && 0.009 <= edge_ratio &&
edge_ratio <= 0.15 /*&& 19.0 <= std_dev && std_dev <= 100.0 */;
};
bool flag = false;
uint64_t hit_count = 0;
uint64_t total_count = 0;
int max_frame_num = 0;
constexpr uint64_t hit_count_threshold = 8;
auto img_check_cb = [&](AVFrame *frame, int frame_num) {
auto stride = frame->linesize[0];
auto beg = frame->data[0];
uint64_t is_same{ 0 };
for (int i = 1; i < frame->height; ++i) {
auto res = memcmp(beg + (i - 1) * stride, beg + i * stride, frame->width);
is_same += (res == 0);
}
static auto count_substr = [](const std::string& str, const std::string& substr) {
if (substr.empty()) return 0;
int count = 0;
size_t pos = 0;
while ((pos = str.find(substr, pos)) != std::string::npos) {
count++;
pos += substr.length();
}
return count;
};
max_frame_num = std::max<int>(max_frame_num, frame_num);
if (is_same >= frame->height / 5 || is_same >= 30
|| (is_same >= 10 && g_ffmpeg_warnings.find("error while decoding") != std::string::npos)) {
if (!g_ffmpeg_warnings.empty())
return;
}
++hit_count;
// std::string filename = std::string("E:/zzz/") + std::to_string(frame_num) + ".pgm";
// pgm_save(frame->data[0], frame->linesize[0], frame->width, frame->height, filename.c_str());
return;
// call opencv and decode
// cv::Mat mat(frame->height, frame->width, CV_8UC1, frame->data[0]);
// if (is_valid_image(mat)) {
// ++hit_count;
// return;
// }
};
while ((ret = av_read_frame(fmt_ctx, pkt)) >= 0) {
if (pkt->stream_index != vid_idx) {
av_packet_unref(pkt);
continue;
}
int ret = avcodec_send_packet(codec_ctx, pkt);
av_packet_unref(pkt);
if (ret < 0) {
break;
}
if(hit_count >= hit_count_threshold || total_count++ >= hit_count_threshold * 5)
break;
handle_packet(codec_ctx, frame, img_check_cb);
}
ret = avcodec_send_packet(codec_ctx, nullptr);
handle_packet(codec_ctx, frame, img_check_cb);
uint64_t result{0};
if(g_ffmpeg_warnings.empty())
result |= e_valid_image::non_ffmpeg_error;
if (hit_count >= hit_count_threshold
|| ((hit_count < hit_count_threshold) && (static_cast<double>(hit_count) / static_cast<double>(total_count) >= 0.8)))
result |= e_valid_image::ok;
return {result, max_frame_num};
}
更多推荐


所有评论(0)