FFmpeg
Loading...
Searching...
No Matches
vf_dnn_classify.c
Go to the documentation of this file.
1/*
2 * This file is part of FFmpeg.
3 *
4 * FFmpeg is free software; you can redistribute it and/or
5 * modify it under the terms of the GNU Lesser General Public
6 * License as published by the Free Software Foundation; either
7 * version 2.1 of the License, or (at your option) any later version.
8 *
9 * FFmpeg is distributed in the hope that it will be useful,
10 * but WITHOUT ANY WARRANTY; without even the implied warranty of
11 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
12 * Lesser General Public License for more details.
13 *
14 * You should have received a copy of the GNU Lesser General Public
15 * License along with FFmpeg; if not, write to the Free Software
16 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
17 */
18
19/**
20 * @file
21 * implementing an classification filter using deep learning networks.
22 */
23
24#include "libavutil/file_open.h"
25#include "libavutil/mem.h"
26#include "libavutil/opt.h"
27#include "filters.h"
28#include "dnn_filter_common.h"
29#include "video.h"
30#include "libavutil/time.h"
31#include "libavutil/avstring.h"
33
43
44#define OFFSET(x) offsetof(DnnClassifyContext, dnnctx.x)
45#define OFFSET2(x) offsetof(DnnClassifyContext, x)
46#define FLAGS AV_OPT_FLAG_FILTERING_PARAM | AV_OPT_FLAG_VIDEO_PARAM
48 { "dnn_backend", "DNN backend", OFFSET(backend_type), AV_OPT_TYPE_INT, { .i64 = DNN_OV }, INT_MIN, INT_MAX, FLAGS, .unit = "backend" },
49#if (CONFIG_LIBOPENVINO == 1)
50 { "openvino", "openvino backend flag", 0, AV_OPT_TYPE_CONST, { .i64 = DNN_OV }, 0, 0, FLAGS, .unit = "backend" },
51#endif
52#if (CONFIG_LIBONNXRUNTIME == 1)
53 { "onnx", "onnx backend flag", 0, AV_OPT_TYPE_CONST, { .i64 = DNN_ONNX }, 0, 0, FLAGS, .unit = "backend" },
54#endif
55 { "confidence", "threshold of confidence", OFFSET2(confidence), AV_OPT_TYPE_FLOAT, { .dbl = 0.5 }, 0, 1, FLAGS},
56 { "labels", "path to labels file", OFFSET2(labels_filename), AV_OPT_TYPE_STRING, { .str = NULL }, 0, 0, FLAGS },
57 { "target", "which one to be classified", OFFSET2(target), AV_OPT_TYPE_STRING, { .str = NULL }, 0, 0, FLAGS },
58 { NULL }
59};
60
62
63static int dnn_classify_post_proc(AVFrame *frame, DNNData *output, uint32_t bbox_index, AVFilterContext *filter_ctx)
64{
66 float conf_threshold = ctx->confidence;
68 AVDetectionBBox *bbox;
69 float *classifications;
70 uint32_t label_id;
71 float confidence;
73 int output_size = output->dims[3] * output->dims[2] * output->dims[1];
74 if (output_size <= 0) {
75 return -1;
76 }
77
79 if (!sd) {
80 av_log(filter_ctx, AV_LOG_ERROR, "Cannot get side data in dnn_classify_post_proc\n");
81 return -1;
82 }
84
85 if (bbox_index == 0) {
86 av_strlcat(header->source, ", ", sizeof(header->source));
87 av_strlcat(header->source, ctx->dnnctx.model_filename, sizeof(header->source));
88 }
89
90 classifications = output->data;
91 label_id = 0;
92 confidence= classifications[0];
93 for (int i = 1; i < output_size; i++) {
94 if (classifications[i] > confidence) {
95 label_id = i;
96 confidence= classifications[i];
97 }
98 }
99
100 if (confidence < conf_threshold) {
101 return 0;
102 }
103
104 bbox = av_get_detection_bbox(header, bbox_index);
105 bbox->classify_confidences[bbox->classify_count] = av_make_q((int)(confidence * 10000), 10000);
106
107 if (ctx->labels && label_id < ctx->label_count) {
108 av_strlcpy(bbox->classify_labels[bbox->classify_count], ctx->labels[label_id], sizeof(bbox->classify_labels[bbox->classify_count]));
109 } else {
110 snprintf(bbox->classify_labels[bbox->classify_count], sizeof(bbox->classify_labels[bbox->classify_count]), "%d", label_id);
111 }
112
113 bbox->classify_count++;
114
115 return 0;
116}
117
119{
120 for (int i = 0; i < ctx->label_count; i++) {
121 av_freep(&ctx->labels[i]);
122 }
123 ctx->label_count = 0;
124 av_freep(&ctx->labels);
125}
126
128{
129 int line_len;
130 FILE *file;
131 DnnClassifyContext *ctx = context->priv;
132
133 file = avpriv_fopen_utf8(ctx->labels_filename, "r");
134 if (!file){
135 av_log(context, AV_LOG_ERROR, "failed to open file %s\n", ctx->labels_filename);
136 return AVERROR(EINVAL);
137 }
138
139 while (!feof(file)) {
140 char *label;
141 char buf[256];
142 if (!fgets(buf, 256, file)) {
143 break;
144 }
145
146 line_len = strlen(buf);
147 while (line_len) {
148 int i = line_len - 1;
149 if (buf[i] == '\n' || buf[i] == '\r' || buf[i] == ' ') {
150 buf[i] = '\0';
151 line_len--;
152 } else {
153 break;
154 }
155 }
156
157 if (line_len == 0) // empty line
158 continue;
159
161 av_log(context, AV_LOG_ERROR, "label %s too long\n", buf);
162 fclose(file);
163 return AVERROR(EINVAL);
164 }
165
166 label = av_strdup(buf);
167 if (!label) {
168 av_log(context, AV_LOG_ERROR, "failed to allocate memory for label %s\n", buf);
169 fclose(file);
170 return AVERROR(ENOMEM);
171 }
172
173 if (av_dynarray_add_nofree(&ctx->labels, &ctx->label_count, label) < 0) {
174 av_log(context, AV_LOG_ERROR, "failed to do av_dynarray_add\n");
175 fclose(file);
176 av_freep(&label);
177 return AVERROR(ENOMEM);
178 }
179 }
180
181 fclose(file);
182 return 0;
183}
184
186{
187 DnnClassifyContext *ctx = context->priv;
188 int ret = ff_dnn_init(&ctx->dnnctx, DFT_ANALYTICS_CLASSIFY, context);
189 if (ret < 0)
190 return ret;
192
193 if (ctx->labels_filename) {
194 return read_classify_label_file(context);
195 }
196 return 0;
197}
198
207
209{
210 DnnClassifyContext *ctx = outlink->src->priv;
211 int ret;
212 DNNAsyncStatusType async_state;
213
214 ret = ff_dnn_flush(&ctx->dnnctx);
215 if (ret != 0) {
216 return -1;
217 }
218
219 do {
220 AVFrame *in_frame = NULL;
221 AVFrame *out_frame = NULL;
222 async_state = ff_dnn_get_result(&ctx->dnnctx, &in_frame, &out_frame);
223 if (async_state == DAST_SUCCESS) {
224 ret = ff_filter_frame(outlink, in_frame);
225 if (ret < 0)
226 return ret;
227 if (out_pts)
228 *out_pts = in_frame->pts + pts;
229 }
230 av_usleep(5000);
231 } while (async_state >= DAST_NOT_READY);
232
233 return 0;
234}
235
237{
238 AVFilterLink *inlink = filter_ctx->inputs[0];
239 AVFilterLink *outlink = filter_ctx->outputs[0];
241 AVFrame *in = NULL;
242 int64_t pts;
243 int ret, status;
244 int got_frame = 0;
245 int async_state;
246
247 FF_FILTER_FORWARD_STATUS_BACK(outlink, inlink);
248
249 do {
250 // drain all input frames
251 ret = ff_inlink_consume_frame(inlink, &in);
252 if (ret < 0)
253 return ret;
254 if (ret > 0) {
255 if (ff_dnn_execute_model_classification(&ctx->dnnctx, in, NULL, ctx->target) != 0) {
256 return AVERROR(EIO);
257 }
258 }
259 } while (ret > 0);
260
261 // drain all processed frames
262 do {
263 AVFrame *in_frame = NULL;
264 AVFrame *out_frame = NULL;
265 async_state = ff_dnn_get_result(&ctx->dnnctx, &in_frame, &out_frame);
266 if (async_state == DAST_SUCCESS) {
267 ret = ff_filter_frame(outlink, in_frame);
268 if (ret < 0)
269 return ret;
270 got_frame = 1;
271 }
272 } while (async_state == DAST_SUCCESS);
273
274 // if frame got, schedule to next filter
275 if (got_frame)
276 return 0;
277
278 if (ff_inlink_acknowledge_status(inlink, &status, &pts)) {
279 if (status == AVERROR_EOF) {
280 int64_t out_pts = pts;
281 ret = dnn_classify_flush_frame(outlink, pts, &out_pts);
282 ff_outlink_set_status(outlink, status, out_pts);
283 return ret;
284 }
285 }
286
287 FF_FILTER_FORWARD_WANTED(outlink, inlink);
288
289 return 0;
290}
291
293{
294 DnnClassifyContext *ctx = context->priv;
295 ff_dnn_uninit(&ctx->dnnctx);
297}
298
300 .p.name = "dnn_classify",
301 .p.description = NULL_IF_CONFIG_SMALL("Apply DNN classify filter to the input."),
302 .p.priv_class = &dnn_classify_class,
303 .priv_size = sizeof(DnnClassifyContext),
310 .activate = dnn_classify_activate,
311};
const FFFilter ff_vf_dnn_classify
static AVFormatContext * ctx
int ff_inlink_acknowledge_status(AVFilterLink *link, int *rstatus, int64_t *rpts)
Test and acknowledge the change of status on the link.
Definition avfilter.c:1472
int ff_filter_frame(AVFilterLink *link, AVFrame *frame)
Send a frame of data to the next filter.
Definition avfilter.c:1073
int ff_inlink_consume_frame(AVFilterLink *link, AVFrame **rframe)
Take a frame from the link's FIFO and update the link's stats.
Definition avfilter.c:1525
#define i(width, name, range_min, range_max)
Definition cbs_h264.c:63
#define FLAGS
Definition cmdutils.c:596
#define NULL
Definition coverity.c:32
long long int64_t
Definition coverity.c:34
static AVFrame * frame
#define AV_DETECTION_BBOX_LABEL_NAME_MAX_SIZE
static av_always_inline AVDetectionBBox * av_get_detection_bbox(const AVDetectionBBoxHeader *header, unsigned int idx)
void ff_dnn_uninit(DnnContext *ctx)
DNNAsyncStatusType ff_dnn_get_result(DnnContext *ctx, AVFrame **in_frame, AVFrame **out_frame)
int ff_dnn_set_classify_post_proc(DnnContext *ctx, ClassifyPostProc post_proc)
int ff_dnn_init(DnnContext *ctx, DNNFunctionType func_type, AVFilterContext *filter_ctx)
int ff_dnn_flush(DnnContext *ctx)
int ff_dnn_filter_init_child_class(AVFilterContext *filter)
int ff_dnn_execute_model_classification(DnnContext *ctx, AVFrame *in_frame, AVFrame *out_frame, const char *target)
common functions for the dnn based filters
#define AVFILTER_DNN_DEFINE_CLASS(fname, backend_mask)
DNNAsyncStatusType
@ DAST_NOT_READY
@ DAST_SUCCESS
@ DNN_OV
@ DNN_ONNX
@ DFT_ANALYTICS_CLASSIFY
int(* init)(AVBSFContext *ctx)
Definition dts2pts.c:608
const char * label
Definition ffplay.c:1187
@ AV_OPT_TYPE_CONST
Special option type for declaring named constants.
Definition opt.h:298
@ AV_OPT_TYPE_INT
Underlying C type is int.
Definition opt.h:258
@ AV_OPT_TYPE_FLOAT
Underlying C type is float.
Definition opt.h:270
@ AV_OPT_TYPE_STRING
Underlying C type is a uint8_t* that is either NULL or points to a C string allocated with the av_mal...
Definition opt.h:275
#define AVERROR_EOF
End of file.
Definition error.h:57
#define AVERROR(e)
Definition error.h:45
AVFrameSideData * av_frame_get_side_data(const AVFrame *frame, enum AVFrameSideDataType type)
Definition frame.c:659
@ AV_FRAME_DATA_DETECTION_BBOXES
Bounding boxes for object detection and classification, as described by AVDetectionBBoxHeader.
Definition frame.h:194
#define AV_LOG_ERROR
Something went wrong and cannot losslessly be recovered.
Definition log.h:210
static AVRational av_make_q(int num, int den)
Create an AVRational.
Definition rational.h:71
int av_dynarray_add_nofree(void *tab_ptr, int *nb_ptr, void *elem)
Add an element to a dynamic array.
Definition mem.c:419
size_t av_strlcat(char *dst, const char *src, size_t size)
Append the string src to the string dst, but to a total length of no more than size - 1 bytes,...
Definition avstring.c:95
size_t av_strlcpy(char *dst, const char *src, size_t size)
Copy the string src to dst, but no more than size - 1 bytes, and null-terminate dst.
Definition avstring.c:85
static av_cold void uninit(AVBitStreamFilterContext *ctx)
#define FILTER_INPUTS(array)
Definition filters.h:264
#define FILTER_OUTPUTS(array)
Definition filters.h:265
#define FF_FILTER_FORWARD_WANTED(outlink, inlink)
Forward the frame_wanted_out flag from an output link to an input link.
Definition filters.h:694
#define FILTER_PIXFMTS_ARRAY(array)
Definition filters.h:244
static void ff_outlink_set_status(AVFilterLink *link, int status, int64_t pts)
Set the status field of a link from the source filter.
Definition filters.h:629
#define FF_FILTER_FORWARD_STATUS_BACK(outlink, inlink)
Forward the status on an output link to an input link.
Definition filters.h:639
#define av_cold
Definition attributes.h:117
FILE * avpriv_fopen_utf8(const char *path, const char *mode)
Open a file using a UTF-8 filename.
Definition file_open.c:160
#define NULL_IF_CONFIG_SMALL(x)
Return NULL if CONFIG_SMALL is true, otherwise the argument without modification.
Definition internal.h:97
static enum AVPixelFormat pix_fmts[]
Definition libkvazaar.c:296
Memory handling functions.
static size_t line_len
Definition mscl.c:66
#define av_strdup(s)
Definition ops_static.c:55
AVOptions.
#define AV_PIX_FMT_GRAYF32
Definition pixfmt.h:588
AVPixelFormat
Pixel format.
Definition pixfmt.h:71
@ AV_PIX_FMT_NV12
planar YUV 4:2:0, 12bpp, 1 plane for Y and 1 plane for the UV components, which are interleaved (firs...
Definition pixfmt.h:96
@ AV_PIX_FMT_NONE
Definition pixfmt.h:72
@ AV_PIX_FMT_RGB24
packed RGB 8:8:8, 24bpp, RGBRGB...
Definition pixfmt.h:75
@ AV_PIX_FMT_YUV420P
planar YUV 4:2:0, 12bpp, (1 Cr & Cb sample per 2x2 Y samples)
Definition pixfmt.h:73
@ AV_PIX_FMT_YUV422P
planar YUV 4:2:2, 16bpp, (1 Cr & Cb sample per 2x1 Y samples)
Definition pixfmt.h:77
@ AV_PIX_FMT_GRAY8
Y , 8bpp.
Definition pixfmt.h:81
@ AV_PIX_FMT_YUV410P
planar YUV 4:1:0, 9bpp, (1 Cr & Cb sample per 4x4 Y samples)
Definition pixfmt.h:79
@ AV_PIX_FMT_YUV411P
planar YUV 4:1:1, 12bpp, (1 Cr & Cb sample per 4x1 Y samples)
Definition pixfmt.h:80
@ AV_PIX_FMT_YUV444P
planar YUV 4:4:4, 24bpp, (1 Cr & Cb sample per 1x1 Y samples)
Definition pixfmt.h:78
@ AV_PIX_FMT_BGR24
packed RGB 8:8:8, 24bpp, BGRBGR...
Definition pixfmt.h:76
static const uint8_t header[24]
Definition sdr2.c:68
static av_cold int preinit(AVBitStreamFilterContext *ctx)
Definition source.c:137
Describe the class of an AVClass context structure.
Definition log.h:76
char classify_labels[AV_NUM_DETECTION_BBOX_CLASSIFY][AV_DETECTION_BBOX_LABEL_NAME_MAX_SIZE]
AVRational classify_confidences[AV_NUM_DETECTION_BBOX_CLASSIFY]
uint32_t classify_count
An instance of a filter.
Definition avfilter.h:279
void * priv
private data for use by the filter
Definition avfilter.h:294
Structure to hold side data for an AVFrame.
Definition frame.h:334
uint8_t * data
Definition frame.h:336
This structure describes decoded (raw) audio or video data.
Definition frame.h:479
int64_t pts
Presentation timestamp in time_base units (time when frame should be shown to user).
Definition frame.h:581
AVOption.
Definition opt.h:428
int dims[4]
void * data
#define av_freep(p)
#define av_log(a,...)
int av_usleep(unsigned usec)
Sleep for a period of time.
Definition time.c:93
static FilteringContext * filter_ctx
Definition transcode.c:52
static int64_t pts
static int read_classify_label_file(AVFilterContext *context)
static av_cold void dnn_classify_uninit(AVFilterContext *context)
static void free_classify_labels(DnnClassifyContext *ctx)
static int dnn_classify_activate(AVFilterContext *filter_ctx)
static const AVOption dnn_classify_options[]
#define OFFSET(x)
static int dnn_classify_flush_frame(AVFilterLink *outlink, int64_t pts, int64_t *out_pts)
#define OFFSET2(x)
static av_cold int dnn_classify_init(AVFilterContext *context)
static int dnn_classify_post_proc(AVFrame *frame, DNNData *output, uint32_t bbox_index, AVFilterContext *filter_ctx)
const AVFilterPad ff_video_default_filterpad[1]
An AVFilterPad array whose only entry has name "default" and is of type AVMEDIA_TYPE_VIDEO.
Definition video.c:37