FFmpeg
Loading...
Searching...
No Matches
dnn_io_proc.c
Go to the documentation of this file.
1/*
2 * Copyright (c) 2020
3 *
4 * This file is part of FFmpeg.
5 *
6 * FFmpeg is free software; you can redistribute it and/or
7 * modify it under the terms of the GNU Lesser General Public
8 * License as published by the Free Software Foundation; either
9 * version 2.1 of the License, or (at your option) any later version.
10 *
11 * FFmpeg is distributed in the hope that it will be useful,
12 * but WITHOUT ANY WARRANTY; without even the implied warranty of
13 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
14 * Lesser General Public License for more details.
15 *
16 * You should have received a copy of the GNU Lesser General Public
17 * License along with FFmpeg; if not, write to the Free Software
18 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
19 */
20
21#include "dnn_io_proc.h"
22#include "libavutil/imgutils.h"
23#include "libavutil/mem.h"
24#include "libswscale/swscale.h"
25#include "libavutil/avassert.h"
27
29{
30 switch (dt)
31 {
32 case DNN_FLOAT:
33 return sizeof(float);
34 case DNN_UINT8:
35 return sizeof(uint8_t);
36 default:
37 av_assert0(!"not supported yet.");
38 return 1;
39 }
40}
41
42int ff_proc_from_dnn_to_frame(AVFrame *frame, DNNData *output, void *log_ctx)
43{
44 struct SwsContext *sws_ctx;
45 int ret = 0;
46 int linesize[4] = { 0 };
47 void **dst_data = NULL;
48 void *middle_data = NULL;
49 uint8_t *planar_data[4] = { 0 };
50 int plane_size = frame->width * frame->height * sizeof(uint8_t);
51 enum AVPixelFormat src_fmt = AV_PIX_FMT_NONE;
52 int src_datatype_size = get_datatype_size(output->dt);
53
54 int bytewidth = av_image_get_linesize(frame->format, frame->width, 0);
55 if (bytewidth < 0) {
56 return AVERROR(EINVAL);
57 }
58 /* scale == 1 and mean == 0 and dt == UINT8: passthrough */
59 if (fabsf(output->scale - 1) < 1e-6f && fabsf(output->mean) < 1e-6 && output->dt == DNN_UINT8)
60 src_fmt = AV_PIX_FMT_GRAY8;
61 /* (scale == 255 or scale == 0) and mean == 0 and dt == FLOAT: normalization */
62 else if ((fabsf(output->scale - 255) < 1e-6f || fabsf(output->scale) < 1e-6f) &&
63 fabsf(output->mean) < 1e-6 && output->dt == DNN_FLOAT)
64 src_fmt = AV_PIX_FMT_GRAYF32;
65 else {
66 av_log(log_ctx, AV_LOG_ERROR, "dnn_process output data doesn't type: UINT8 "
67 "scale: %f, mean: %f\n", output->scale, output->mean);
68 return AVERROR(ENOSYS);
69 }
70
71 dst_data = (void **)frame->data;
72 linesize[0] = frame->linesize[0];
73 if (output->layout == DL_NCHW) {
74 middle_data = av_malloc(plane_size * output->dims[1]);
75 if (!middle_data) {
76 ret = AVERROR(ENOMEM);
77 goto err;
78 }
79 dst_data = &middle_data;
80 linesize[0] = frame->width * 3;
81 }
82
83 switch (frame->format) {
86 sws_ctx = sws_getContext(frame->width * 3,
87 frame->height,
88 src_fmt,
89 frame->width * 3,
90 frame->height,
92 0, NULL, NULL, NULL);
93 if (!sws_ctx) {
94 av_log(log_ctx, AV_LOG_ERROR, "Impossible to create scale context for the conversion "
95 "fmt:%s s:%dx%d -> fmt:%s s:%dx%d\n",
96 av_get_pix_fmt_name(src_fmt), frame->width * 3, frame->height,
97 av_get_pix_fmt_name(AV_PIX_FMT_GRAY8), frame->width * 3, frame->height);
98 ret = AVERROR(EINVAL);
99 goto err;
100 }
101 sws_scale(sws_ctx, (const uint8_t *[4]){(const uint8_t *)output->data, 0, 0, 0},
102 (const int[4]){frame->width * 3 * src_datatype_size, 0, 0, 0}, 0, frame->height,
103 (uint8_t * const*)dst_data, linesize);
104 sws_freeContext(sws_ctx);
105 // convert data from planar to packed
106 if (output->layout == DL_NCHW) {
107 sws_ctx = sws_getContext(frame->width,
108 frame->height,
110 frame->width,
111 frame->height,
112 frame->format,
113 0, NULL, NULL, NULL);
114 if (!sws_ctx) {
115 av_log(log_ctx, AV_LOG_ERROR, "Impossible to create scale context for the conversion "
116 "fmt:%s s:%dx%d -> fmt:%s s:%dx%d\n",
118 av_get_pix_fmt_name(frame->format),frame->width, frame->height);
119 ret = AVERROR(EINVAL);
120 goto err;
121 }
122 if (frame->format == AV_PIX_FMT_RGB24) {
123 planar_data[0] = (uint8_t *)middle_data + plane_size;
124 planar_data[1] = (uint8_t *)middle_data + plane_size * 2;
125 planar_data[2] = (uint8_t *)middle_data;
126 } else if (frame->format == AV_PIX_FMT_BGR24) {
127 planar_data[0] = (uint8_t *)middle_data + plane_size;
128 planar_data[1] = (uint8_t *)middle_data;
129 planar_data[2] = (uint8_t *)middle_data + plane_size * 2;
130 }
131 sws_scale(sws_ctx, (const uint8_t * const *)planar_data,
132 (const int [4]){frame->width * sizeof(uint8_t),
133 frame->width * sizeof(uint8_t),
134 frame->width * sizeof(uint8_t), 0},
135 0, frame->height, frame->data, frame->linesize);
136 sws_freeContext(sws_ctx);
137 }
138 break;
140 av_image_copy_plane(frame->data[0], frame->linesize[0],
141 output->data, bytewidth,
142 bytewidth, frame->height);
143 break;
149 case AV_PIX_FMT_GRAY8:
150 case AV_PIX_FMT_NV12:
151 sws_ctx = sws_getContext(frame->width,
152 frame->height,
154 frame->width,
155 frame->height,
157 0, NULL, NULL, NULL);
158 if (!sws_ctx) {
159 av_log(log_ctx, AV_LOG_ERROR, "Impossible to create scale context for the conversion "
160 "fmt:%s s:%dx%d -> fmt:%s s:%dx%d\n",
161 av_get_pix_fmt_name(src_fmt), frame->width, frame->height,
163 ret = AVERROR(EINVAL);
164 goto err;
165 }
166 sws_scale(sws_ctx, (const uint8_t *[4]){(const uint8_t *)output->data, 0, 0, 0},
167 (const int[4]){frame->width * src_datatype_size, 0, 0, 0}, 0, frame->height,
168 (uint8_t * const*)frame->data, frame->linesize);
169 sws_freeContext(sws_ctx);
170 break;
171 default:
173 ret = AVERROR(ENOSYS);
174 goto err;
175 }
176
177err:
178 av_free(middle_data);
179 return ret;
180}
181
182int ff_proc_from_frame_to_dnn(AVFrame *frame, DNNData *input, void *log_ctx)
183{
184 struct SwsContext *sws_ctx;
185 int ret = 0;
186 int linesize[4] = { 0 };
187 void **src_data = NULL;
188 void *middle_data = NULL;
189 uint8_t *planar_data[4] = { 0 };
190 int plane_size = frame->width * frame->height * sizeof(uint8_t);
191 enum AVPixelFormat dst_fmt = AV_PIX_FMT_NONE;
192 int dst_datatype_size = get_datatype_size(input->dt);
193 int bytewidth = av_image_get_linesize(frame->format, frame->width, 0);
194 if (bytewidth < 0) {
195 return AVERROR(EINVAL);
196 }
197 /* scale == 1 and mean == 0 and dt == UINT8: passthrough */
198 if (fabsf(input->scale - 1) < 1e-6f && fabsf(input->mean) < 1e-6 && input->dt == DNN_UINT8)
199 dst_fmt = AV_PIX_FMT_GRAY8;
200 /* (scale == 255 or scale == 0) and mean == 0 and dt == FLOAT: normalization */
201 else if ((fabsf(input->scale - 255) < 1e-6f || fabsf(input->scale) < 1e-6f) &&
202 fabsf(input->mean) < 1e-6 && input->dt == DNN_FLOAT)
203 dst_fmt = AV_PIX_FMT_GRAYF32;
204 else {
205 av_log(log_ctx, AV_LOG_ERROR, "dnn_process input data doesn't support type: UINT8 "
206 "scale: %f, mean: %f\n", input->scale, input->mean);
207 return AVERROR(ENOSYS);
208 }
209
210 src_data = (void **)frame->data;
211 linesize[0] = frame->linesize[0];
212 if (input->layout == DL_NCHW) {
213 middle_data = av_malloc(plane_size * input->dims[1]);
214 if (!middle_data) {
215 ret = AVERROR(ENOMEM);
216 goto err;
217 }
218 src_data = &middle_data;
219 linesize[0] = frame->width * 3;
220 }
221
222 switch (frame->format) {
223 case AV_PIX_FMT_RGB24:
224 case AV_PIX_FMT_BGR24:
225 // convert data from planar to packed
226 if (input->layout == DL_NCHW) {
227 sws_ctx = sws_getContext(frame->width,
228 frame->height,
229 frame->format,
230 frame->width,
231 frame->height,
233 0, NULL, NULL, NULL);
234 if (!sws_ctx) {
235 av_log(log_ctx, AV_LOG_ERROR, "Impossible to create scale context for the conversion "
236 "fmt:%s s:%dx%d -> fmt:%s s:%dx%d\n",
237 av_get_pix_fmt_name(frame->format), frame->width, frame->height,
239 ret = AVERROR(EINVAL);
240 goto err;
241 }
242 if (frame->format == AV_PIX_FMT_RGB24) {
243 planar_data[0] = (uint8_t *)middle_data + plane_size;
244 planar_data[1] = (uint8_t *)middle_data + plane_size * 2;
245 planar_data[2] = (uint8_t *)middle_data;
246 } else if (frame->format == AV_PIX_FMT_BGR24) {
247 planar_data[0] = (uint8_t *)middle_data + plane_size;
248 planar_data[1] = (uint8_t *)middle_data;
249 planar_data[2] = (uint8_t *)middle_data + plane_size * 2;
250 }
251 sws_scale(sws_ctx, (const uint8_t * const *)frame->data,
252 frame->linesize, 0, frame->height, planar_data,
253 (const int [4]){frame->width * sizeof(uint8_t),
254 frame->width * sizeof(uint8_t),
255 frame->width * sizeof(uint8_t), 0});
256 sws_freeContext(sws_ctx);
257 }
258 sws_ctx = sws_getContext(frame->width * 3,
259 frame->height,
261 frame->width * 3,
262 frame->height,
263 dst_fmt,
264 0, NULL, NULL, NULL);
265 if (!sws_ctx) {
266 av_log(log_ctx, AV_LOG_ERROR, "Impossible to create scale context for the conversion "
267 "fmt:%s s:%dx%d -> fmt:%s s:%dx%d\n",
268 av_get_pix_fmt_name(AV_PIX_FMT_GRAY8), frame->width * 3, frame->height,
269 av_get_pix_fmt_name(dst_fmt),frame->width * 3, frame->height);
270 ret = AVERROR(EINVAL);
271 goto err;
272 }
273 sws_scale(sws_ctx, (const uint8_t **)src_data,
274 linesize, 0, frame->height,
275 (uint8_t * const [4]){input->data, 0, 0, 0},
276 (const int [4]){frame->width * 3 * dst_datatype_size, 0, 0, 0});
277 sws_freeContext(sws_ctx);
278 break;
280 av_image_copy_plane(input->data, bytewidth,
281 frame->data[0], frame->linesize[0],
282 bytewidth, frame->height);
283 break;
289 case AV_PIX_FMT_GRAY8:
290 case AV_PIX_FMT_NV12:
291 sws_ctx = sws_getContext(frame->width,
292 frame->height,
294 frame->width,
295 frame->height,
296 dst_fmt,
297 0, NULL, NULL, NULL);
298 if (!sws_ctx) {
299 av_log(log_ctx, AV_LOG_ERROR, "Impossible to create scale context for the conversion "
300 "fmt:%s s:%dx%d -> fmt:%s s:%dx%d\n",
302 av_get_pix_fmt_name(dst_fmt),frame->width, frame->height);
303 ret = AVERROR(EINVAL);
304 goto err;
305 }
306 sws_scale(sws_ctx, (const uint8_t **)frame->data,
307 frame->linesize, 0, frame->height,
308 (uint8_t * const [4]){input->data, 0, 0, 0},
309 (const int [4]){frame->width * dst_datatype_size, 0, 0, 0});
310 sws_freeContext(sws_ctx);
311 break;
312 default:
314 ret = AVERROR(ENOSYS);
315 goto err;
316 }
317err:
318 av_free(middle_data);
319 return ret;
320}
321
323{
324 if (data->dt == DNN_UINT8) {
325 switch (data->order) {
326 case DCO_BGR:
327 return AV_PIX_FMT_BGR24;
328 case DCO_RGB:
329 return AV_PIX_FMT_RGB24;
330 default:
331 av_assert0(!"unsupported data pixel format.\n");
332 return AV_PIX_FMT_BGR24;
333 }
334 }
335
336 av_assert0(!"unsupported data type.\n");
337 return AV_PIX_FMT_BGR24;
338}
339
340int ff_frame_to_dnn_classify(AVFrame *frame, DNNData *input, uint32_t bbox_index, void *log_ctx)
341{
343 int offsetx[4], offsety[4];
344 uint8_t *bbox_data[4];
345 struct SwsContext *sws_ctx;
346 int linesizes[4];
347 int ret = 0;
348 enum AVPixelFormat fmt;
349 int left, top, width, height;
350 int width_idx, height_idx;
352 const AVDetectionBBox *bbox;
354 int max_step[4] = { 0 };
355 av_assert0(sd);
356
357 /* (scale != 1 and scale != 0) or mean != 0 */
358 if ((fabsf(input->scale - 1) > 1e-6f && fabsf(input->scale) > 1e-6f) ||
359 fabsf(input->mean) > 1e-6f) {
360 av_log(log_ctx, AV_LOG_ERROR, "dnn_classify input data doesn't support "
361 "scale: %f, mean: %f\n", input->scale, input->mean);
362 return AVERROR(ENOSYS);
363 }
364
365 if (input->layout == DL_NCHW) {
366 av_log(log_ctx, AV_LOG_ERROR, "dnn_classify input data doesn't support layout: NCHW\n");
367 return AVERROR(ENOSYS);
368 }
369
370 width_idx = dnn_get_width_idx_by_layout(input->layout);
371 height_idx = dnn_get_height_idx_by_layout(input->layout);
372
373 header = (const AVDetectionBBoxHeader *)sd->data;
374 bbox = av_get_detection_bbox(header, bbox_index);
375
376 left = bbox->x;
377 width = bbox->w;
378 top = bbox->y;
379 height = bbox->h;
380
381 fmt = get_pixel_format(input);
382 sws_ctx = sws_getContext(width, height, frame->format,
383 input->dims[width_idx],
384 input->dims[height_idx], fmt,
386 if (!sws_ctx) {
387 av_log(log_ctx, AV_LOG_ERROR, "Failed to create scale context for the conversion "
388 "fmt:%s s:%dx%d -> fmt:%s s:%dx%d\n",
391 input->dims[width_idx],
392 input->dims[height_idx]);
393 return AVERROR(EINVAL);
394 }
395
396 ret = av_image_fill_linesizes(linesizes, fmt, input->dims[width_idx]);
397 if (ret < 0) {
398 av_log(log_ctx, AV_LOG_ERROR, "unable to get linesizes with av_image_fill_linesizes");
399 sws_freeContext(sws_ctx);
400 return ret;
401 }
402
403 desc = av_pix_fmt_desc_get(frame->format);
404 offsetx[1] = offsetx[2] = AV_CEIL_RSHIFT(left, desc->log2_chroma_w);
405 offsetx[0] = offsetx[3] = left;
406
407 offsety[1] = offsety[2] = AV_CEIL_RSHIFT(top, desc->log2_chroma_h);
408 offsety[0] = offsety[3] = top;
409
411 for (int k = 0; frame->data[k]; k++)
412 bbox_data[k] = frame->data[k] + offsety[k] * frame->linesize[k] + offsetx[k] * max_step[k];
413
414 sws_scale(sws_ctx, (const uint8_t *const *)&bbox_data, frame->linesize,
415 0, height,
416 (uint8_t *const [4]){input->data, 0, 0, 0}, linesizes);
417
418 sws_freeContext(sws_ctx);
419
420 return ret;
421}
422
423/*
424 * Write packed 24bit RGB/BGR samples into the tensor type/layout the model expects.
425 */
426static void detect_write_tensor(DNNData *input, const uint8_t *src,
427 int src_linesize, int w, int h)
428{
429 const size_t plane_size = (size_t)w * h;
430
431 if (input->dt == DNN_FLOAT) {
432 float *dst = input->data;
433 if (input->layout == DL_NCHW) {
434 /* FLOAT + NCHW: deinterleave into planes while widening uint8 -> float. */
435 for (int c = 0; c < 3; c++)
436 for (int y = 0; y < h; y++) {
437 const uint8_t *row = src + (ptrdiff_t)y * src_linesize;
438 for (int x = 0; x < w; x++)
439 dst[c * plane_size + (size_t)y * w + x] = row[x * 3 + c];
440 }
441 } else {
442 /* FLOAT + NHWC : keep it packed, only widen uint8 -> float byte by byte. */
443 for (int y = 0; y < h; y++) {
444 const uint8_t *row = src + (ptrdiff_t)y * src_linesize;
445 for (int x = 0; x < w * 3; x++)
446 dst[(size_t)y * w * 3 + x] = row[x];
447 }
448 }
449 } else {
450 /* UINT8 + NCHW: deinterleave into 3 uint8 planes. */
451 uint8_t *dst = input->data;
452 for (int c = 0; c < 3; c++)
453 for (int y = 0; y < h; y++) {
454 const uint8_t *row = src + (ptrdiff_t)y * src_linesize;
455 for (int x = 0; x < w; x++)
456 dst[c * plane_size + (size_t)y * w + x] = row[x * 3 + c];
457 }
458 }
459}
460
461int ff_frame_to_dnn_detect(AVFrame *frame, DNNData *input, void *log_ctx)
462{
463 struct SwsContext *sws_ctx = NULL;
464 uint8_t *tmp_buf = NULL;
465 const uint8_t *src;
466 int src_linesize;
467 int linesizes[4];
468 int ret = 0, width_idx, height_idx, channel_idx;
469 int packed_u8, w, h;
470 enum AVPixelFormat fmt;
471
472 switch (input->order) {
473 case DCO_BGR: fmt = AV_PIX_FMT_BGR24; break;
474 case DCO_RGB: fmt = AV_PIX_FMT_RGB24; break;
475 default: fmt = AV_PIX_FMT_NONE; break;
476 }
477
478 /* (scale != 1 and scale != 0) or mean != 0 */
479 if ((fabsf(input->scale - 1) > 1e-6f && fabsf(input->scale) > 1e-6f) ||
480 fabsf(input->mean) > 1e-6f) {
481 av_log(log_ctx, AV_LOG_ERROR, "dnn_detect input data doesn't support "
482 "scale: %f, mean: %f\n", input->scale, input->mean);
483 return AVERROR(ENOSYS);
484 }
485
486 if (fmt == AV_PIX_FMT_NONE) {
487 av_log(log_ctx, AV_LOG_ERROR, "dnn_detect input data doesn't support "
488 "color order: %d\n", input->order);
489 return AVERROR(ENOSYS);
490 }
491
492 if (input->dt != DNN_UINT8 && input->dt != DNN_FLOAT) {
493 av_log(log_ctx, AV_LOG_ERROR, "dnn_detect input data doesn't support "
494 "data type: %d\n", input->dt);
495 return AVERROR(ENOSYS);
496 }
497
498 width_idx = dnn_get_width_idx_by_layout(input->layout);
499 height_idx = dnn_get_height_idx_by_layout(input->layout);
500 channel_idx = dnn_get_channel_idx_by_layout(input->layout);
501
502 w = input->dims[width_idx];
503 h = input->dims[height_idx];
504 if (w <= 0 || h <= 0) {
505 av_log(log_ctx, AV_LOG_ERROR, "dnn_detect input data has invalid "
506 "dimensions: %dx%d\n", w, h);
507 return AVERROR(EINVAL);
508 }
509
510 packed_u8 = (input->dt == DNN_UINT8) && (input->layout != DL_NCHW);
511 if (!packed_u8 && input->dims[channel_idx] != 3) {
512 av_log(log_ctx, AV_LOG_ERROR, "dnn_detect requires a 3-channel input, "
513 "but the model has %d channels\n",
514 input->dims[channel_idx]);
515 return AVERROR(ENOSYS);
516 }
517
518 ret = av_image_fill_linesizes(linesizes, fmt, w);
519 if (ret < 0) {
520 av_log(log_ctx, AV_LOG_ERROR, "unable to get linesizes with av_image_fill_linesizes");
521 return ret;
522 }
523
524 if (packed_u8) {
525 /* UINT8 + NHWC: sws_scale() writes straight into input->data in one pass. */
526 sws_ctx = sws_getContext(frame->width, frame->height, frame->format,
527 w, h, fmt, SWS_FAST_BILINEAR, NULL, NULL, NULL);
528 if (!sws_ctx) {
529 av_log(log_ctx, AV_LOG_ERROR, "Impossible to create scale context for the conversion "
530 "fmt:%s s:%dx%d -> fmt:%s s:%dx%d\n",
531 av_get_pix_fmt_name(frame->format), frame->width, frame->height,
532 av_get_pix_fmt_name(fmt), w, h);
533 return AVERROR(EINVAL);
534 }
535
536 sws_scale(sws_ctx, (const uint8_t *const *)frame->data, frame->linesize, 0, frame->height,
537 (uint8_t *const [4]){input->data, 0, 0, 0}, linesizes);
538 sws_freeContext(sws_ctx);
539 return 0;
540 }
541
542 if (frame->format == fmt && frame->width == w && frame->height == h) {
543 src = frame->data[0];
544 src_linesize = frame->linesize[0];
545 } else {
546 size_t tmp_size;
547
548 if (av_size_mult((size_t)linesizes[0], (size_t)h, &tmp_size) < 0) {
549 av_log(log_ctx, AV_LOG_ERROR, "dnn_detect temporary buffer size overflows\n");
550 return AVERROR(EINVAL);
551 }
552
553 tmp_buf = av_malloc(tmp_size);
554 if (!tmp_buf)
555 return AVERROR(ENOMEM);
556
557 sws_ctx = sws_getContext(frame->width, frame->height, frame->format,
558 w, h, fmt, SWS_FAST_BILINEAR, NULL, NULL, NULL);
559 if (!sws_ctx) {
560 av_log(log_ctx, AV_LOG_ERROR, "Impossible to create scale context for the conversion "
561 "fmt:%s s:%dx%d -> fmt:%s s:%dx%d\n",
562 av_get_pix_fmt_name(frame->format), frame->width, frame->height,
563 av_get_pix_fmt_name(fmt), w, h);
564 ret = AVERROR(EINVAL);
565 goto end;
566 }
567
568 sws_scale(sws_ctx, (const uint8_t *const *)frame->data, frame->linesize, 0, frame->height,
569 (uint8_t *const [4]){tmp_buf, 0, 0, 0}, linesizes);
570 sws_freeContext(sws_ctx);
571 sws_ctx = NULL;
572
573 src = tmp_buf;
574 src_linesize = linesizes[0];
575 }
576
577 detect_write_tensor(input, src, src_linesize, w, h);
578
579end:
580 av_freep(&tmp_buf);
581 return ret;
582}
uint8_t ptrdiff_t const uint8_t ptrdiff_t int intptr_t intptr_t int int16_t * dst
Definition dsp.h:87
simple assert() macros that are a bit more flexible than ISO C assert().
#define av_assert0(cond)
assert() equivalent, that is always enabled.
Definition avassert.h:42
static int BS_FUNC left(const BSCTX *bc)
Return the number of the bits left in a buffer.
#define AV_CEIL_RSHIFT(a, b)
Definition common.h:60
#define NULL
Definition coverity.c:32
static __device__ float fabsf(float a)
static AVFrame * frame
static av_always_inline AVDetectionBBox * av_get_detection_bbox(const AVDetectionBBoxHeader *header, unsigned int idx)
static int dnn_get_height_idx_by_layout(DNNLayout layout)
@ DL_NCHW
static int dnn_get_width_idx_by_layout(DNNLayout layout)
DNNDataType
@ DNN_UINT8
@ DNN_FLOAT
@ DCO_RGB
@ DCO_BGR
static int dnn_get_channel_idx_by_layout(DNNLayout layout)
static int get_datatype_size(DNNDataType dt)
Definition dnn_io_proc.c:28
int ff_proc_from_frame_to_dnn(AVFrame *frame, DNNData *input, void *log_ctx)
int ff_frame_to_dnn_detect(AVFrame *frame, DNNData *input, void *log_ctx)
static void detect_write_tensor(DNNData *input, const uint8_t *src, int src_linesize, int w, int h)
int ff_frame_to_dnn_classify(AVFrame *frame, DNNData *input, uint32_t bbox_index, void *log_ctx)
static enum AVPixelFormat get_pixel_format(DNNData *data)
int ff_proc_from_dnn_to_frame(AVFrame *frame, DNNData *output, void *log_ctx)
Definition dnn_io_proc.c:42
DNN input&output process between AVFrame and DNNData.
#define AVERROR(e)
Definition error.h:45
AVFrameSideData * av_frame_get_side_data(const AVFrame *frame, enum AVFrameSideDataType type)
Definition frame.c:659
@ AV_FRAME_DATA_DETECTION_BBOXES
Bounding boxes for object detection and classification, as described by AVDetectionBBoxHeader.
Definition frame.h:194
#define AV_LOG_ERROR
Something went wrong and cannot losslessly be recovered.
Definition log.h:210
int av_size_mult(size_t a, size_t b, size_t *r)
Multiply two size_t values checking for overflow.
Definition mem.c:565
void av_image_copy_plane(uint8_t *dst, int dst_linesize, const uint8_t *src, int src_linesize, int bytewidth, int height)
Copy image plane from src to dst.
Definition imgutils.c:374
int av_image_get_linesize(enum AVPixelFormat pix_fmt, int width, int plane)
Compute the size of an image line with format pix_fmt and width width for the plane plane.
Definition imgutils.c:76
int av_image_fill_linesizes(int linesizes[4], enum AVPixelFormat pix_fmt, int width)
Fill plane linesizes for an image with pixel format pix_fmt and width width.
Definition imgutils.c:89
void av_image_fill_max_pixsteps(int max_pixsteps[4], int max_pixstep_comps[4], const AVPixFmtDescriptor *pixdesc)
Compute the max pixel step for each plane of an image with a format described by pixdesc.
Definition imgutils.c:35
int attribute_align_arg sws_scale(SwsContext *sws, const uint8_t *const srcSlice[], const int srcStride[], int srcSliceY, int srcSliceH, uint8_t *const dst[], const int dstStride[])
swscale wrapper, so we don't need to export the SwsContext.
Definition swscale.c:1626
SwsContext * sws_getContext(int srcW, int srcH, enum AVPixelFormat srcFormat, int dstW, int dstH, enum AVPixelFormat dstFormat, int flags, SwsFilter *srcFilter, SwsFilter *dstFilter, const double *param)
Allocate and return an SwsContext.
Definition utils.c:1922
void sws_freeContext(SwsContext *swsContext)
Free the swscaler context swsContext.
Definition utils.c:2253
@ SWS_FAST_BILINEAR
Scaler selection options.
Definition swscale.h:197
misc image utilities
void avpriv_report_missing_feature(void *avc, const char *msg,...) av_printf_format(2
Log a generic warning message about a missing feature.
const char * desc
Definition libsvtav1.c:83
uint8_t w
Definition llvidencdsp.c:39
Memory handling functions.
const char data[16]
Definition mxf.c:149
#define av_malloc(s)
Definition ops_static.c:52
const char * av_get_pix_fmt_name(enum AVPixelFormat pix_fmt)
Return the short name for a pixel format, NULL in case pix_fmt is unknown.
Definition pixdesc.c:3380
const AVPixFmtDescriptor * av_pix_fmt_desc_get(enum AVPixelFormat pix_fmt)
Definition pixdesc.c:3460
#define AV_PIX_FMT_GRAYF32
Definition pixfmt.h:588
AVPixelFormat
Pixel format.
Definition pixfmt.h:71
@ AV_PIX_FMT_NV12
planar YUV 4:2:0, 12bpp, 1 plane for Y and 1 plane for the UV components, which are interleaved (firs...
Definition pixfmt.h:96
@ AV_PIX_FMT_NONE
Definition pixfmt.h:72
@ AV_PIX_FMT_RGB24
packed RGB 8:8:8, 24bpp, RGBRGB...
Definition pixfmt.h:75
@ AV_PIX_FMT_YUV420P
planar YUV 4:2:0, 12bpp, (1 Cr & Cb sample per 2x2 Y samples)
Definition pixfmt.h:73
@ AV_PIX_FMT_YUV422P
planar YUV 4:2:2, 16bpp, (1 Cr & Cb sample per 2x1 Y samples)
Definition pixfmt.h:77
@ AV_PIX_FMT_GRAY8
Y , 8bpp.
Definition pixfmt.h:81
@ AV_PIX_FMT_YUV410P
planar YUV 4:1:0, 9bpp, (1 Cr & Cb sample per 4x4 Y samples)
Definition pixfmt.h:79
@ AV_PIX_FMT_YUV411P
planar YUV 4:1:1, 12bpp, (1 Cr & Cb sample per 4x1 Y samples)
Definition pixfmt.h:80
@ AV_PIX_FMT_YUV444P
planar YUV 4:4:4, 24bpp, (1 Cr & Cb sample per 1x1 Y samples)
Definition pixfmt.h:78
@ AV_PIX_FMT_BGR24
packed RGB 8:8:8, 24bpp, BGRBGR...
Definition pixfmt.h:76
@ AV_PIX_FMT_GBRP
planar GBR 4:4:4 24bpp
Definition pixfmt.h:165
static const uint8_t header[24]
Definition sdr2.c:68
int x
Distance in pixels from the left/top edge of the frame, together with width and height,...
Structure to hold side data for an AVFrame.
Definition frame.h:327
uint8_t * data
Definition frame.h:329
This structure describes decoded (raw) audio or video data.
Definition frame.h:472
Descriptor that unambiguously describes how the bits of a pixel are stored in the up to 4 data planes...
Definition pixdesc.h:69
float scale
DNNDataType dt
int dims[4]
DNNColorOrder order
void * data
DNNLayout layout
float mean
Main external API structure.
Definition swscale.h:227
external API header
#define av_free(p)
#define av_freep(p)
#define av_log(a,...)
#define src
Definition vp8dsp.c:248
#define height
Definition dsp.h:89
#define width
Definition dsp.h:89
static double c[64]