FFmpeg
Loading...
Searching...
No Matches
dnn_backend_onnx.c
Go to the documentation of this file.
1/*
2 * Copyright (c) 2026 Advanced Micro Devices, Inc.
3 *
4 * This file is part of FFmpeg.
5 *
6 * FFmpeg is free software; you can redistribute it and/or
7 * modify it under the terms of the GNU Lesser General Public
8 * License as published by the Free Software Foundation; either
9 * version 2.1 of the License, or (at your option) any later version.
10 *
11 * FFmpeg is distributed in the hope that it will be useful,
12 * but WITHOUT ANY WARRANTY; without even the implied warranty of
13 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
14 * Lesser General Public License for more details.
15 *
16 * You should have received a copy of the GNU Lesser General Public
17 * License along with FFmpeg; if not, write to the Free Software
18 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
19 */
20
21/**
22 * @file
23 * DNN ONNX Runtime backend implementation.
24 */
25
26#include "libavutil/opt.h"
27#include "libavutil/avassert.h"
29#include "libavutil/imgutils.h"
30#include "libavutil/mem.h"
31#include "libavutil/avstring.h"
32#include "libavutil/thread.h"
34#include "../filters.h"
35#include "dnn_io_proc.h"
36#include "dnn_backend_common.h"
37#include "queue.h"
38#include "safe_queue.h"
39#include <onnxruntime_c_api.h>
40#include <inttypes.h>
41#include <stdio.h>
42#include <string.h>
43
58
59typedef struct ONNXInferRequest {
60 OrtValue *input_tensor;
61 OrtValue **output_tensors;
62 uint32_t nb_outputs;
65
71
72#define OFFSET(x) offsetof(ONNXOptions, x)
73#define FLAGS AV_OPT_FLAG_FILTERING_PARAM
74static const AVOption dnn_onnx_options[] = {
75 { "threads_per_operation", "number of CPU threads per ORT operator (device=cpu only)",
76 OFFSET(num_threads), AV_OPT_TYPE_INT, { .i64 = 0 }, 0, INT_MAX, FLAGS },
77 { NULL }
78};
79
81
82static const OrtApi *g_ort = NULL;
84
85static void init_ort_api(void)
86{
87 g_ort = OrtGetApiBase()->GetApi(ORT_API_VERSION);
88}
89
90#define ORT_ABORT_ON_ERROR(expr) \
91 do { \
92 OrtStatus *status = (expr); \
93 if (status != NULL) { \
94 const char *msg = g_ort->GetErrorMessage(status); \
95 av_log(ctx, AV_LOG_ERROR, "ONNX Runtime error: %s\n", msg); \
96 g_ort->ReleaseStatus(status); \
97 goto err; \
98 } \
99 } while (0)
100
102{
103 AVFrameSideData *sd;
105 const AVDetectionBBox *bbox;
106
108 if (!sd)
109 return 0;
110
111 if (!sd->size)
112 return 0;
113
114 header = (const AVDetectionBBoxHeader *)sd->data;
115 if (!header->nb_bboxes)
116 return 0;
117
118 for (uint32_t i = 0; i < header->nb_bboxes; i++) {
120 if (bbox->x < 0 || bbox->w < 0 || bbox->x + bbox->w > frame->width)
121 return 0;
122 if (bbox->y < 0 || bbox->h < 0 || bbox->y + bbox->h > frame->height)
123 return 0;
125 return 0;
126 }
127
128 return 1;
129}
130
132 Queue *lltask_queue, DNNExecBaseParams *exec_params)
133{
134 ONNXModel *onnx_model = (ONNXModel *)task->model;
135 DnnContext *ctx = onnx_model->ctx;
136
137 switch (func_type) {
140 {
141 LastLevelTaskItem *lltask = av_malloc(sizeof(*lltask));
142 if (!lltask) {
143 av_log(ctx, AV_LOG_ERROR, "Failed to allocate memory for LastLevelTaskItem\n");
144 return AVERROR(ENOMEM);
145 }
146 task->inference_todo = 1;
147 task->inference_done = 0;
148 lltask->task = task;
149 if (ff_queue_push_back(lltask_queue, lltask) < 0) {
150 av_log(ctx, AV_LOG_ERROR, "Failed to push back lltask_queue.\n");
151 av_freep(&lltask);
152 return AVERROR(ENOMEM);
153 }
154 return 0;
155 }
157 {
159 AVFrame *frame = task->in_frame;
160 AVFrameSideData *sd;
162
163 task->inference_todo = 0;
164 task->inference_done = 0;
165
167 return 0;
168
170 header = (const AVDetectionBBoxHeader *)sd->data;
171
172 for (uint32_t i = 0; i < header->nb_bboxes; i++) {
173 LastLevelTaskItem *lltask;
175
176 if (params->target) {
177 if (av_strncasecmp(bbox->detect_label, params->target, sizeof(bbox->detect_label)) != 0)
178 continue;
179 }
180
181 lltask = av_malloc(sizeof(*lltask));
182 if (!lltask) {
183 av_log(ctx, AV_LOG_ERROR, "Failed to allocate memory for LastLevelTaskItem\n");
184 return AVERROR(ENOMEM);
185 }
186 task->inference_todo++;
187 lltask->task = task;
188 lltask->bbox_index = i;
189 if (ff_queue_push_back(lltask_queue, lltask) < 0) {
190 av_log(ctx, AV_LOG_ERROR, "Failed to push back lltask_queue.\n");
191 av_freep(&lltask);
192 return AVERROR(ENOMEM);
193 }
194 }
195 return 0;
196 }
197 default:
198 avpriv_report_missing_feature(ctx, "model function type %d", func_type);
199 return AVERROR(ENOSYS);
200 }
201}
202
204{
205 if (!request)
206 return;
207 if (request->input_tensor) {
208 g_ort->ReleaseValue(request->input_tensor);
209 request->input_tensor = NULL;
210 }
211 av_freep(&request->input_data);
212 if (request->output_tensors) {
213 for (uint32_t i = 0; i < request->nb_outputs; i++) {
214 if (request->output_tensors[i]) {
215 g_ort->ReleaseValue(request->output_tensors[i]);
216 request->output_tensors[i] = NULL;
217 }
218 }
219 av_freep(&request->output_tensors);
220 }
221 request->nb_outputs = 0;
222}
223
225{
226 ONNXRequestItem *item;
227 if (!arg || !*arg)
228 return;
229 item = *arg;
231 av_freep(&item->infer_request);
232 av_freep(&item->lltask);
234 av_freep(arg);
235}
236
237static void dnn_free_model_onnx(DNNModel **model)
238{
239 ONNXModel *onnx_model;
240 if (!model || !*model)
241 return;
242
243 onnx_model = (ONNXModel *)(*model);
244
245 ff_dnn_wait_requests(onnx_model->request_queue, onnx_model->ctx->nireq);
246 while (ff_safe_queue_size(onnx_model->request_queue) != 0) {
249 }
251
252 while (ff_queue_size(onnx_model->lltask_queue) != 0) {
254 av_freep(&item);
255 }
256 ff_queue_destroy(onnx_model->lltask_queue);
257
258 while (ff_queue_size(onnx_model->task_queue) != 0) {
259 TaskItem *item = (TaskItem *)ff_queue_pop_front(onnx_model->task_queue);
260 av_frame_free(&item->in_frame);
261 av_frame_free(&item->out_frame);
262 av_freep(&item);
263 }
264 ff_queue_destroy(onnx_model->task_queue);
265
266 if (onnx_model->session)
267 g_ort->ReleaseSession(onnx_model->session);
268 if (onnx_model->session_options)
269 g_ort->ReleaseSessionOptions(onnx_model->session_options);
270 if (onnx_model->env)
271 g_ort->ReleaseEnv(onnx_model->env);
272
273 av_freep(&onnx_model);
274 *model = NULL;
275}
276
277static int get_input_onnx(DNNModel *model, DNNData *input, const char *input_name)
278{
279 ONNXModel *onnx_model = (ONNXModel *)model;
280 DnnContext *ctx = onnx_model->ctx;
281 OrtTypeInfo *type_info = NULL;
282 const OrtTensorTypeAndShapeInfo *tensor_info = NULL;
283 size_t num_dims;
284 size_t input_count = 0;
285 size_t input_index = 0;
286 int found_input = 0;
287 int64_t *dims;
288 ONNXTensorElementDataType tensor_type;
289 OrtStatus *status;
290
291 if (!input_name || !*input_name) {
292 av_log(ctx, AV_LOG_ERROR, "ONNX input name is not specified\n");
293 return AVERROR(EINVAL);
294 }
295
296 if (onnx_model->input_resolved) {
297 *input = onnx_model->input_info;
298 return 0;
299 }
300
301 status = g_ort->SessionGetInputCount(onnx_model->session, &input_count);
302 if (status != NULL) {
303 const char *msg = g_ort->GetErrorMessage(status);
304 av_log(ctx, AV_LOG_ERROR, "Failed to get input count: %s\n", msg);
305 g_ort->ReleaseStatus(status);
306 return AVERROR(EINVAL);
307 }
308
309 for (size_t i = 0; i < input_count; i++) {
310 char *name = NULL;
311 status = g_ort->SessionGetInputName(onnx_model->session, i,
312 onnx_model->allocator, &name);
313 if (status != NULL) {
314 g_ort->ReleaseStatus(status);
315 continue;
316 }
317 if (!strcmp(name, input_name)) {
318 input_index = i;
319 found_input = 1;
320 }
321 onnx_model->allocator->Free(onnx_model->allocator, name);
322 if (found_input)
323 break;
324 }
325
326 if (!found_input) {
327 av_log(ctx, AV_LOG_ERROR, "Input name '%s' not found in ONNX model\n",
328 input_name);
329 return AVERROR(EINVAL);
330 }
331
332 status = g_ort->SessionGetInputTypeInfo(onnx_model->session, input_index,
333 &type_info);
334 if (status != NULL) {
335 const char *msg = g_ort->GetErrorMessage(status);
336 av_log(ctx, AV_LOG_ERROR, "Failed to get input type info: %s\n", msg);
337 g_ort->ReleaseStatus(status);
338 return AVERROR(EINVAL);
339 }
340
341 status = g_ort->CastTypeInfoToTensorInfo(type_info, &tensor_info);
342 if (status != NULL) {
343 g_ort->ReleaseTypeInfo(type_info);
344 g_ort->ReleaseStatus(status);
345 return AVERROR(EINVAL);
346 }
347
348 status = g_ort->GetDimensionsCount(tensor_info, &num_dims);
349 if (status != NULL) {
350 g_ort->ReleaseTypeInfo(type_info);
351 g_ort->ReleaseStatus(status);
352 return AVERROR(EINVAL);
353 }
354
355 if (num_dims != 4) {
356 avpriv_report_missing_feature(ctx, "Support for %zu dimensional input", num_dims);
357 g_ort->ReleaseTypeInfo(type_info);
358 return AVERROR(ENOSYS);
359 }
360
361 dims = av_malloc(num_dims * sizeof(int64_t));
362 if (!dims) {
363 g_ort->ReleaseTypeInfo(type_info);
364 return AVERROR(ENOMEM);
365 }
366
367 g_ort->GetDimensions(tensor_info, dims, num_dims);
368 g_ort->GetTensorElementType(tensor_info, &tensor_type);
369
370 if (dims[0] > 1) {
372 "ONNX model has fixed batch size %"PRId64", but the backend "
373 "only supports a batch size of 1\n", dims[0]);
374 av_free(dims);
375 g_ort->ReleaseTypeInfo(type_info);
376 return AVERROR(ENOSYS);
377 }
378
379 for (size_t i = 1; i < num_dims; i++) {
380 if (dims[i] > INT_MAX) {
382 "ONNX model input dimension %zu (%"PRId64") is too large to be represented\n",
383 i, dims[i]);
384 av_free(dims);
385 g_ort->ReleaseTypeInfo(type_info);
386 return AVERROR(ENOSYS);
387 }
388 }
389
390 /*
391 * The ONNX backend assumes a 4-D NCHW input tensor (the rank check
392 * above already rejects anything else).
393 */
394 input->layout = DL_NCHW;
395 input->dims[0] = dims[0] > 0 ? dims[0] : 1;
396 input->dims[1] = dims[1] > 0 ? dims[1] : 3;
397 input->dims[2] = dims[2] > 0 ? dims[2] : -1;
398 input->dims[3] = dims[3] > 0 ? dims[3] : -1;
399
400 if (tensor_type == ONNX_TENSOR_ELEMENT_DATA_TYPE_FLOAT) {
401 input->dt = DNN_FLOAT;
402 } else {
403 av_log(ctx, AV_LOG_ERROR, "Unsupported input tensor data type, only float is supported\n");
404 av_free(dims);
405 g_ort->ReleaseTypeInfo(type_info);
406 return AVERROR(ENOSYS);
407 }
408
409 /*
410 * The DCO_RGB setting below is only consulted by the dnn_detect and dnn_classify;
411 * the dnn_processing path lets the source AVFrame pixel format determine the
412 * tensor channel order, so both RGB24 and BGR24 inputs work transparently
413 * for that flow.
414 */
415 input->order = DCO_RGB;
416 av_free(dims);
417 g_ort->ReleaseTypeInfo(type_info);
418
419 onnx_model->input_info = *input;
420 onnx_model->input_resolved = 1;
421 return 0;
422}
423
424static int fill_model_input_onnx(ONNXModel *onnx_model, ONNXRequestItem *request)
425{
426 LastLevelTaskItem *lltask = NULL;
427 TaskItem *task = NULL;
428 ONNXInferRequest *infer_request = NULL;
429 DNNData input = { 0 };
430 DnnContext *ctx = onnx_model->ctx;
431 int ret, width_idx, height_idx, channel_idx;
432 int64_t input_shape[4];
433 size_t input_tensor_size;
434 OrtMemoryInfo *memory_info;
435 OrtStatus *status;
436
437 lltask = (LastLevelTaskItem *)ff_queue_pop_front(onnx_model->lltask_queue);
438 if (!lltask) {
439 ret = AVERROR(EINVAL);
440 goto err;
441 }
442 request->lltask = lltask;
443 task = lltask->task;
444 infer_request = request->infer_request;
445
446 ret = get_input_onnx(&onnx_model->model, &input, task->input_name);
447 if (ret != 0) {
448 goto err;
449 }
450
451 width_idx = dnn_get_width_idx_by_layout(input.layout);
452 height_idx = dnn_get_height_idx_by_layout(input.layout);
453 channel_idx = dnn_get_channel_idx_by_layout(input.layout);
454
455 if (input.dims[height_idx] < 0)
456 input.dims[height_idx] = task->in_frame->height;
457 if (input.dims[width_idx] < 0)
458 input.dims[width_idx] = task->in_frame->width;
459
460 if (input.dims[0] <= 0 || input.dims[channel_idx] <= 0 ||
461 input.dims[height_idx] <= 0 || input.dims[width_idx] <= 0) {
462 av_log(ctx, AV_LOG_ERROR, "ONNX input tensor has a non-positive dimension\n");
463 ret = AVERROR(EINVAL);
464 goto err;
465 }
466
467 ret = av_image_check_size((unsigned)input.dims[width_idx],
468 (unsigned)input.dims[height_idx], 0, ctx);
469 if (ret < 0) {
470 av_log(ctx, AV_LOG_ERROR, "ONNX input image dimensions %dx%d are not supported\n",
471 input.dims[width_idx], input.dims[height_idx]);
472 goto err;
473 }
474
475 input_shape[0] = input.dims[0];
476 input_shape[1] = input.dims[channel_idx];
477 input_shape[2] = input.dims[height_idx];
478 input_shape[3] = input.dims[width_idx];
479
480 /*
481 * Build the byte count with checked size_t multiplications instead of
482 * multiplying four int64_t shape values in one expression.
483 */
484 input_tensor_size = sizeof(float);
485 if (av_size_mult(input_tensor_size, (size_t)input_shape[0], &input_tensor_size) < 0 ||
486 av_size_mult(input_tensor_size, (size_t)input_shape[1], &input_tensor_size) < 0 ||
487 av_size_mult(input_tensor_size, (size_t)input_shape[2], &input_tensor_size) < 0 ||
488 av_size_mult(input_tensor_size, (size_t)input_shape[3], &input_tensor_size) < 0) {
489 av_log(ctx, AV_LOG_ERROR, "ONNX input tensor size overflows\n");
490 ret = AVERROR(EINVAL);
491 goto err;
492 }
493
494 input.data = av_malloc(input_tensor_size);
495 if (!input.data) {
496 ret = AVERROR(ENOMEM);
497 goto err;
498 }
499 infer_request->input_data = input.data;
500
501 switch (onnx_model->model.func_type) {
503 input.scale = 255;
504 if (task->do_ioproc) {
505 if (onnx_model->model.frame_pre_proc != NULL) {
506 ret = onnx_model->model.frame_pre_proc(task->in_frame, &input,
507 onnx_model->model.filter_ctx);
508 } else {
509 ret = ff_proc_from_frame_to_dnn(task->in_frame, &input, ctx);
510 }
511 if (ret < 0)
512 goto err;
513 }
514 break;
516 ret = ff_frame_to_dnn_detect(task->in_frame, &input, ctx);
517 if (ret < 0)
518 goto err;
519 break;
521 ret = ff_frame_to_dnn_classify(task->in_frame, &input, lltask->bbox_index, ctx);
522 if (ret < 0)
523 goto err;
524 break;
525 default:
526 avpriv_report_missing_feature(ctx, "model function type %d", onnx_model->model.func_type);
527 ret = AVERROR(ENOSYS);
528 goto err;
529 }
530
531 status = g_ort->CreateCpuMemoryInfo(OrtArenaAllocator, OrtMemTypeDefault, &memory_info);
532 if (status != NULL) {
533 ret = AVERROR(ENOMEM);
534 goto err;
535 }
536
537 status = g_ort->CreateTensorWithDataAsOrtValue(
538 memory_info, input.data, input_tensor_size,
539 input_shape, 4, ONNX_TENSOR_ELEMENT_DATA_TYPE_FLOAT,
540 &infer_request->input_tensor);
541
542 g_ort->ReleaseMemoryInfo(memory_info);
543
544 if (status != NULL) {
545 const char *msg = g_ort->GetErrorMessage(status);
546 av_log(ctx, AV_LOG_ERROR, "Failed to create input tensor: %s\n", msg);
547 g_ort->ReleaseStatus(status);
548 ret = AVERROR(ENOMEM);
549 goto err;
550 }
551
552 return 0;
553
554err:
555 onnx_free_request(infer_request);
556 return ret;
557}
558
559static int onnx_start_inference(void *args)
560{
561 ONNXRequestItem *request = (ONNXRequestItem *)args;
562 ONNXInferRequest *infer_request = NULL;
563 LastLevelTaskItem *lltask = NULL;
564 TaskItem *task = NULL;
565 ONNXModel *onnx_model = NULL;
567 OrtStatus *status;
568 const char *input_names[1];
569 int ret = DNN_GENERIC_ERROR;
570
571 if (!request) {
572 av_log(NULL, AV_LOG_ERROR, "ONNXRequestItem is NULL\n");
573 return AVERROR(EINVAL);
574 }
575
576 infer_request = request->infer_request;
577 lltask = request->lltask;
578 task = lltask->task;
579 onnx_model = (ONNXModel *)task->model;
580 ctx = onnx_model->ctx;
581
582 if (!task->input_name || !task->output_names || !task->output_names[0]) {
584 "ONNX backend: input/output tensor name was not resolved at load time\n");
585 return AVERROR(EINVAL);
586 }
587
588 if (!infer_request->input_tensor) {
589 av_log(ctx, AV_LOG_ERROR, "Input tensor is NULL\n");
590 return DNN_GENERIC_ERROR;
591 }
592
593 if (!onnx_model->output_resolved) {
594 size_t output_count = 0;
595 int found_output = 0;
596
597 status = g_ort->SessionGetOutputCount(onnx_model->session, &output_count);
598 if (status != NULL) {
599 const char *msg = g_ort->GetErrorMessage(status);
600 av_log(ctx, AV_LOG_ERROR, "Failed to get output count: %s\n", msg);
601 g_ort->ReleaseStatus(status);
602 return AVERROR(EINVAL);
603 }
604
605 for (uint32_t req = 0; req < task->nb_output; req++) {
606 found_output = 0;
607 for (size_t i = 0; i < output_count; i++) {
608 char *name = NULL;
609 status = g_ort->SessionGetOutputName(onnx_model->session, i,
610 onnx_model->allocator, &name);
611 if (status != NULL) {
612 g_ort->ReleaseStatus(status);
613 continue;
614 }
615 if (!strcmp(name, task->output_names[req]))
616 found_output = 1;
617 onnx_model->allocator->Free(onnx_model->allocator, name);
618 if (found_output)
619 break;
620 }
621 if (!found_output) {
623 "Output name '%s' not found in ONNX model\n",
624 task->output_names[req]);
625 return AVERROR(EINVAL);
626 }
627 }
628
629 onnx_model->output_resolved = 1;
630 }
631
632 input_names[0] = task->input_name;
633
634 /* ORT writes task->nb_output result handles into this array; it must be
635 * allocated (and NULL-initialised) before Run() so ORT owns each slot. */
636 av_freep(&infer_request->output_tensors);
637 infer_request->output_tensors = av_calloc(task->nb_output,
638 sizeof(*infer_request->output_tensors));
639 if (!infer_request->output_tensors) {
640 infer_request->nb_outputs = 0;
641 return AVERROR(ENOMEM);
642 }
643 infer_request->nb_outputs = task->nb_output;
644
645 status = g_ort->Run(onnx_model->session, NULL,
646 input_names, (const OrtValue *const *)&infer_request->input_tensor, 1,
647 task->output_names, task->nb_output, infer_request->output_tensors);
648
649 if (status != NULL) {
650 const char *msg = g_ort->GetErrorMessage(status);
651 av_log(ctx, AV_LOG_ERROR, "ONNX inference failed: %s\n", msg);
652 g_ort->ReleaseStatus(status);
653 goto err;
654 }
655
656 return 0;
657
658err:
659 av_freep(&infer_request->output_tensors);
660 infer_request->nb_outputs = 0;
661 return ret;
662}
663
664static void infer_completion_callback(void *args)
665{
666 ONNXRequestItem *request = (ONNXRequestItem *)args;
667 LastLevelTaskItem *lltask = request->lltask;
668 TaskItem *task = lltask->task;
670 ONNXInferRequest *infer_request = request->infer_request;
671 ONNXModel *onnx_model = (ONNXModel *)task->model;
672 DnnContext *ctx = onnx_model->ctx;
673 OrtTensorTypeAndShapeInfo *tensor_info;
674 ONNXTensorElementDataType tensor_type;
675 size_t num_dims;
676 int64_t *dims;
677 OrtStatus *status;
678 int ret;
679
680 outputs = av_calloc(infer_request->nb_outputs, sizeof(*outputs));
681 if (!outputs) {
682 av_log(ctx, AV_LOG_ERROR, "Failed to allocate output DNNData array\n");
683 goto err;
684 }
685
686 for (uint32_t i = 0; i < infer_request->nb_outputs; i++) {
687 status = g_ort->GetTensorTypeAndShape(infer_request->output_tensors[i],
688 &tensor_info);
689 if (status != NULL) {
690 av_log(ctx, AV_LOG_ERROR, "Failed to get output tensor[%u] type/shape\n", i);
691 g_ort->ReleaseStatus(status);
692 goto err;
693 }
694
695 status = g_ort->GetDimensionsCount(tensor_info, &num_dims);
696 if (status != NULL) {
697 av_log(ctx, AV_LOG_ERROR, "Failed to get output tensor[%u] dimension count\n", i);
698 g_ort->ReleaseStatus(status);
699 g_ort->ReleaseTensorTypeAndShapeInfo(tensor_info);
700 goto err;
701 }
702
703 dims = av_malloc(num_dims * sizeof(int64_t));
704 if (!dims) {
705 av_log(ctx, AV_LOG_ERROR, "Failed to allocate dims array\n");
706 g_ort->ReleaseTensorTypeAndShapeInfo(tensor_info);
707 goto err;
708 }
709
710 status = g_ort->GetDimensions(tensor_info, dims, num_dims);
711 if (status != NULL) {
712 av_log(ctx, AV_LOG_ERROR, "Failed to get output tensor[%u] dimensions\n", i);
713 g_ort->ReleaseStatus(status);
714 av_free(dims);
715 g_ort->ReleaseTensorTypeAndShapeInfo(tensor_info);
716 goto err;
717 }
718
719 for (size_t d = 0; d < num_dims; d++) {
720 if (dims[d] < 0 || dims[d] > INT_MAX) {
722 "Output tensor[%u] dimension %zu (%"PRId64") is out of representable range\n",
723 i, d, dims[d]);
724 av_free(dims);
725 g_ort->ReleaseTensorTypeAndShapeInfo(tensor_info);
726 goto err;
727 }
728 }
729
730 status = g_ort->GetTensorElementType(tensor_info, &tensor_type);
731 if (status != NULL) {
732 av_log(ctx, AV_LOG_ERROR, "Failed to get output tensor[%u] element type\n", i);
733 g_ort->ReleaseStatus(status);
734 av_free(dims);
735 g_ort->ReleaseTensorTypeAndShapeInfo(tensor_info);
736 goto err;
737 }
738 if (tensor_type == ONNX_TENSOR_ELEMENT_DATA_TYPE_FLOAT) {
739 outputs[i].dt = DNN_FLOAT;
740 } else {
742 "Unsupported output tensor[%u] data type, only float supported\n", i);
743 av_free(dims);
744 g_ort->ReleaseTensorTypeAndShapeInfo(tensor_info);
745 goto err;
746 }
747
748 /* Output is interpreted as NCHW, matching the input assumption. */
749 outputs[i].layout = DL_NCHW;
750 outputs[i].order = DCO_RGB;
751
752 if (num_dims == 4) {
753 outputs[i].dims[0] = dims[0];
754 outputs[i].dims[1] = dims[1];
755 outputs[i].dims[2] = dims[2];
756 outputs[i].dims[3] = dims[3];
757 } else if (num_dims == 3) {
758 /* Some detection models output [1, N, D]; promote it to [1, 1, N, D]. */
759 outputs[i].dims[0] = dims[0];
760 outputs[i].dims[1] = 1;
761 outputs[i].dims[2] = dims[1];
762 outputs[i].dims[3] = dims[2];
763 } else if (num_dims == 2) {
764 /* [N, C] -> [N, C, 1, 1] */
765 outputs[i].dims[0] = dims[0];
766 outputs[i].dims[1] = dims[1];
767 outputs[i].dims[2] = 1;
768 outputs[i].dims[3] = 1;
769 } else if (num_dims == 1) {
770 /* [C] -> [1, C, 1, 1] */
771 outputs[i].dims[0] = 1;
772 outputs[i].dims[1] = dims[0];
773 outputs[i].dims[2] = 1;
774 outputs[i].dims[3] = 1;
775 } else {
777 "Support for %zu-dimensional output (tensor[%u])", num_dims, i);
778 av_free(dims);
779 g_ort->ReleaseTensorTypeAndShapeInfo(tensor_info);
780 goto err;
781 }
782
783 if (outputs[i].dims[0] != 1) {
785 "Output tensor[%u] batch size %d unsupported, must be 1\n",
786 i, outputs[i].dims[0]);
787 av_free(dims);
788 g_ort->ReleaseTensorTypeAndShapeInfo(tensor_info);
789 goto err;
790 }
791
792 status = g_ort->GetTensorMutableData(infer_request->output_tensors[i], &outputs[i].data);
793 if (status != NULL) {
794 av_log(ctx, AV_LOG_ERROR, "Failed to get tensor[%u] data pointer\n", i);
795 g_ort->ReleaseStatus(status);
796 av_free(dims);
797 g_ort->ReleaseTensorTypeAndShapeInfo(tensor_info);
798 goto err;
799 }
800
801 av_free(dims);
802 g_ort->ReleaseTensorTypeAndShapeInfo(tensor_info);
803 }
804
805 switch (onnx_model->model.func_type) {
807 if (task->do_ioproc) {
808 outputs[0].scale = 255;
809 if (onnx_model->model.frame_post_proc != NULL) {
810 onnx_model->model.frame_post_proc(task->out_frame, outputs, onnx_model->model.filter_ctx);
811 } else {
813 }
814 } else {
817 }
818 break;
820 ret = onnx_model->model.detect_post_proc(task->in_frame, outputs,
821 infer_request->nb_outputs,
822 onnx_model->model.filter_ctx);
823 if (ret < 0)
824 goto err;
825 break;
827 if (!onnx_model->model.classify_post_proc) {
828 av_log(ctx, AV_LOG_ERROR, "classify filter needs to provide classify_post_proc\n");
829 goto err;
830 }
831 onnx_model->model.classify_post_proc(task->in_frame, outputs,
832 request->lltask->bbox_index,
833 onnx_model->model.filter_ctx);
834 break;
835 default:
836 avpriv_report_missing_feature(ctx, "model function type %d", onnx_model->model.func_type);
837 goto err;
838 }
839
840 task->inference_done++;
841
842err:
844 av_freep(&request->lltask);
845 onnx_free_request(infer_request);
846 if (ff_safe_queue_push_back(onnx_model->request_queue, request) < 0) {
847 destroy_request_item(&request);
848 av_log(ctx, AV_LOG_ERROR, "Unable to push back request_queue.\n");
849 }
850}
851
852static int execute_model_onnx(ONNXModel *onnx_model, ONNXRequestItem *request, Queue *lltask_queue)
853{
854 LastLevelTaskItem *lltask;
855 TaskItem *task = NULL;
856 int ret = 0;
857
858 if (ff_queue_size(lltask_queue) == 0) {
859 if (ff_safe_queue_push_back(onnx_model->request_queue, request) < 0) {
860 destroy_request_item(&request);
861 }
862 return 0;
863 }
864
865 /* Drain all lltasks for the current frame. */
866 for (;;) {
867 lltask = (LastLevelTaskItem *)ff_queue_peek_front(lltask_queue);
868 if (lltask == NULL) {
869 av_log(NULL, AV_LOG_ERROR, "Failed to get LastLevelTaskItem\n");
870 ret = AVERROR(EINVAL);
871 goto err;
872 }
873 task = lltask->task;
874
875 ret = fill_model_input_onnx(onnx_model, request);
876 if (ret != 0) {
877 goto err;
878 }
879
880 if (task->async) {
881 avpriv_report_missing_feature(onnx_model->ctx, "ONNX async inference");
882 ret = AVERROR(ENOSYS);
883 goto err;
884 }
885
886 ret = onnx_start_inference((void *)request);
887 if (ret != 0) {
888 goto err;
889 }
891
892 if (ff_queue_size(lltask_queue) == 0) {
893 break;
894 }
895 request = (ONNXRequestItem *)ff_safe_queue_pop_front(onnx_model->request_queue);
896 }
897
898 return (task->inference_done == task->inference_todo) ? 0 : DNN_GENERIC_ERROR;
899
900err:
901 av_freep(&request->lltask);
903 if (ff_safe_queue_push_back(onnx_model->request_queue, request) < 0) {
904 destroy_request_item(&request);
905 }
906 return ret;
907}
908
909static int get_output_onnx(DNNModel *model, const char *input_name, int input_width, int input_height,
910 const char *output_name, int *output_width, int *output_height)
911{
912 int ret = 0;
913 ONNXModel *onnx_model = (ONNXModel *)model;
914 DnnContext *ctx = onnx_model->ctx;
915 TaskItem task = { 0 };
916 ONNXRequestItem *request = NULL;
917 DNNExecBaseParams exec_params = {
918 .input_name = input_name,
919 .output_names = &output_name,
920 .nb_output = 1,
921 .in_frame = NULL,
922 .out_frame = NULL,
923 };
924
925 ret = ff_dnn_fill_gettingoutput_task(&task, &exec_params, onnx_model, input_height, input_width, ctx);
926 if (ret != 0) {
927 goto err;
928 }
929
931 if (ret != 0) {
932 av_log(ctx, AV_LOG_ERROR, "Unable to extract last level task from task.\n");
933 goto err;
934 }
935
936 request = (ONNXRequestItem *)ff_safe_queue_pop_front(onnx_model->request_queue);
937 if (!request) {
938 av_log(ctx, AV_LOG_ERROR, "Unable to get infer request.\n");
939 ret = AVERROR(EINVAL);
940 goto err;
941 }
942
943 ret = execute_model_onnx(onnx_model, request, onnx_model->lltask_queue);
944 *output_width = task.out_frame->width;
945 *output_height = task.out_frame->height;
946
947err:
949 av_frame_free(&task.in_frame);
950 return ret;
951}
952
954{
955 ONNXInferRequest *request = av_mallocz(sizeof(ONNXInferRequest));
956 if (!request)
957 return NULL;
958 return request;
959}
960
962{
963 DNNModel *model = NULL;
964 ONNXModel *onnx_model = NULL;
965 ONNXRequestItem *item = NULL;
966 ONNXOptions *options = &ctx->onnx_option;
967 OrtStatus *status;
968
970 if (!g_ort) {
971 av_log(ctx, AV_LOG_ERROR, "Failed to get ONNX Runtime API\n");
972 return NULL;
973 }
974
975 onnx_model = av_mallocz(sizeof(ONNXModel));
976 if (!onnx_model)
977 return NULL;
978
979 model = &onnx_model->model;
980 onnx_model->ctx = ctx;
981
982 if (ctx->nireq > 1) {
984 "nireq=%d is not supported by the ONNX Runtime backend, "
985 "which only allocates a single, synchronous inference "
986 "request. Rolling back to nireq=1.\n", ctx->nireq);
987 }
988 ctx->nireq = 1;
989
990 status = g_ort->CreateEnv(ORT_LOGGING_LEVEL_WARNING, "FFmpeg", &onnx_model->env);
991 if (status != NULL) {
992 av_log(ctx, AV_LOG_ERROR, "Failed to create ONNX Runtime environment\n");
993 goto fail;
994 }
995
996 status = g_ort->CreateSessionOptions(&onnx_model->session_options);
997 if (status != NULL) {
998 av_log(ctx, AV_LOG_ERROR, "Failed to create session options\n");
999 goto fail;
1000 }
1001
1002 if (options->num_threads > 0 &&
1003 (!ctx->device || av_strcasecmp(ctx->device, "cpu") == 0)) {
1004 g_ort->SetIntraOpNumThreads(onnx_model->session_options, options->num_threads);
1005 }
1006 g_ort->SetSessionGraphOptimizationLevel(onnx_model->session_options, ORT_ENABLE_ALL);
1007
1008 if (ctx->device && av_strcasecmp(ctx->device, "cpu") != 0) {
1009 if (av_strcasecmp(ctx->device, "cuda") == 0) {
1010 if (g_ort->SessionOptionsAppendExecutionProvider_CUDA) {
1011 OrtCUDAProviderOptions cuda_options;
1012 memset(&cuda_options, 0, sizeof(cuda_options));
1013 cuda_options.device_id = ctx->device_id;
1014
1015 status = g_ort->SessionOptionsAppendExecutionProvider_CUDA(
1016 onnx_model->session_options, &cuda_options);
1017 if (status != NULL) {
1018 const char *msg = g_ort->GetErrorMessage(status);
1019 av_log(ctx, AV_LOG_WARNING, "Failed to enable CUDA (device %d): %s. Falling back to CPU\n",
1020 ctx->device_id, msg);
1021 g_ort->ReleaseStatus(status);
1022 } else {
1023 av_log(ctx, AV_LOG_INFO, "Using CUDA execution provider on device %d\n", ctx->device_id);
1024 }
1025 } else {
1026 av_log(ctx, AV_LOG_WARNING, "CUDA provider function not available in this ONNX Runtime API version. Falling back to CPU\n");
1027 }
1028 } else if (av_strcasecmp(ctx->device, "dml") == 0) {
1029#ifdef _WIN32
1030 const char* dml_options_keys[] = {"device_id"};
1031 const char* dml_options_values[] = {NULL};
1032 char device_id_str[32];
1033 snprintf(device_id_str, sizeof(device_id_str), "%d", ctx->device_id);
1034 dml_options_values[0] = device_id_str;
1035
1036 /* DirectML cannot use ORT's memory-pattern optimizer and only
1037 * supports sequential execution. */
1038 status = g_ort->SetSessionExecutionMode(onnx_model->session_options, ORT_SEQUENTIAL);
1039 if (status)
1040 g_ort->ReleaseStatus(status);
1041 status = g_ort->DisableMemPattern(onnx_model->session_options);
1042 if (status)
1043 g_ort->ReleaseStatus(status);
1044
1045 if (g_ort->SessionOptionsAppendExecutionProvider) {
1046 status = g_ort->SessionOptionsAppendExecutionProvider(
1047 onnx_model->session_options, "DML",
1048 dml_options_keys, dml_options_values, 1);
1049 if (status != NULL) {
1050 const char *msg = g_ort->GetErrorMessage(status);
1051 av_log(ctx, AV_LOG_WARNING, "Failed to enable DirectML (device %d): %s. Falling back to CPU\n",
1052 ctx->device_id, msg);
1053 g_ort->ReleaseStatus(status);
1054 } else {
1055 av_log(ctx, AV_LOG_INFO, "Using DirectML execution provider on device %d\n", ctx->device_id);
1056 }
1057 } else {
1058 av_log(ctx, AV_LOG_WARNING, "DirectML provider function not available in this ONNX Runtime API version. Falling back to CPU\n");
1059 }
1060#else
1061 av_log(ctx, AV_LOG_WARNING, "DirectML is only available on Windows. Falling back to CPU\n");
1062#endif
1063 } else if (av_strcasecmp(ctx->device, "vitisai") == 0) {
1064 if (g_ort->SessionOptionsAppendExecutionProvider) {
1065 status = g_ort->SessionOptionsAppendExecutionProvider(
1066 onnx_model->session_options, "VitisAI",
1067 NULL, NULL, 0);
1068 if (status != NULL) {
1069 const char *msg = g_ort->GetErrorMessage(status);
1071 "Failed to enable VitisAI EP: %s. Falling back to CPU\n", msg);
1072 g_ort->ReleaseStatus(status);
1073 } else {
1074 av_log(ctx, AV_LOG_INFO, "Using VitisAI execution provider (AMD Ryzen AI NPU)\n");
1075 }
1076 } else {
1078 "VitisAI provider function not available in this ONNX Runtime API version. Falling back to CPU.\n");
1079 }
1080 } else {
1081#ifdef _WIN32
1083 "Unknown device '%s'. Supported: cpu, cuda, dml, vitisai. Using CPU\n",
1084 ctx->device);
1085#else
1087 "Unknown device '%s'. Supported: cpu, cuda, vitisai. Using CPU\n",
1088 ctx->device);
1089#endif
1090 }
1091 } else {
1092 av_log(ctx, AV_LOG_INFO, "Using CPU execution provider\n");
1093 }
1094
1095#ifdef _WIN32
1096 {
1097 wchar_t *wfilename = NULL;
1098 if (utf8towchar(ctx->model_filename, &wfilename)) {
1099 av_log(ctx, AV_LOG_ERROR, "Failed to convert model filename to UTF-16\n");
1100 goto fail;
1101 }
1102 if (!wfilename) {
1103 av_log(ctx, AV_LOG_ERROR, "Failed to convert model filename to UTF-16\n");
1104 goto fail;
1105 }
1106
1107 status = g_ort->CreateSession(onnx_model->env, wfilename,
1108 onnx_model->session_options, &onnx_model->session);
1109 av_free(wfilename);
1110 }
1111#else
1112 status = g_ort->CreateSession(onnx_model->env, ctx->model_filename,
1113 onnx_model->session_options, &onnx_model->session);
1114#endif
1115 if (status != NULL) {
1116 const char *msg = g_ort->GetErrorMessage(status);
1117 av_log(ctx, AV_LOG_ERROR, "Failed to create ONNX session: %s\n", msg);
1118 g_ort->ReleaseStatus(status);
1119 goto fail;
1120 }
1121
1122 status = g_ort->GetAllocatorWithDefaultOptions(&onnx_model->allocator);
1123 if (status != NULL) {
1124 av_log(ctx, AV_LOG_ERROR, "Failed to get allocator\n");
1125 goto fail;
1126 }
1127
1128 /*
1129 * The ONNX backend binds exactly one input tensor to Run(), so only
1130 * single-input models are supported.
1131 */
1132 {
1133 size_t input_count = 0;
1134 status = g_ort->SessionGetInputCount(onnx_model->session, &input_count);
1135 if (status != NULL) {
1136 const char *msg = g_ort->GetErrorMessage(status);
1137 av_log(ctx, AV_LOG_ERROR, "Failed to get model input count: %s\n", msg);
1138 g_ort->ReleaseStatus(status);
1139 goto fail;
1140 }
1141 if (input_count == 0) {
1142 av_log(ctx, AV_LOG_ERROR, "ONNX model exposes no input tensors\n");
1143 goto fail;
1144 }
1145 if (input_count > 1) {
1147 "ONNX model exposes %zu input tensors; the ONNX backend "
1148 "supports single-input models only.\n",
1149 input_count);
1150 goto fail;
1151 }
1152 }
1153
1154 /* Auto-detect the input tensor name when the user did not pass input=NAME. */
1155 if (!ctx->model_inputname || !*ctx->model_inputname) {
1156 char *name = NULL;
1157 status = g_ort->SessionGetInputName(onnx_model->session, 0,
1158 onnx_model->allocator, &name);
1159 if (status != NULL) {
1160 const char *msg = g_ort->GetErrorMessage(status);
1161 av_log(ctx, AV_LOG_ERROR, "Failed to get model input name: %s\n", msg);
1162 g_ort->ReleaseStatus(status);
1163 goto fail;
1164 }
1165 av_freep(&ctx->model_inputname);
1166 ctx->model_inputname = av_strdup(name);
1167 onnx_model->allocator->Free(onnx_model->allocator, name);
1168 if (!ctx->model_inputname)
1169 goto fail;
1170 av_log(ctx, AV_LOG_INFO, "Auto-detected ONNX input tensor '%s'\n",
1171 ctx->model_inputname);
1172 }
1173
1174 /* Auto-detect the output tensor name when the user did not pass output=NAME. */
1175 if (!ctx->model_outputnames) {
1176 size_t output_count = 0;
1177 char *name = NULL;
1178 status = g_ort->SessionGetOutputCount(onnx_model->session, &output_count);
1179 if (status != NULL) {
1180 const char *msg = g_ort->GetErrorMessage(status);
1181 av_log(ctx, AV_LOG_ERROR, "Failed to get model output count: %s\n", msg);
1182 g_ort->ReleaseStatus(status);
1183 goto fail;
1184 }
1185 if (output_count == 0) {
1186 av_log(ctx, AV_LOG_ERROR, "ONNX model exposes no output tensors\n");
1187 goto fail;
1188 }
1189 status = g_ort->SessionGetOutputName(onnx_model->session, 0,
1190 onnx_model->allocator, &name);
1191 if (status != NULL) {
1192 const char *msg = g_ort->GetErrorMessage(status);
1193 av_log(ctx, AV_LOG_ERROR, "Failed to get model output name: %s\n", msg);
1194 g_ort->ReleaseStatus(status);
1195 goto fail;
1196 }
1197 ctx->model_outputnames = av_calloc(1, sizeof(*ctx->model_outputnames));
1198 if (!ctx->model_outputnames) {
1199 onnx_model->allocator->Free(onnx_model->allocator, name);
1200 goto fail;
1201 }
1202 ctx->model_outputnames[0] = av_strdup(name);
1203 onnx_model->allocator->Free(onnx_model->allocator, name);
1204 if (!ctx->model_outputnames[0]) {
1205 av_freep(&ctx->model_outputnames);
1206 goto fail;
1207 }
1208 ctx->nb_outputs = 1;
1209 if (output_count == 1) {
1210 av_log(ctx, AV_LOG_INFO, "Auto-detected ONNX output tensor '%s'\n",
1211 ctx->model_outputnames[0]);
1212 } else {
1214 "ONNX model exposes %zu output tensors; auto-using index 0 ('%s'). "
1215 "Specify output=NAME to choose a different one.\n",
1216 output_count, ctx->model_outputnames[0]);
1217 }
1218 }
1219
1220 onnx_model->request_queue = ff_safe_queue_create();
1221 if (!onnx_model->request_queue) {
1222 goto fail;
1223 }
1224
1225 item = av_mallocz(sizeof(ONNXRequestItem));
1226 if (!item) {
1227 goto fail;
1228 }
1229 item->lltask = NULL;
1231 if (!item->infer_request) {
1232 av_log(ctx, AV_LOG_ERROR, "Failed to allocate memory for ONNX inference request\n");
1233 goto fail;
1234 }
1237 item->exec_module.args = item;
1238
1239 if (ff_safe_queue_push_back(onnx_model->request_queue, item) < 0) {
1240 goto fail;
1241 }
1242 item = NULL;
1243
1244 onnx_model->task_queue = ff_queue_create();
1245 if (!onnx_model->task_queue) {
1246 goto fail;
1247 }
1248
1249 onnx_model->lltask_queue = ff_queue_create();
1250 if (!onnx_model->lltask_queue) {
1251 goto fail;
1252 }
1253
1254 model->get_input = &get_input_onnx;
1255 model->get_output = &get_output_onnx;
1256 model->filter_ctx = filter_ctx;
1257 model->func_type = func_type;
1258
1259 return model;
1260
1261fail:
1262 if (item) {
1263 destroy_request_item(&item);
1264 }
1265 dnn_free_model_onnx(&model);
1266 return NULL;
1267}
1268
1269static int dnn_execute_model_onnx(const DNNModel *model, DNNExecBaseParams *exec_params)
1270{
1271 ONNXModel *onnx_model = (ONNXModel *)model;
1272 DnnContext *ctx = onnx_model->ctx;
1273 TaskItem *task;
1274 ONNXRequestItem *request;
1275 int ret = 0;
1276
1277 ret = ff_check_exec_params(ctx, DNN_ONNX, model->func_type, exec_params);
1278 if (ret != 0) {
1279 av_log(ctx, AV_LOG_ERROR, "Exec parameter checking failed.\n");
1280 return ret;
1281 }
1282
1283 task = av_malloc(sizeof(TaskItem));
1284 if (!task) {
1285 av_log(ctx, AV_LOG_ERROR, "Unable to alloc memory for task item.\n");
1286 return AVERROR(ENOMEM);
1287 }
1288
1289 ret = ff_dnn_fill_task(task, exec_params, onnx_model, 0, 1);
1290 if (ret != 0) {
1291 av_freep(&task);
1292 av_log(ctx, AV_LOG_ERROR, "Unable to fill task.\n");
1293 return ret;
1294 }
1295
1296 ret = ff_queue_push_back(onnx_model->task_queue, task);
1297 if (ret < 0) {
1298 av_freep(&task);
1299 av_log(ctx, AV_LOG_ERROR, "Unable to push back task_queue.\n");
1300 return ret;
1301 }
1302
1303 ret = extract_lltask_from_task(model->func_type, task, onnx_model->lltask_queue, exec_params);
1304 if (ret != 0) {
1305 av_log(ctx, AV_LOG_ERROR, "Unable to extract last level task from task.\n");
1306 return ret;
1307 }
1308
1309 /* No lltasks queued, nothing to infer. */
1310 if (ff_queue_size(onnx_model->lltask_queue) == 0) {
1311 return 0;
1312 }
1313
1314 request = (ONNXRequestItem *)ff_safe_queue_pop_front(onnx_model->request_queue);
1315 if (!request) {
1316 av_log(ctx, AV_LOG_ERROR, "Unable to get infer request.\n");
1317 return AVERROR(EINVAL);
1318 }
1319
1320 return execute_model_onnx(onnx_model, request, onnx_model->lltask_queue);
1321}
1322
1324{
1325 ONNXModel *onnx_model = (ONNXModel *)model;
1326 return ff_dnn_get_result_common(onnx_model->task_queue, in, out);
1327}
1328
1329static int dnn_flush_onnx(const DNNModel *model)
1330{
1331 ONNXModel *onnx_model = (ONNXModel *)model;
1332 ONNXRequestItem *request;
1333
1334 if (ff_queue_size(onnx_model->lltask_queue) == 0)
1335 return 0;
1336
1337 request = (ONNXRequestItem *)ff_safe_queue_pop_front(onnx_model->request_queue);
1338 if (!request) {
1339 av_log(onnx_model->ctx, AV_LOG_ERROR, "Unable to get infer request.\n");
1340 return AVERROR(EINVAL);
1341 }
1342
1343 return execute_model_onnx(onnx_model, request, onnx_model->lltask_queue);
1344}
1345
1347 .clazz = DNN_DEFINE_CLASS(dnn_onnx),
1348 .type = DNN_ONNX,
1349 .load_model = dnn_load_model_onnx,
1350 .execute_model = dnn_execute_model_onnx,
1351 .get_result = dnn_get_result_onnx,
1352 .flush = dnn_flush_onnx,
1353 .free_model = dnn_free_model_onnx,
1354};
SwsAArch64OpImplParams params
Definition ops.c:51
static const AVFilterPad outputs[]
Definition af_aap.c:310
static FILE * out
static AVFormatContext * ctx
simple assert() macros that are a bit more flexible than ISO C assert().
#define i(width, name, range_min, range_max)
Definition cbs_h264.c:63
#define FLAGS
Definition cmdutils.c:596
#define NULL
Definition coverity.c:32
long long int64_t
Definition coverity.c:34
static AVFrame * frame
#define AV_NUM_DETECTION_BBOX_CLASSIFY
At most 4 classifications based on the detected bounding box.
static av_always_inline AVDetectionBBox * av_get_detection_bbox(const AVDetectionBBoxHeader *header, unsigned int idx)
int ff_check_exec_params(void *ctx, DNNBackendType backend, DNNFunctionType func_type, DNNExecBaseParams *exec_params)
void ff_dnn_wait_requests(SafeQueue *request_queue, int nireq)
Wait for all inference requests to complete before teardown.
DNNAsyncStatusType ff_dnn_get_result_common(Queue *task_queue, AVFrame **in, AVFrame **out)
Extract input and output frame from the Task Queue after asynchronous inference.
int ff_dnn_async_module_cleanup(DNNAsyncExecModule *async_module)
Join the Async Execution thread and set module pointers to NULL.
int ff_dnn_fill_task(TaskItem *task, DNNExecBaseParams *exec_params, void *backend_model, int async, int do_ioproc)
Fill the Task for Backend Execution.
int ff_dnn_fill_gettingoutput_task(TaskItem *task, DNNExecBaseParams *exec_params, void *backend_model, int input_height, int input_width, void *ctx)
Allocate input and output frames and fill the Task with execution parameters.
DNN common functions different backends.
#define DNN_DEFINE_CLASS(fname)
static int dnn_execute_model_onnx(const DNNModel *model, DNNExecBaseParams *exec_params)
const DNNModule ff_dnn_backend_onnx
static const AVOption dnn_onnx_options[]
static int dnn_flush_onnx(const DNNModel *model)
static DNNModel * dnn_load_model_onnx(DnnContext *ctx, DNNFunctionType func_type, AVFilterContext *filter_ctx)
static int get_output_onnx(DNNModel *model, const char *input_name, int input_width, int input_height, const char *output_name, int *output_width, int *output_height)
static int fill_model_input_onnx(ONNXModel *onnx_model, ONNXRequestItem *request)
static ONNXInferRequest * onnx_create_inference_request(void)
static void init_ort_api(void)
static void onnx_free_request(ONNXInferRequest *request)
static const OrtApi * g_ort
static void dnn_free_model_onnx(DNNModel **model)
static DNNAsyncStatusType dnn_get_result_onnx(const DNNModel *model, AVFrame **in, AVFrame **out)
static AVOnce g_ort_init_once
static int onnx_start_inference(void *args)
static int extract_lltask_from_task(DNNFunctionType func_type, TaskItem *task, Queue *lltask_queue, DNNExecBaseParams *exec_params)
#define OFFSET(x)
static int contain_valid_detection_bbox(AVFrame *frame)
static void destroy_request_item(ONNXRequestItem **arg)
static int execute_model_onnx(ONNXModel *onnx_model, ONNXRequestItem *request, Queue *lltask_queue)
static void infer_completion_callback(void *args)
static int get_input_onnx(DNNModel *model, DNNData *input, const char *input_name)
static int dnn_get_height_idx_by_layout(DNNLayout layout)
DNNAsyncStatusType
@ DL_NCHW
@ DNN_ONNX
DNNFunctionType
@ DFT_ANALYTICS_CLASSIFY
@ DFT_PROCESS_FRAME
@ DFT_ANALYTICS_DETECT
static int dnn_get_width_idx_by_layout(DNNLayout layout)
#define DNN_GENERIC_ERROR
@ DNN_FLOAT
@ DCO_RGB
static int dnn_get_channel_idx_by_layout(DNNLayout layout)
int ff_proc_from_frame_to_dnn(AVFrame *frame, DNNData *input, void *log_ctx)
int ff_frame_to_dnn_detect(AVFrame *frame, DNNData *input, void *log_ctx)
int ff_frame_to_dnn_classify(AVFrame *frame, DNNData *input, uint32_t bbox_index, void *log_ctx)
int ff_proc_from_dnn_to_frame(AVFrame *frame, DNNData *output, void *log_ctx)
Definition dnn_io_proc.c:42
DNN input&output process between AVFrame and DNNData.
#define fail
Definition test.h:479
@ AV_OPT_TYPE_INT
Underlying C type is int.
Definition opt.h:258
#define AVERROR(e)
Definition error.h:45
AVFrameSideData * av_frame_get_side_data(const AVFrame *frame, enum AVFrameSideDataType type)
Definition frame.c:659
void av_frame_free(AVFrame **frame)
Free the frame and any dynamically allocated objects in it, e.g.
Definition frame.c:64
@ AV_FRAME_DATA_DETECTION_BBOXES
Bounding boxes for object detection and classification, as described by AVDetectionBBoxHeader.
Definition frame.h:194
#define AV_LOG_WARNING
Something somehow does not look correct.
Definition log.h:216
#define AV_LOG_INFO
Standard information.
Definition log.h:221
#define AV_LOG_ERROR
Something went wrong and cannot losslessly be recovered.
Definition log.h:210
int av_size_mult(size_t a, size_t b, size_t *r)
Multiply two size_t values checking for overflow.
Definition mem.c:671
int av_image_check_size(unsigned int w, unsigned int h, int log_offset, void *log_ctx)
Check if the given dimension of an image is valid, meaning that all bytes of the image can be address...
Definition imgutils.c:318
int av_strcasecmp(const char *a, const char *b)
Locale-independent case-insensitive compare.
Definition avstring.c:208
int av_strncasecmp(const char *a, const char *b, size_t n)
Locale-independent case-insensitive compare.
Definition avstring.c:218
misc image utilities
const char * arg
Definition jacosubdec.c:65
#define AVFILTER_DEFINE_CLASS(fname)
Definition filters.h:478
void avpriv_report_missing_feature(void *avc, const char *msg,...) av_printf_format(2
Log a generic warning message about a missing feature.
#define AVOnce
Definition thread.h:202
static int ff_thread_once(char *control, void(*routine)(void))
Definition thread.h:205
#define AV_ONCE_INIT
Definition thread.h:203
uint64_t layout
void * av_calloc(size_t nmemb, size_t size)
Definition mem.c:370
Memory handling functions.
#define av_strdup(s)
Definition ops_static.c:55
#define av_malloc(s)
Definition ops_static.c:52
AVOptions.
const char * name
Definition qsvenc.c:142
void ff_queue_destroy(Queue *q)
Destroy the Queue instance.
Definition queue.c:72
void * ff_queue_pop_front(Queue *q)
Remove and free first element from the Queue.
Definition queue.c:151
int ff_queue_push_back(Queue *q, void *v)
Add data to the tail of the queue.
Definition queue.c:130
void * ff_queue_peek_front(Queue *q)
Return a pointer to the data at the head of the queue.
Definition queue.c:93
size_t ff_queue_size(Queue *q)
Return the length of the Queue.
Definition queue.c:88
Queue * ff_queue_create(void)
Create a Queue instance.
Definition queue.c:47
int ff_safe_queue_push_back(SafeQueue *sq, void *v)
Add data to the tail of queue in the SafeQueue after locking mutex.
Definition safe_queue.c:87
void * ff_safe_queue_pop_front(SafeQueue *sq)
Remove and free first element from the queue in SafeQueue.
Definition safe_queue.c:97
size_t ff_safe_queue_size(SafeQueue *sq)
Return the length of the SafeQueue.
Definition safe_queue.c:61
SafeQueue * ff_safe_queue_create(void)
Create and initialize a SafeQueue instance.
Definition safe_queue.c:33
void ff_safe_queue_destroy(SafeQueue *sq)
Destroy the SafeQueue instance.
Definition safe_queue.c:50
static const uint8_t header[24]
Definition sdr2.c:68
char detect_label[AV_DETECTION_BBOX_LABEL_NAME_MAX_SIZE]
Detect result with confidence.
int x
Distance in pixels from the left/top edge of the frame, together with width and height,...
uint32_t classify_count
An instance of a filter.
Definition avfilter.h:279
Structure to hold side data for an AVFrame.
Definition frame.h:334
size_t size
Definition frame.h:337
uint8_t * data
Definition frame.h:336
This structure describes decoded (raw) audio or video data.
Definition frame.h:479
int width
Definition frame.h:551
int height
Definition frame.h:551
AVOption.
Definition opt.h:428
Common Async Execution Mechanism for the DNN Backends.
void * args
Argument for the execution functions.
int(* start_inference)(void *request)
Synchronous inference function for the backend with corresponding request item as the argument.
void(* callback)(void *args)
Completion Callback for the backend.
float scale
DNNDataType dt
int dims[4]
DNNColorOrder order
void * data
DNNLayout layout
int(* get_input)(struct DNNModel *model, DNNData *input, const char *input_name)
int(* get_output)(struct DNNModel *model, const char *input_name, int input_width, int input_height, const char *output_name, int *output_width, int *output_height)
FramePrePostProc frame_pre_proc
ClassifyPostProc classify_post_proc
FramePrePostProc frame_post_proc
DetectPostProc detect_post_proc
AVFilterContext * filter_ctx
DNNFunctionType func_type
OrtValue * input_tensor
OrtValue ** output_tensors
Queue * task_queue
DnnContext * ctx
DNNData input_info
OrtSessionOptions * session_options
SafeQueue * request_queue
OrtSession * session
DNNModel model
OrtAllocator * allocator
Queue * lltask_queue
DNNAsyncExecModule exec_module
ONNXInferRequest * infer_request
LastLevelTaskItem * lltask
Linear double-ended data structure.
Definition executor.c:51
Double-ended queue with mutex locks ensuring data consistency while multithreading.
Definition safe_queue.c:27
uint32_t inference_done
AVFrame * in_frame
const char ** output_names
uint8_t do_ioproc
uint32_t inference_todo
const char * input_name
AVFrame * out_frame
uint32_t nb_output
#define av_free(p)
#define av_mallocz(s)
#define av_freep(p)
#define av_log(a,...)
static FilteringContext * filter_ctx
Definition transcode.c:52