FFmpeg
Loading...
Searching...
No Matches
ops.c
Go to the documentation of this file.
1/**
2 * Copyright (C) 2026 Lynne
3 *
4 * This file is part of FFmpeg.
5 *
6 * FFmpeg is free software; you can redistribute it and/or
7 * modify it under the terms of the GNU Lesser General Public
8 * License as published by the Free Software Foundation; either
9 * version 2.1 of the License, or (at your option) any later version.
10 *
11 * FFmpeg is distributed in the hope that it will be useful,
12 * but WITHOUT ANY WARRANTY; without even the implied warranty of
13 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
14 * Lesser General Public License for more details.
15 *
16 * You should have received a copy of the GNU Lesser General Public
17 * License along with FFmpeg; if not, write to the Free Software
18 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
19 */
20
21#include "libavutil/mem.h"
22#include "libavutil/refstruct.h"
23
24#include "../graph.h"
25#include "../ops_internal.h"
26#include "../swscale_internal.h"
27
28#include "ops.h"
29
30#if HAVE_SPIRV_HEADERS_SPIRV_H || HAVE_SPIRV_UNIFIED1_SPIRV_H
31#include "spvasm.h"
32#endif
33
34static void ff_sws_vk_uninit(AVRefStructOpaque opaque, void *obj)
35{
37
38 ff_vk_uninit(&s->vkctx);
39}
40
42{
43 int err;
45
46 if (!c->hw_priv) {
47 c->hw_priv = av_refstruct_alloc_ext(sizeof(FFVulkanOpsCtx), 0, NULL,
49 if (!c->hw_priv)
50 return AVERROR(ENOMEM);
51 }
52
53 FFVulkanOpsCtx *s = c->hw_priv;
54 if (s->vkctx.device_ref && s->vkctx.device_ref->data != dev_ref->data) {
55 /* Reinitialize with new context */
56 ff_vk_uninit(&s->vkctx);
57 } else if (s->vkctx.device_ref && s->vkctx.device_ref->data == dev_ref->data) {
58 return 0;
59 }
60
61 err = ff_vk_init(&s->vkctx, sws, dev_ref, NULL);
62 if (err < 0)
63 return err;
64
65 s->qf = ff_vk_qf_find(&s->vkctx, VK_QUEUE_COMPUTE_BIT, 0);
66 if (!s->qf) {
67 av_log(sws, AV_LOG_ERROR, "Device has no compute queues\n");
68 return AVERROR(ENOTSUP);
69 }
70
71 return 0;
72}
73
75{
77 FFVulkanOpsCtx *s = c->hw_priv;
78 return s ? s->vkctx.device_ref : NULL;
79}
80
81#define MAX_DITHER_BUFS 4
82#define MAX_FILT_BUFS 4
83#define MAX_DATA_BUFS (MAX_DITHER_BUFS + MAX_FILT_BUFS*4)
84
95
96static void process(const SwsFrame *dst, const SwsFrame *src, int y, int h,
97 const SwsPass *pass)
98{
99 VulkanPriv *p = (VulkanPriv *) pass->priv;
100 FFVkExecContext *ec = ff_vk_exec_get(&p->s->vkctx, &p->e);
101 FFVulkanFunctions *vk = &p->s->vkctx.vkfn;
102 ff_vk_exec_start(&p->s->vkctx, ec);
103
104 AVFrame *src_f = (AVFrame *) src->avframe;
105 AVFrame *dst_f = (AVFrame *) dst->avframe;
106 ff_vk_exec_add_dep_frame(&p->s->vkctx, ec, src_f,
107 VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT,
108 VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT);
109 ff_vk_exec_add_dep_frame(&p->s->vkctx, ec, dst_f,
110 VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT,
111 VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT);
112
113 VkImageView src_views[AV_NUM_DATA_POINTERS];
114 VkImageView dst_views[AV_NUM_DATA_POINTERS];
115 ff_vk_create_imageviews(&p->s->vkctx, ec, src_views, src_f, p->src_rep);
116 ff_vk_create_imageviews(&p->s->vkctx, ec, dst_views, dst_f, p->dst_rep);
117
118 ff_vk_shader_update_img_array(&p->s->vkctx, ec, &p->shd, src_f, src_views,
119 0, 0, VK_IMAGE_LAYOUT_GENERAL, VK_NULL_HANDLE);
120 ff_vk_shader_update_img_array(&p->s->vkctx, ec, &p->shd, dst_f, dst_views,
121 0, 1, VK_IMAGE_LAYOUT_GENERAL, VK_NULL_HANDLE);
122
123 int nb_img_bar = 0;
124 VkImageMemoryBarrier2 img_bar[8];
125 ff_vk_frame_barrier(&p->s->vkctx, ec, src_f, img_bar, &nb_img_bar,
126 VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT,
127 VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT,
128 VK_ACCESS_SHADER_READ_BIT,
129 VK_IMAGE_LAYOUT_GENERAL,
130 VK_QUEUE_FAMILY_IGNORED);
131 ff_vk_frame_barrier(&p->s->vkctx, ec, dst_f, img_bar, &nb_img_bar,
132 VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT,
133 VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT,
134 VK_ACCESS_SHADER_WRITE_BIT,
135 VK_IMAGE_LAYOUT_GENERAL,
136 VK_QUEUE_FAMILY_IGNORED);
137 vk->CmdPipelineBarrier2(ec->buf, &(VkDependencyInfo) {
138 .sType = VK_STRUCTURE_TYPE_DEPENDENCY_INFO,
139 .pImageMemoryBarriers = img_bar,
140 .imageMemoryBarrierCount = nb_img_bar,
141 });
142
143 if (p->interlaced) {
144 uint32_t field = pass->graph ? pass->graph->dst.field : 0;
145 ff_vk_shader_update_push_const(&p->s->vkctx, ec, &p->shd,
146 VK_SHADER_STAGE_COMPUTE_BIT,
147 0, sizeof(field), &field);
148 }
149
150 ff_vk_exec_bind_shader(&p->s->vkctx, ec, &p->shd);
151
152 vk->CmdDispatch(ec->buf,
153 FFALIGN(dst->width, p->shd.lg_size[0])/p->shd.lg_size[0],
154 FFALIGN(dst->height, p->shd.lg_size[1])/p->shd.lg_size[1],
155 1);
156
157 ff_vk_exec_submit(&p->s->vkctx, ec);
158 ff_vk_exec_wait(&p->s->vkctx, ec);
159}
160
161static void free_fn(void *priv)
162{
163 VulkanPriv *p = priv;
164 ff_vk_exec_pool_free(&p->s->vkctx, &p->e);
165 ff_vk_shader_free(&p->s->vkctx, &p->shd);
166 for (int i = 0; i < p->nb_data_bufs; i++)
167 ff_vk_free_buf(&p->s->vkctx, &p->data_bufs[i]);
168 av_refstruct_unref(&p->s);
169 av_free(priv);
170}
171
173 const SwsFilterWeights *wd, FFVkBuffer *buf)
174{
175 int err;
176
177 /* Weights */
178 err = ff_vk_create_buf(&s->vkctx, buf,
179 wd->num_weights*sizeof(float) +
180 wd->dst_size*sizeof(int32_t), NULL, NULL,
181 VK_BUFFER_USAGE_UNIFORM_BUFFER_BIT,
182 VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT |
183 VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT);
184 if (err < 0)
185 goto fail;
186
187 float *weights_data;
188 err = ff_vk_map_buffer(&s->vkctx, buf,
189 (uint8_t **)&weights_data, 0);
190 if (err < 0)
191 goto fail;
192 for (int i = 0; i < wd->num_weights; i++)
193 weights_data[i] = (float) wd->weights[i] / SWS_FILTER_SCALE;
194
195 memcpy(weights_data + wd->num_weights,
196 wd->offsets, wd->dst_size*sizeof(int32_t));
197
198 ff_vk_unmap_buffer(&s->vkctx, buf, 1);
199
200 return 0;
201
202fail:
203 ff_vk_free_buf(&p->s->vkctx, buf);
204 return 0;
205}
206
208 const SwsDitherOp *dd, FFVkBuffer *buf)
209{
210 int err;
211
212 int size = (1 << dd->size_log2);
213 err = ff_vk_create_buf(&s->vkctx, buf,
214 size*size*sizeof(float), NULL, NULL,
215 VK_BUFFER_USAGE_UNIFORM_BUFFER_BIT,
216 VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT |
217 VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT);
218 if (err < 0)
219 return err;
220
221 float *dither_data;
222 err = ff_vk_map_buffer(&s->vkctx, buf, (uint8_t **)&dither_data, 0);
223 if (err < 0)
224 goto fail;
225
226 for (int i = 0; i < size; i++) {
227 for (int j = 0; j < size; j++) {
228 const AVRational64 r = dd->matrix[i*size + j];
229 dither_data[i*size + j] = r.num/(float)r.den;
230 }
231 }
232
233 ff_vk_unmap_buffer(&s->vkctx, buf, 1);
234
235 return 0;
236
237fail:
238 ff_vk_free_buf(&p->s->vkctx, buf);
239 return err;
240}
241
242static int create_bufs(FFVulkanOpsCtx *s, VulkanPriv *p, const SwsOpList *ops)
243{
244 int err;
245 p->nb_data_bufs = 0;
246 for (int n = 0; n < ops->num_ops; n++) {
247 const SwsOp *op = &ops->ops[n];
248 if (op->op == SWS_OP_DITHER) {
249 av_assert0(p->nb_data_bufs + 1 <= FF_ARRAY_ELEMS(p->data_bufs));
250 err = create_dither_buf(s, p, &op->dither,
251 &p->data_bufs[p->nb_data_bufs]);
252 if (err < 0)
253 goto fail;
254 p->nb_data_bufs++;
255 } else if (op->op == SWS_OP_FILTER_H || op->op == SWS_OP_FILTER_V) {
256 av_assert0(p->nb_data_bufs + 1 <= FF_ARRAY_ELEMS(p->data_bufs));
257 err = create_filter_buf(s, p, op->filter.kernel,
258 &p->data_bufs[p->nb_data_bufs]);
259 if (err < 0)
260 goto fail;
261 p->nb_data_bufs++;
262 } else if ((op->op == SWS_OP_READ ||
263 op->op == SWS_OP_WRITE) && op->rw.filter.op) {
264 av_assert0(p->nb_data_bufs + 1 <= FF_ARRAY_ELEMS(p->data_bufs));
265 err = create_filter_buf(s, p, op->rw.filter.kernel,
266 &p->data_bufs[p->nb_data_bufs]);
267 if (err < 0)
268 goto fail;
269 p->nb_data_bufs++;
270 }
271 }
272
273 return 0;
274
275fail:
276 for (int i = 0; i < p->nb_data_bufs; i++)
277 ff_vk_free_buf(&p->s->vkctx, &p->data_bufs[i]);
278 return err;
279}
280
281#if HAVE_SPIRV_HEADERS_SPIRV_H || HAVE_SPIRV_UNIFIED1_SPIRV_H
282struct DitherData {
283 int size;
284 int arr_1d_id;
285 int arr_2d_id;
286 int struct_id;
287 int struct_ptr_id;
288 int id;
289 int mask_id;
290 int binding;
291};
292
293struct FilterData {
295 int filter_size;
296 int dst_size;
297 int num_weights;
298
299 int arr_w_in_id;
300 int arr_w_out_id;
301 int arr_o_id;
302 int struct_id;
303 int struct_ptr_id;
304
305 int id; /* buffer ID */
306 int binding; /* descriptor idx in desc set 1 */
307
308 int tap_const_base;
309};
310
311typedef struct SPIRVIDs {
312 int in_vars[3 + MAX_DATA_BUFS + 1];
313
314 int glfn;
315 int ep;
316
317 /* Types */
318 int void_type;
319 int b_type;
320 int u32_type;
321 int i32_type;
322 int f32_type;
323 int void_fn_type;
324
325 /* Define vector types */
326 int bvec2_type;
327 int u32vec2_type;
328 int i32vec2_type;
329
330 int u32vec3_type;
331
332 int u32vec4_type;
333 int f32vec4_type;
334 int f32mat4_type;
335
336 /* Constants */
337 int u32_p;
338 int f32_p;
339 int f32_0;
340 int u32_cid[5];
341
342 int const_ids[128];
343 int nb_const_ids;
344
345 int linear_deco_off[16];
346 int linear_deco_ops[16];
347 int nb_linear_ops;
348
349 struct DitherData dither[MAX_DITHER_BUFS];
350 int dither_ptr_elem_id;
351 int nb_dither_bufs;
352
353 struct FilterData filt[MAX_FILT_BUFS];
354 int filt_o_ptr_id;
355 int nb_filter_bufs;
356
357 int out_img_type;
358 int out_img_array_id;
359
360 int in_img_type;
361 int in_img_array_id;
362
363 /* Pointer types for images */
364 int u32vec3_tptr;
365 int out_img_tptr;
366 int out_img_sptr;
367
368 int in_img_tptr;
369 int in_img_sptr;
370
371 /* Interlaced handling */
372 int interlaced;
373 int push_const_struct_id;
374 int push_const_ptr_id;
375 int push_const_elem_ptr_id;
376 int push_const_var_id;
377 int field_i32;
378} SPIRVIDs;
379
380/* Section 1: Function to define all shader header data, and decorations */
381static void define_shader_header(SwsContext *sws, FFVulkanShader *shd,
382 const SwsOpList *ops, SPICtx *spi, SPIRVIDs *id)
383{
384 spi_OpCapability(spi, SpvCapabilityShader); /* Shader type */
385
386 /* Declare required capabilities */
387 spi_OpCapability(spi, SpvCapabilityInt16);
388 spi_OpCapability(spi, SpvCapabilityInt8);
389 spi_OpCapability(spi, SpvCapabilityImageQuery);
390 spi_OpCapability(spi, SpvCapabilityStorageImageReadWithoutFormat);
391 spi_OpCapability(spi, SpvCapabilityStorageImageWriteWithoutFormat);
392 spi_OpCapability(spi, SpvCapabilityStorageBuffer8BitAccess);
393 /* Import the GLSL set of functions (used for min/max) */
394 id->glfn = spi_OpExtInstImport(spi, "GLSL.std.450");
395
396 /* Next section starts here */
397 spi_OpMemoryModel(spi, SpvAddressingModelLogical, SpvMemoryModelGLSL450);
398
399 /* Entrypoint */
400 id->ep = spi_OpEntryPoint(spi, SpvExecutionModelGLCompute, "main",
401 id->in_vars,
402 3 + id->nb_dither_bufs + id->nb_filter_bufs +
403 (id->interlaced ? 1 : 0));
404 spi_OpExecutionMode(spi, id->ep, SpvExecutionModeLocalSize,
405 shd->lg_size, 3);
406
407 if (id->interlaced) {
408 spi_OpDecorate(spi, id->push_const_struct_id, SpvDecorationBlock);
409 spi_OpMemberDecorate(spi, id->push_const_struct_id, 0,
410 SpvDecorationOffset, 0);
411 }
412
413 /* gl_GlobalInvocationID descriptor decorations */
414 spi_OpDecorate(spi, id->in_vars[0], SpvDecorationBuiltIn,
415 SpvBuiltInGlobalInvocationId);
416
417 /* Input image descriptor decorations */
418 spi_OpDecorate(spi, id->in_vars[1], SpvDecorationNonWritable);
419 spi_OpDecorate(spi, id->in_vars[1], SpvDecorationDescriptorSet, 0);
420 spi_OpDecorate(spi, id->in_vars[1], SpvDecorationBinding, 0);
421
422 /* Output image descriptor decorations */
423 spi_OpDecorate(spi, id->in_vars[2], SpvDecorationNonReadable);
424 spi_OpDecorate(spi, id->in_vars[2], SpvDecorationDescriptorSet, 0);
425 spi_OpDecorate(spi, id->in_vars[2], SpvDecorationBinding, 1);
426
427 for (int i = 0; i < id->nb_dither_bufs; i++) {
428 spi_OpDecorate(spi, id->dither[i].arr_1d_id, SpvDecorationArrayStride,
429 sizeof(float));
430 spi_OpDecorate(spi, id->dither[i].arr_2d_id, SpvDecorationArrayStride,
431 id->dither[i].size*sizeof(float));
432 spi_OpDecorate(spi, id->dither[i].struct_id, SpvDecorationBlock);
433 spi_OpMemberDecorate(spi, id->dither[i].struct_id, 0, SpvDecorationOffset, 0);
434 spi_OpDecorate(spi, id->dither[i].id, SpvDecorationDescriptorSet, 1);
435 spi_OpDecorate(spi, id->dither[i].id, SpvDecorationBinding,
436 id->dither[i].binding);
437 }
438
439 for (int i = 0; i < id->nb_filter_bufs; i++) {
440 struct FilterData *f = &id->filt[i];
441 spi_OpDecorate(spi, f->arr_w_in_id, SpvDecorationArrayStride,
442 sizeof(float));
443 spi_OpDecorate(spi, f->arr_w_out_id, SpvDecorationArrayStride,
444 f->filter_size*sizeof(float));
445 spi_OpDecorate(spi, f->arr_o_id, SpvDecorationArrayStride,
446 sizeof(int32_t));
447 spi_OpDecorate(spi, f->struct_id, SpvDecorationBlock);
448 spi_OpMemberDecorate(spi, f->struct_id, 0, SpvDecorationOffset, 0);
449 spi_OpMemberDecorate(spi, f->struct_id, 1, SpvDecorationOffset,
450 f->num_weights*sizeof(float));
451 spi_OpDecorate(spi, f->id, SpvDecorationDescriptorSet, 1);
452 spi_OpDecorate(spi, f->id, SpvDecorationBinding, f->binding);
453 }
454
455 if (!(sws->flags & SWS_BITEXACT))
456 return;
457
458 /* All linear arithmetic ops must be decorated with NoContraction */
459 for (int n = 0; n < ops->num_ops; n++) {
460 const SwsOp *op = &ops->ops[n];
461 if (op->op != SWS_OP_LINEAR)
462 continue;
463 av_assert0((id->nb_linear_ops + 1) <= FF_ARRAY_ELEMS(id->linear_deco_off));
464
465 int nb_ops = 0;
466 for (int j = 0; j < 4; j++) {
467 nb_ops += !!op->lin.m[j][0].num;
468 nb_ops += op->lin.m[j][0].num && op->lin.m[j][4].num;
469 for (int i = 1; i < 4; i++) {
470 nb_ops += !!op->lin.m[j][i].num;
471 nb_ops += op->lin.m[j][i].num &&
472 (op->lin.m[j][0].num || op->lin.m[j][4].num);
473 }
474 }
475
476 id->linear_deco_off[id->nb_linear_ops] = spi_reserve(spi, nb_ops*4*3);
477 id->linear_deco_ops[id->nb_linear_ops] = nb_ops;
478 id->nb_linear_ops++;
479 }
480}
481
482/* Section 2: Define all types and constants */
483static void define_shader_consts(SwsContext *sws, const SwsOpList *ops,
484 SPICtx *spi, SPIRVIDs *id)
485{
486 /* Define scalar types */
487 id->void_type = spi_OpTypeVoid(spi);
488 id->b_type = spi_OpTypeBool(spi);
489 int u32_type =
490 id->u32_type = spi_OpTypeInt(spi, 32, 0);
491 id->i32_type = spi_OpTypeInt(spi, 32, 1);
492 int f32_type =
493 id->f32_type = spi_OpTypeFloat(spi, 32);
494 id->void_fn_type = spi_OpTypeFunction(spi, id->void_type, NULL, 0);
495
496 /* Define vector types */
497 id->bvec2_type = spi_OpTypeVector(spi, id->b_type, 2);
498 id->u32vec2_type = spi_OpTypeVector(spi, u32_type, 2);
499 id->i32vec2_type = spi_OpTypeVector(spi, id->i32_type, 2);
500
501 id->u32vec3_type = spi_OpTypeVector(spi, u32_type, 3);
502
503 id->u32vec4_type = spi_OpTypeVector(spi, u32_type, 4);
504 id->f32vec4_type = spi_OpTypeVector(spi, f32_type, 4);
505 id->f32mat4_type = spi_OpTypeMatrix(spi, id->f32vec4_type, 4);
506
507 /* Constants */
508 id->u32_p = spi_OpUndef(spi, u32_type);
509 id->f32_p = spi_OpUndef(spi, f32_type);
510 id->f32_0 = spi_OpConstantFloat(spi, f32_type, 0);
511 for (int i = 0; i < 5; i++)
512 id->u32_cid[i] = spi_OpConstantUInt(spi, u32_type, i);
513
514 /* Operation constants */
515 id->nb_const_ids = 0;
516 for (int n = 0; n < ops->num_ops; n++) {
517 /* Make sure there's always enough space for the maximum number of
518 * constants a single operation needs (currently linear, 31 consts). */
519 av_assert0((id->nb_const_ids + 31) <= FF_ARRAY_ELEMS(id->const_ids));
520 const SwsOp *op = &ops->ops[n];
521 switch (op->op) {
522 case SWS_OP_CLEAR:
523 for (int i = 0; i < 4; i++) {
524 if (!SWS_COMP_TEST(op->clear.mask, i))
525 continue;
526 AVRational64 cv = op->clear.value[i];
527 if (op->type == SWS_PIXEL_F32) {
528 float q = (float)cv.num/cv.den;
529 id->const_ids[id->nb_const_ids++] =
530 spi_OpConstantFloat(spi, f32_type, q);
531 } else {
532 av_assert0(cv.den == 1);
533 id->const_ids[id->nb_const_ids++] =
534 spi_OpConstantUInt(spi, u32_type, cv.num);
535 }
536 }
537 break;
538 case SWS_OP_LSHIFT:
539 case SWS_OP_RSHIFT: {
540 int tmp = spi_OpConstantUInt(spi, u32_type, op->shift.amount);
541 tmp = spi_OpConstantComposite(spi, id->u32vec4_type,
542 tmp, tmp, tmp, tmp);
543 id->const_ids[id->nb_const_ids++] = tmp;
544 break;
545 }
546 case SWS_OP_SCALE: {
547 int tmp;
548 if (op->type == SWS_PIXEL_F32) {
549 float q = op->scale.factor.num/(float)op->scale.factor.den;
550 tmp = spi_OpConstantFloat(spi, f32_type, q);
551 tmp = spi_OpConstantComposite(spi, id->f32vec4_type,
552 tmp, tmp, tmp, tmp);
553 } else {
554 av_assert0(op->scale.factor.den == 1);
555 tmp = spi_OpConstantUInt(spi, u32_type, op->scale.factor.num);
556 tmp = spi_OpConstantComposite(spi, id->u32vec4_type,
557 tmp, tmp, tmp, tmp);
558 }
559 id->const_ids[id->nb_const_ids++] = tmp;
560 break;
561 }
562 case SWS_OP_MIN:
563 case SWS_OP_MAX:
564 for (int i = 0; i < 4; i++) {
565 int tmp;
566 AVRational64 cl = op->clamp.limit[i];
567 if (!op->clamp.limit[i].den) {
568 continue;
569 } else if (op->type == SWS_PIXEL_F32) {
570 float q = (float)cl.num/((float)cl.den);
571 tmp = spi_OpConstantFloat(spi, f32_type, q);
572 } else {
573 av_assert0(cl.den == 1);
574 tmp = spi_OpConstantUInt(spi, u32_type, cl.num);
575 }
576 id->const_ids[id->nb_const_ids++] = tmp;
577 }
578 break;
579 case SWS_OP_DITHER:
580 for (int i = 0; i < 4; i++) {
581 if (op->dither.y_offset[i] < 0)
582 continue;
583 int tmp = spi_OpConstantUInt(spi, u32_type, op->dither.y_offset[i]);
584 id->const_ids[id->nb_const_ids++] = tmp;
585 }
586 break;
587 case SWS_OP_LINEAR: {
588 int tmp;
589 float val;
590 for (int i = 0; i < 4; i++) {
591 for (int j = 0; j < 4; j++) {
592 int k = sws->flags & SWS_BITEXACT ? i : j;
593 int l = sws->flags & SWS_BITEXACT ? j : i;
594 val = op->lin.m[k][l].num/(float)op->lin.m[k][l].den;
595 id->const_ids[id->nb_const_ids++] =
596 spi_OpConstantFloat(spi, f32_type, val);
597 }
598 tmp = spi_OpConstantComposite(spi, id->f32vec4_type,
599 id->const_ids[id->nb_const_ids - 4],
600 id->const_ids[id->nb_const_ids - 3],
601 id->const_ids[id->nb_const_ids - 2],
602 id->const_ids[id->nb_const_ids - 1]);
603 id->const_ids[id->nb_const_ids++] = tmp;
604 }
605
606 tmp = spi_OpConstantComposite(spi, id->f32mat4_type,
607 id->const_ids[id->nb_const_ids - 5*4 + 4],
608 id->const_ids[id->nb_const_ids - 5*3 + 4],
609 id->const_ids[id->nb_const_ids - 5*2 + 4],
610 id->const_ids[id->nb_const_ids - 5*1 + 4]);
611 id->const_ids[id->nb_const_ids++] = tmp;
612
613 for (int i = 0; i < 4; i++) {
614 val = op->lin.m[i][4].num/(float)op->lin.m[i][4].den;
615 id->const_ids[id->nb_const_ids++] =
616 spi_OpConstantFloat(spi, f32_type, val);
617 }
618
619 tmp = spi_OpConstantComposite(spi, id->f32vec4_type,
620 id->const_ids[id->nb_const_ids - 4],
621 id->const_ids[id->nb_const_ids - 3],
622 id->const_ids[id->nb_const_ids - 2],
623 id->const_ids[id->nb_const_ids - 1]);
624 id->const_ids[id->nb_const_ids++] = tmp;
625 break;
626 }
627 default:
628 break;
629 }
630 }
631}
632
633/* Section 3: Define bindings */
634static void define_shader_bindings(const SwsOpList *ops, SPICtx *spi, SPIRVIDs *id,
635 int in_img_count, int out_img_count)
636{
637 id->dither_ptr_elem_id = spi_OpTypePointer(spi, SpvStorageClassUniform,
638 id->f32_type);
639
640 struct DitherData *dither = id->dither;
641 for (int i = 0; i < id->nb_dither_bufs; i++) {
642 int size_id = spi_OpConstantUInt(spi, id->u32_type, dither[i].size);
643 dither[i].mask_id = spi_OpConstantUInt(spi, id->u32_type, dither[i].size - 1);
644 spi_OpTypeArray(spi, id->f32_type, dither[i].arr_1d_id, size_id);
645 spi_OpTypeArray(spi, dither[i].arr_1d_id, dither[i].arr_2d_id, size_id);
646 spi_OpTypeStruct(spi, dither[i].struct_id, dither[i].arr_2d_id);
647 dither[i].struct_ptr_id = spi_OpTypePointer(spi, SpvStorageClassUniform,
648 dither[i].struct_id);
649 dither[i].id = spi_OpVariable(spi, dither[i].id, dither[i].struct_ptr_id,
650 SpvStorageClassUniform, 0);
651 }
652
653 /* Filter buffers: struct { float w[dst_size][filter_size]; int o[dst_size]; } */
654 id->filt_o_ptr_id = 0;
655 if (id->nb_filter_bufs)
656 id->filt_o_ptr_id = spi_OpTypePointer(spi, SpvStorageClassUniform,
657 id->i32_type);
658
659 for (int i = 0; i < id->nb_filter_bufs; i++) {
660 struct FilterData *f = &id->filt[i];
661 int fs_id = spi_OpConstantUInt(spi, id->u32_type, f->filter_size);
662 int ds_id = spi_OpConstantUInt(spi, id->u32_type, f->dst_size);
663
664 spi_OpTypeArray(spi, id->f32_type, f->arr_w_in_id, fs_id);
665 spi_OpTypeArray(spi, f->arr_w_in_id, f->arr_w_out_id, ds_id);
666 spi_OpTypeArray(spi, id->i32_type, f->arr_o_id, ds_id);
667 spi_OpTypeStruct(spi, f->struct_id, f->arr_w_out_id, f->arr_o_id);
668 f->struct_ptr_id = spi_OpTypePointer(spi, SpvStorageClassUniform,
669 f->struct_id);
670 f->id = spi_OpVariable(spi, f->id, f->struct_ptr_id,
671 SpvStorageClassUniform, 0);
672
673 /* Signed tap-index constants 0..filter_size-1 (consecutive <id>s) */
674 f->tap_const_base = spi_OpConstantInt(spi, id->i32_type, 0);
675 for (int t = 1; t < f->filter_size; t++)
676 spi_OpConstantInt(spi, id->i32_type, t);
677 }
678
679 const SwsOp *op_w = ff_sws_op_list_output(ops);
680 const SwsOp *op_r = ff_sws_op_list_input(ops);
681
682 /* Define image types for descriptors */
683 id->out_img_type = spi_OpTypeImage(spi,
684 op_w->type == SWS_PIXEL_F32 ?
685 id->f32_type : id->u32_type,
686 2, 0, 0, 0, 2, SpvImageFormatUnknown);
687 id->out_img_array_id = spi_OpTypeArray(spi, id->out_img_type, spi_get_id(spi),
688 id->u32_cid[out_img_count]);
689
690 id->in_img_type = 0;
691 id->in_img_array_id = 0;
692 if (op_r) {
693 /* If the formats match, we have to reuse the types due to SPIR-V not
694 * allowing redundant type defines */
695 int match = ((op_w->type == SWS_PIXEL_F32) ==
696 (op_r->type == SWS_PIXEL_F32));
697 id->in_img_type = match ? id->out_img_type :
698 spi_OpTypeImage(spi,
699 op_r->type == SWS_PIXEL_F32 ?
700 id->f32_type : id->u32_type,
701 2, 0, 0, 0, 2, SpvImageFormatUnknown);
702 id->in_img_array_id = spi_OpTypeArray(spi, id->in_img_type, spi_get_id(spi),
703 id->u32_cid[in_img_count]);
704 }
705
706 /* Pointer types for images */
707 id->u32vec3_tptr = spi_OpTypePointer(spi, SpvStorageClassInput,
708 id->u32vec3_type);
709 id->out_img_tptr = spi_OpTypePointer(spi, SpvStorageClassUniformConstant,
710 id->out_img_array_id);
711 id->out_img_sptr = spi_OpTypePointer(spi, SpvStorageClassUniformConstant,
712 id->out_img_type);
713
714 id->in_img_tptr = 0;
715 id->in_img_sptr = 0;
716 if (op_r) {
717 id->in_img_tptr= spi_OpTypePointer(spi, SpvStorageClassUniformConstant,
718 id->in_img_array_id);
719 id->in_img_sptr= spi_OpTypePointer(spi, SpvStorageClassUniformConstant,
720 id->in_img_type);
721 }
722
723 /* Define inputs */
724 spi_OpVariable(spi, id->in_vars[0], id->u32vec3_tptr,
725 SpvStorageClassInput, 0);
726 if (op_r) {
727 spi_OpVariable(spi, id->in_vars[1], id->in_img_tptr,
728 SpvStorageClassUniformConstant, 0);
729 }
730 spi_OpVariable(spi, id->in_vars[2], id->out_img_tptr,
731 SpvStorageClassUniformConstant, 0);
732
733 if (id->interlaced) {
734 spi_OpTypeStruct(spi, id->push_const_struct_id, id->u32_type);
735 id->push_const_ptr_id = spi_OpTypePointer(spi, SpvStorageClassPushConstant,
736 id->push_const_struct_id);
737 id->push_const_elem_ptr_id = spi_OpTypePointer(spi, SpvStorageClassPushConstant,
738 id->u32_type);
739 spi_OpVariable(spi, id->push_const_var_id, id->push_const_ptr_id,
740 SpvStorageClassPushConstant, 0);
741 }
742}
743
744static int insert_vmat_linear(const SwsOp *op, SPICtx *spi, SPIRVIDs *id,
745 int data, int const_off)
746{
747 data = spi_OpMatrixTimesVector(spi, id->f32vec4_type,
748 id->const_ids[const_off + 4*5],
749 data);
750 return spi_OpFAdd(spi, id->f32vec4_type,
751 id->const_ids[const_off + 4*5 + 1 + 4], data);
752}
753
754static int insert_bitexact_linear(const SwsOp *op, SPICtx *spi, SPIRVIDs *id,
755 int data, int linear_ops_idx, int const_off)
756{
757 int type_s = op->type == SWS_PIXEL_F32 ? id->f32_type : id->u32_type;
758 int type_v = op->type == SWS_PIXEL_F32 ? id->f32vec4_type : id->u32vec4_type;
759
760 int tmp[4];
761 tmp[0] = spi_OpCompositeExtract(spi, type_s, data, 0);
762 tmp[1] = spi_OpCompositeExtract(spi, type_s, data, 1);
763 tmp[2] = spi_OpCompositeExtract(spi, type_s, data, 2);
764 tmp[3] = spi_OpCompositeExtract(spi, type_s, data, 3);
765
766 int off = spi_reserve(spi, 0); /* Current offset */
767 spi->off = id->linear_deco_off[linear_ops_idx];
768 for (int i = 0; i < id->linear_deco_ops[linear_ops_idx]; i++)
769 spi_OpDecorate(spi, spi->id + i, SpvDecorationNoContraction);
770 spi->off = off;
771
772 int res[4];
773 for (int j = 0; j < 4; j++) {
774 res[j] = op->type == SWS_PIXEL_F32 ? id->f32_0 : id->u32_cid[0];
775 if (op->lin.m[j][0].num)
776 res[j] = spi_OpFMul(spi, type_s, tmp[0],
777 id->const_ids[const_off + j*5 + 0]);
778
779 if (op->lin.m[j][0].num && op->lin.m[j][4].num)
780 res[j] = spi_OpFAdd(spi, type_s,
781 id->const_ids[const_off + 4*5 + 1 + j], res[j]);
782 else if (op->lin.m[j][4].num)
783 res[j] = id->const_ids[const_off + 4*5 + 1 + j];
784
785 for (int i = 1; i < 4; i++) {
786 if (!op->lin.m[j][i].num)
787 continue;
788
789 int v = spi_OpFMul(spi, type_s, tmp[i],
790 id->const_ids[const_off + j*5 + i]);
791 if (op->lin.m[j][0].num || op->lin.m[j][4].num)
792 res[j] = spi_OpFAdd(spi, type_s, res[j], v);
793 else
794 res[j] = v;
795 }
796 }
797
798 return spi_OpCompositeConstruct(spi, type_v,
799 res[0], res[1], res[2], res[3]);
800}
801
802static int read_filtered(SPICtx *spi, SPIRVIDs *id, const SwsOpList *ops,
803 const SwsOp *op, const struct FilterData *f,
804 const int *in_img, int gid, int gi2)
805{
806 const int is_h = f->filter == SWS_OP_FILTER_H;
807 const int src_interlaced = ops->src.interlaced;
808
809 const int src_float = op->type == SWS_PIXEL_F32;
810 const int read_vtype = src_float ? id->f32vec4_type : id->u32vec4_type;
811
812 /* Buffer array index along the filtered axis: pos.x (H) or pos.y (V) */
813 int axis = spi_OpCompositeExtract(spi, id->u32_type, gid, is_h ? 0 : 1);
814
815 /* int o = filter_o[axis]; */
816 int o_ptr = spi_OpAccessChain(spi, id->filt_o_ptr_id, f->id,
817 id->u32_cid[1], axis);
818 int o = spi_OpLoad(spi, id->i32_type, o_ptr, SpvMemoryAccessMaskNone, 0);
819
820 /* Signed pixel position, for the non-filtered coordinate axis */
821 int pos_x = spi_OpCompositeExtract(spi, id->i32_type, gi2, 0);
822 int pos_y = spi_OpCompositeExtract(spi, id->i32_type, gi2, 1);
823
824 /* For interlaced horizontal filtering, the y coordinate of every tap is
825 * the (constant) destination y mapped into the source image. */
826 if (src_interlaced && is_h) {
827 pos_y = spi_OpShiftLeftLogical(spi, id->i32_type, pos_y, id->u32_cid[1]);
828 pos_y = spi_OpIAdd(spi, id->i32_type, pos_y, id->field_i32);
829 }
830
831 /* Accumulators, initialized to zero */
832 int acc_s[4] = { id->f32_0, id->f32_0, id->f32_0, id->f32_0 };
833 int acc_v = id->f32_0;
834 if (op->rw.mode == SWS_RW_PACKED)
835 acc_v = spi_OpCompositeConstruct(spi, id->f32vec4_type,
836 id->f32_0, id->f32_0,
837 id->f32_0, id->f32_0);
838
839 for (int t = 0; t < f->filter_size; t++) {
840 /* float w = filter_w[axis][t]; */
841 int w_ptr = spi_OpAccessChain(spi, id->dither_ptr_elem_id, f->id,
842 id->u32_cid[0], axis,
843 f->tap_const_base + t);
844 int w = spi_OpLoad(spi, id->f32_type, w_ptr,
845 SpvMemoryAccessMaskNone, 0);
846
847 /* Source coordinate, filtered axis offset by the tap index */
848 int c = t ? spi_OpIAdd(spi, id->i32_type, o, f->tap_const_base + t) : o;
849 /* For interlaced vertical filtering, the per-tap source row is
850 * field-local; map it to the actual image row. */
851 if (src_interlaced && !is_h) {
852 c = spi_OpShiftLeftLogical(spi, id->i32_type, c, id->u32_cid[1]);
853 c = spi_OpIAdd(spi, id->i32_type, c, id->field_i32);
854 }
855 int coord = is_h ?
856 spi_OpCompositeConstruct(spi, id->i32vec2_type, c, pos_y) :
857 spi_OpCompositeConstruct(spi, id->i32vec2_type, pos_x, c);
858
859 if (op->rw.mode == SWS_RW_PACKED) {
860 int px = spi_OpImageRead(spi, read_vtype,
861 in_img[ops->plane_src[0]], coord,
862 SpvImageOperandsMaskNone);
863 if (!src_float)
864 px = spi_OpConvertUToF(spi, id->f32vec4_type, px);
865 px = spi_OpVectorTimesScalar(spi, id->f32vec4_type, px, w);
866 acc_v = spi_OpFAdd(spi, id->f32vec4_type, acc_v, px);
867 } else {
868 for (int e = 0; e < op->rw.elems; e++) {
869 int px = spi_OpImageRead(spi, read_vtype,
870 in_img[ops->plane_src[e]], coord,
871 SpvImageOperandsMaskNone);
872 if (src_float) {
873 px = spi_OpCompositeExtract(spi, id->f32_type, px, 0);
874 } else {
875 px = spi_OpCompositeExtract(spi, id->u32_type, px, 0);
876 px = spi_OpConvertUToF(spi, id->f32_type, px);
877 }
878 px = spi_OpFMul(spi, id->f32_type, w, px);
879 acc_s[e] = spi_OpFAdd(spi, id->f32_type, acc_s[e], px);
880 }
881 }
882 }
883
884 if (op->rw.mode == SWS_RW_PACKED)
885 return acc_v;
886 return spi_OpCompositeConstruct(spi, id->f32vec4_type,
887 acc_s[0], acc_s[1], acc_s[2], acc_s[3]);
888}
889
890/* Plane indices refer to actual frame planes, so the image handle arrays
891 * have to cover the highest plane referenced, not just the plane count. */
892static int rw_op_img_count(const SwsOp *op, const uint8_t *planes)
893{
894 int count = 0;
895 for (int i = 0; i < ff_sws_rw_op_planes(op); i++)
896 count = FFMAX(count, planes[i] + 1);
897 return count;
898}
899
900static int add_ops_spirv(SwsContext *sws, VulkanPriv *p, FFVulkanOpsCtx *s,
901 const SwsOpList *ops, FFVulkanShader *shd)
902{
903 uint8_t spvbuf[1024*16];
904 SPICtx spi_context = { 0 }, *spi = &spi_context;
905 SPIRVIDs spid_data = { 0 }, *id = &spid_data;
906 spi_init(spi, spvbuf, sizeof(spvbuf));
907
908 id->interlaced = ops->src.interlaced || ops->dst.interlaced;
909 p->interlaced = id->interlaced;
910
911 ff_vk_shader_load(shd, VK_SHADER_STAGE_COMPUTE_BIT, NULL,
912 (uint32_t []) { 32, 32, 1 }, 0);
913 shd->precompiled = 0;
914
915 if (id->interlaced)
916 ff_vk_shader_add_push_const(shd, 0, sizeof(uint32_t),
917 VK_SHADER_STAGE_COMPUTE_BIT);
918
919 /* Image ops, to determine types */
920 const SwsOp *op_w = ff_sws_op_list_output(ops);
921 int out_img_count = rw_op_img_count(op_w, ops->plane_dst);
922 p->dst_rep = op_w->type == SWS_PIXEL_F32 ? FF_VK_REP_FLOAT : FF_VK_REP_UINT;
923
924 const SwsOp *op_r = ff_sws_op_list_input(ops);
925 int in_img_count = op_r ? rw_op_img_count(op_r, ops->plane_src) : 0;
926 if (op_r)
927 p->src_rep = op_r->type == SWS_PIXEL_F32 ? FF_VK_REP_FLOAT : FF_VK_REP_UINT;
928
930 {
931 .type = VK_DESCRIPTOR_TYPE_STORAGE_IMAGE,
932 .stages = VK_SHADER_STAGE_COMPUTE_BIT,
933 .elems = 4,
934 },
935 {
936 .type = VK_DESCRIPTOR_TYPE_STORAGE_IMAGE,
937 .stages = VK_SHADER_STAGE_COMPUTE_BIT,
938 .elems = 4,
939 },
940 };
941 ff_vk_shader_add_descriptor_set(&s->vkctx, shd, desc_set, 2, 0);
942
943 /* Create dither buffers */
944 int err = create_bufs(s, p, ops);
945 if (err < 0)
946 return err;
947
948 /* Entrypoint inputs; gl_GlobalInvocationID, input and output images, dither */
949 id->in_vars[0] = spi_get_id(spi);
950 id->in_vars[1] = spi_get_id(spi);
951 id->in_vars[2] = spi_get_id(spi);
952
953 /* Create dither and filter buffer descriptor set. Both are collected in
954 * op order, so the bindings match the buffer order from create_bufs().*/
955 id->nb_dither_bufs = 0;
956 id->nb_filter_bufs = 0;
957 int nb_data_bufs = 0;
958 for (int n = 0; n < ops->num_ops; n++) {
959 const SwsOp *op = &ops->ops[n];
960 int var_id = 0;
961
962 if (op->op == SWS_OP_DITHER) {
963 if (id->nb_dither_bufs >= MAX_DITHER_BUFS)
964 return AVERROR(ENOTSUP);
965 struct DitherData *d = &id->dither[id->nb_dither_bufs++];
966 d->size = 1 << op->dither.size_log2;
967 d->arr_1d_id = spi_get_id(spi);
968 d->arr_2d_id = spi_get_id(spi);
969 d->struct_id = spi_get_id(spi);
970 d->id = spi_get_id(spi);
971 d->binding = nb_data_bufs;
972 var_id = d->id;
973 } else if (op->op == SWS_OP_READ && op->rw.filter.op) {
974 if (id->nb_filter_bufs >= MAX_FILT_BUFS)
975 return AVERROR(ENOTSUP);
976 const SwsFilterWeights *wd = op->rw.filter.kernel;
977 struct FilterData *f = &id->filt[id->nb_filter_bufs++];
978 f->filter = op->rw.filter.op;
979 f->filter_size = wd->filter_size;
980 f->dst_size = wd->dst_size;
981 f->num_weights = wd->num_weights;
982 f->arr_w_in_id = spi_get_id(spi);
983 f->arr_w_out_id = spi_get_id(spi);
984 f->arr_o_id = spi_get_id(spi);
985 f->struct_id = spi_get_id(spi);
986 f->id = spi_get_id(spi);
987 f->binding = nb_data_bufs;
988 var_id = f->id;
989 } else {
990 continue;
991 }
992
993 id->in_vars[3 + nb_data_bufs] = var_id;
994 desc_set[nb_data_bufs++] = (FFVulkanDescriptorSetBinding) {
995 .type = VK_DESCRIPTOR_TYPE_UNIFORM_BUFFER,
996 .stages = VK_SHADER_STAGE_COMPUTE_BIT,
997 };
998 }
999 if (nb_data_bufs)
1000 ff_vk_shader_add_descriptor_set(&s->vkctx, shd, desc_set,
1001 nb_data_bufs, 1);
1002
1003 if (id->interlaced) {
1004 id->push_const_struct_id = spi_get_id(spi);
1005 id->push_const_var_id = spi_get_id(spi);
1006 id->in_vars[3 + id->nb_dither_bufs + id->nb_filter_bufs] =
1007 id->push_const_var_id;
1008 }
1009
1010 /* Define shader header sections */
1011 define_shader_header(sws, shd, ops, spi, id);
1012 define_shader_consts(sws, ops, spi, id);
1013 define_shader_bindings(ops, spi, id, in_img_count, out_img_count);
1014
1015 /* Main function starts here */
1016 spi_OpFunction(spi, id->ep, id->void_type, 0, id->void_fn_type);
1017 spi_OpLabel(spi, spi_get_id(spi));
1018
1019 /* Load input image handles */
1020 int in_img[4] = { 0 };
1021 for (int i = 0; i < in_img_count; i++) {
1022 /* Deref array and then the pointer */
1023 int img = spi_OpAccessChain(spi, id->in_img_sptr,
1024 id->in_vars[1], id->u32_cid[i]);
1025 in_img[i] = spi_OpLoad(spi, id->in_img_type, img,
1026 SpvMemoryAccessMaskNone, 0);
1027 }
1028
1029 /* Load output image handles */
1030 int out_img[4] = { 0 };
1031 for (int i = 0; i < out_img_count; i++) {
1032 int img = spi_OpAccessChain(spi, id->out_img_sptr,
1033 id->in_vars[2], id->u32_cid[i]);
1034 out_img[i] = spi_OpLoad(spi, id->out_img_type, img,
1035 SpvMemoryAccessMaskNone, 0);
1036 }
1037
1038 /* Load gl_GlobalInvocationID */
1039 int gid = spi_OpLoad(spi, id->u32vec3_type, id->in_vars[0],
1040 SpvMemoryAccessMaskNone, 0);
1041
1042 /* ivec2(gl_GlobalInvocationID.xy) */
1043 gid = spi_OpVectorShuffle(spi, id->u32vec2_type, gid, gid, 0, 1);
1044 int gi2 = spi_OpBitcast(spi, id->i32vec2_type, gid);
1045
1046 /* For interlaced sources/destinations the shader operates on field-local
1047 * coordinates, while images contain the full frame. Map the y axis to the
1048 * actual image row: image_y = field_y * 2 + field. */
1049 int dst_gid = gid, dst_gi2 = gi2;
1050 int src_gid = gid;
1051 if (id->interlaced) {
1052 int field_u32_ptr = spi_OpAccessChain(spi, id->push_const_elem_ptr_id,
1053 id->push_const_var_id,
1054 id->u32_cid[0]);
1055 int field_u32 = spi_OpLoad(spi, id->u32_type, field_u32_ptr,
1056 SpvMemoryAccessMaskNone, 0);
1057 id->field_i32 = spi_OpBitcast(spi, id->i32_type, field_u32);
1058
1059 int img_y_i32 = spi_OpShiftLeftLogical(spi, id->i32_type,
1060 spi_OpCompositeExtract(spi, id->i32_type, gi2, 1),
1061 id->u32_cid[1]);
1062 img_y_i32 = spi_OpIAdd(spi, id->i32_type, img_y_i32, id->field_i32);
1063
1064 int gi2_x = spi_OpCompositeExtract(spi, id->i32_type, gi2, 0);
1065 int mapped_gi2 = spi_OpCompositeConstruct(spi, id->i32vec2_type,
1066 gi2_x, img_y_i32);
1067 int mapped_gid = spi_OpBitcast(spi, id->u32vec2_type, mapped_gi2);
1068
1069 if (ops->src.interlaced)
1070 src_gid = mapped_gid;
1071 if (ops->dst.interlaced) {
1072 dst_gid = mapped_gid;
1073 dst_gi2 = mapped_gi2;
1074 }
1075 }
1076
1077 /* imageSize(out_img[0]); */
1078 int img1_s = spi_OpImageQuerySize(spi, id->i32vec2_type, out_img[0]);
1079 int scmp = spi_OpSGreaterThanEqual(spi, id->bvec2_type, dst_gi2, img1_s);
1080 scmp = spi_OpAny(spi, id->b_type, scmp);
1081
1082 /* if (out of bounds) return */
1083 int quit_label = spi_get_id(spi), merge_label = spi_get_id(spi);
1084 spi_OpSelectionMerge(spi, merge_label, SpvSelectionControlMaskNone);
1085 spi_OpBranchConditional(spi, scmp, quit_label, merge_label, 0);
1086
1087 spi_OpLabel(spi, quit_label);
1088 spi_OpReturn(spi); /* Quit if out of bounds here */
1089 spi_OpLabel(spi, merge_label);
1090
1091 /* Initialize main data state */
1092 int data;
1093 if (ops->ops[0].type == SWS_PIXEL_F32)
1094 data = spi_OpCompositeConstruct(spi, id->f32vec4_type,
1095 id->f32_p, id->f32_p,
1096 id->f32_p, id->f32_p);
1097 else
1098 data = spi_OpCompositeConstruct(spi, id->u32vec4_type,
1099 id->u32_p, id->u32_p,
1100 id->u32_p, id->u32_p);
1101
1102 /* Keep track of which constant/buffer to use */
1103 int nb_const_ids = 0;
1104 int nb_dither_bufs = 0;
1105 int nb_linear_ops = 0;
1106 int nb_filter_used = 0;
1107
1108 /* Operations */
1109 for (int n = 0; n < ops->num_ops; n++) {
1110 const SwsOp *op = &ops->ops[n];
1111 SwsPixelType cur_type = op->op == SWS_OP_CONVERT ?
1112 op->convert.to : op->type;
1113 int type_v = cur_type == SWS_PIXEL_F32 ?
1114 id->f32vec4_type : id->u32vec4_type;
1115 int type_s = cur_type == SWS_PIXEL_F32 ?
1116 id->f32_type : id->u32_type;
1117 int uid = cur_type == SWS_PIXEL_F32 ?
1118 id->f32_p : id->u32_p;
1119
1120 switch (op->op) {
1121 case SWS_OP_READ:
1122 if (op->rw.frac) {
1123 return AVERROR(ENOTSUP);
1124 } else if (op->rw.filter.op) {
1125 av_assert0(op->rw.mode != SWS_RW_PALETTE);
1126 data = read_filtered(spi, id, ops, op,
1127 &id->filt[nb_filter_used++],
1128 in_img, gid, gi2);
1129 } else if (op->rw.mode == SWS_RW_PACKED) {
1130 data = spi_OpImageRead(spi, type_v, in_img[ops->plane_src[0]],
1131 src_gid, SpvImageOperandsMaskNone);
1132 } else if (op->rw.mode == SWS_RW_PLANAR) {
1133 int tmp[4] = { uid, uid, uid, uid };
1134 for (int i = 0; i < op->rw.elems; i++) {
1135 tmp[i] = spi_OpImageRead(spi, type_v,
1136 in_img[ops->plane_src[i]], src_gid,
1137 SpvImageOperandsMaskNone);
1138 tmp[i] = spi_OpCompositeExtract(spi, type_s, tmp[i], 0);
1139 }
1140 data = spi_OpCompositeConstruct(spi, type_v,
1141 tmp[0], tmp[1], tmp[2], tmp[3]);
1142 } else {
1143 return AVERROR(ENOTSUP);
1144 }
1145 break;
1146 case SWS_OP_WRITE:
1147 if (op->rw.frac || op->rw.filter.op) {
1148 return AVERROR(ENOTSUP);
1149 } else if (op->rw.mode == SWS_RW_PACKED) {
1150 spi_OpImageWrite(spi, out_img[ops->plane_dst[0]], dst_gid, data,
1151 SpvImageOperandsMaskNone);
1152 } else {
1153 for (int i = 0; i < op->rw.elems; i++) {
1154 int tmp = spi_OpCompositeExtract(spi, type_s, data, i);
1155 tmp = spi_OpCompositeConstruct(spi, type_v, tmp, tmp, tmp, tmp);
1156 spi_OpImageWrite(spi, out_img[ops->plane_dst[i]], dst_gid, tmp,
1157 SpvImageOperandsMaskNone);
1158 }
1159 }
1160 break;
1161 case SWS_OP_CLEAR:
1162 for (int i = 0; i < 4; i++) {
1163 if (!SWS_COMP_TEST(op->clear.mask, i))
1164 continue;
1165 data = spi_OpCompositeInsert(spi, type_v,
1166 id->const_ids[nb_const_ids++],
1167 data, i);
1168 }
1169 break;
1170 case SWS_OP_SWIZZLE:
1171 data = spi_OpVectorShuffle(spi, type_v, data, data,
1172 op->swizzle.in[0],
1173 op->swizzle.in[1],
1174 op->swizzle.in[2],
1175 op->swizzle.in[3]);
1176 break;
1177 case SWS_OP_CONVERT:
1178 if (op->type == SWS_PIXEL_F32 && type_s == id->u32_type)
1179 data = spi_OpConvertFToU(spi, type_v, data);
1180 else if (op->type != SWS_PIXEL_F32 && type_s == id->f32_type)
1181 data = spi_OpConvertUToF(spi, type_v, data);
1182 break;
1183 case SWS_OP_LSHIFT:
1184 data = spi_OpShiftLeftLogical(spi, type_v, data,
1185 id->const_ids[nb_const_ids++]);
1186 break;
1187 case SWS_OP_RSHIFT:
1188 data = spi_OpShiftRightLogical(spi, type_v, data,
1189 id->const_ids[nb_const_ids++]);
1190 break;
1191 case SWS_OP_SCALE:
1192 if (op->type == SWS_PIXEL_F32)
1193 data = spi_OpFMul(spi, type_v, data,
1194 id->const_ids[nb_const_ids++]);
1195 else
1196 data = spi_OpIMul(spi, type_v, data,
1197 id->const_ids[nb_const_ids++]);
1198 break;
1199 case SWS_OP_MIN:
1200 case SWS_OP_MAX: {
1201 int t = op->type == SWS_PIXEL_F32 ?
1202 op->op == SWS_OP_MIN ? GLSLstd450FMin : GLSLstd450FMax :
1203 op->op == SWS_OP_MIN ? GLSLstd450UMin : GLSLstd450UMax;
1204 for (int i = 0; i < 4; i++) {
1205 if (!op->clamp.limit[i].den)
1206 continue;
1207 int tmp = spi_OpCompositeExtract(spi, type_s, data, i);
1208 tmp = spi_OpExtInst(spi, type_s, id->glfn, t,
1209 tmp, id->const_ids[nb_const_ids++]);
1210 data = spi_OpCompositeInsert(spi, type_v, tmp, data, i);
1211 }
1212 break;
1213 }
1214 case SWS_OP_DITHER: {
1215 int did = nb_dither_bufs++;
1216 int x_id = spi_OpCompositeExtract(spi, id->u32_type, gid, 0);
1217 int y_pos = spi_OpCompositeExtract(spi, id->u32_type, gid, 1);
1218 x_id = spi_OpBitwiseAnd(spi, id->u32_type, x_id,
1219 id->dither[did].mask_id);
1220 for (int i = 0; i < 4; i++) {
1221 if (op->dither.y_offset[i] < 0)
1222 continue;
1223
1224 int y_id = spi_OpIAdd(spi, id->u32_type, y_pos,
1225 id->const_ids[nb_const_ids++]);
1226 y_id = spi_OpBitwiseAnd(spi, id->u32_type, y_id,
1227 id->dither[did].mask_id);
1228
1229 int ptr = spi_OpAccessChain(spi, id->dither_ptr_elem_id,
1230 id->dither[did].id, id->u32_cid[0],
1231 y_id, x_id);
1232 int val = spi_OpLoad(spi, id->f32_type, ptr,
1233 SpvMemoryAccessMaskNone, 0);
1234
1235 int tmp = spi_OpCompositeExtract(spi, type_s, data, i);
1236 tmp = spi_OpFAdd(spi, type_s, tmp, val);
1237 data = spi_OpCompositeInsert(spi, type_v, tmp, data, i);
1238 }
1239 break;
1240 }
1241 case SWS_OP_LINEAR: {
1242 if (op->type != SWS_PIXEL_F32)
1243 return AVERROR(ENOTSUP);
1244 if (sws->flags & SWS_BITEXACT)
1245 data = insert_bitexact_linear(op, spi, id, data, nb_linear_ops, nb_const_ids);
1246 else
1247 data = insert_vmat_linear(op, spi, id, data, nb_const_ids);
1248 nb_linear_ops++;
1249 nb_const_ids += 5*5 + 1;
1250 break;
1251 }
1252 case SWS_OP_UNPACK:
1253 if (ops->src.format == AV_PIX_FMT_X2BGR10)
1254 data = spi_OpVectorShuffle(spi, type_v, data, data, 3, 2, 1, 0);
1255 else
1256 data = spi_OpVectorShuffle(spi, type_v, data, data, 3, 0, 1, 2);
1257 break;
1258 case SWS_OP_PACK:
1259 if (ops->dst.format == AV_PIX_FMT_X2BGR10)
1260 data = spi_OpVectorShuffle(spi, type_v, data, data, 3, 2, 1, 0);
1261 else
1262 data = spi_OpVectorShuffle(spi, type_v, data, data, 1, 2, 3, 0);
1263 break;
1264 default:
1265 return AVERROR(ENOTSUP);
1266 }
1267 }
1268
1269 /* Return and finalize */
1270 spi_OpReturn(spi);
1271 spi_OpFunctionEnd(spi);
1272
1273 int len = spi_end(spi);
1274 if (len < 0)
1275 return AVERROR_INVALIDDATA;
1276
1277 return ff_vk_shader_link(&s->vkctx, shd, spvbuf, len, "main");
1278}
1279#endif
1280
1281static int compile(SwsContext *sws, const SwsOpList *ops, SwsCompiledOp *out)
1282{
1283 int err;
1284 SwsInternal *c = sws_internal(sws);
1285 FFVulkanOpsCtx *s = c->hw_priv;
1286 if (!s)
1287 return AVERROR(ENOTSUP);
1288
1289 VulkanPriv *p = av_mallocz(sizeof(*p));
1290 if (!p)
1291 return AVERROR(ENOMEM);
1292 p->s = av_refstruct_ref(c->hw_priv);
1293
1294 err = ff_vk_exec_pool_init(&s->vkctx, s->qf, &p->e, 1,
1295 0, 0, 0, NULL);
1296 if (err < 0)
1297 goto fail;
1298
1299 err = AVERROR(ENOTSUP);
1300#if HAVE_SPIRV_HEADERS_SPIRV_H || HAVE_SPIRV_UNIFIED1_SPIRV_H
1301 err = add_ops_spirv(sws, p, s, ops, &p->shd);
1302#endif
1303 if (err < 0)
1304 goto fail;
1305
1306 err = ff_vk_shader_register_exec(&s->vkctx, &p->e, &p->shd);
1307 if (err < 0)
1308 goto fail;
1309
1310 for (int i = 0; i < p->nb_data_bufs; i++)
1311 ff_vk_shader_update_desc_buffer(&s->vkctx, &p->e.contexts[0], &p->shd,
1312 1, i, 0, &p->data_bufs[i],
1313 0, VK_WHOLE_SIZE, VK_FORMAT_UNDEFINED);
1314
1315 *out = (SwsCompiledOp) {
1316 .opaque = true,
1317 .func_opaque = process,
1318 .priv = p,
1319 .free = free_fn,
1320 };
1321
1322 return 0;
1323
1324fail:
1325 free_fn(p);
1326 return err;
1327}
1328
1329#if HAVE_SPIRV_HEADERS_SPIRV_H || HAVE_SPIRV_UNIFIED1_SPIRV_H
1330static int compile_spirv(SwsContext *sws, const SwsOpList *ops,
1332{
1333 return compile(sws, ops, out);
1334}
1335
1336const SwsOpBackend backend_spirv = {
1337 .name = "spirv",
1338 .flags = SWS_BACKEND_SPIRV,
1339 .compile = compile_spirv,
1340 .hw_format = AV_PIX_FMT_VULKAN,
1341};
1342#endif
uint8_t ptrdiff_t const uint8_t ptrdiff_t int intptr_t intptr_t int int16_t * dst
Definition dsp.h:87
static double val(void *priv, double ch)
Definition aeval.c:77
static const int8_t filt[NUMTAPS *2]
Definition af_earwax.c:40
static FILE * out
int32_t
#define av_assert0(cond)
assert() equivalent, that is always enabled.
Definition avassert.h:42
#define i(width, name, range_min, range_max)
Definition cbs_h264.c:63
#define f(width, name)
Definition cbs_vp8.c:236
#define s(width, name)
Definition cbs_vp9.c:198
#define NULL
Definition coverity.c:32
enum AVCodecID id
Definition dts2pts.c:607
#define AV_NUM_DATA_POINTERS
Definition frame.h:480
#define fail
Definition test.h:479
#define AVERROR_INVALIDDATA
Invalid data found when processing input.
Definition error.h:61
#define AVERROR(e)
Definition error.h:45
#define AV_LOG_ERROR
Something went wrong and cannot losslessly be recovered.
Definition log.h:210
@ SWS_BACKEND_SPIRV
Vulkan SPIR-V backend.
Definition swscale.h:120
@ SWS_BITEXACT
Definition swscale.h:178
#define r
Definition input.c:42
static int op(uint8_t **dst, const uint8_t *dst_end, GetByteContext *gb, int pixel, int count, int *x, int width, int linesize)
Perform decode operation.
Definition anm.c:76
enum TrimAxis axis
Definition trim.c:49
void ff_vk_shader_update_img_array(FFVulkanContext *s, FFVkExecContext *e, FFVulkanShader *shd, AVFrame *f, VkImageView *views, int set, int binding, VkImageLayout layout, VkSampler sampler)
Update a descriptor in a buffer with an image array.
Definition vulkan.c:2690
int ff_vk_shader_load(FFVulkanShader *shd, VkPipelineStageFlags stage, VkSpecializationInfo *spec, uint32_t wg_size[3], uint32_t required_subgroup_size)
Initialize a shader object.
Definition vulkan.c:2262
void ff_vk_shader_add_descriptor_set(FFVulkanContext *s, FFVulkanShader *shd, const FFVulkanDescriptorSetBinding *desc, int nb, int singular)
Add descriptor to a shader.
Definition vulkan.c:2521
void ff_vk_exec_pool_free(FFVulkanContext *s, FFVkExecPool *pool)
Definition vulkan.c:335
int ff_vk_create_buf(FFVulkanContext *s, FFVkBuffer *buf, size_t size, void *pNext, void *alloc_pNext, VkBufferUsageFlags usage, VkMemoryPropertyFlagBits flags)
Definition vulkan.c:1204
int ff_vk_exec_pool_init(FFVulkanContext *s, AVVulkanDeviceQueueFamily *qf, FFVkExecPool *pool, int nb_contexts, int nb_queries, VkQueryType query_type, int query_64bit, const void *query_create_pnext)
Allocates/frees an execution pool.
Definition vulkan.c:399
void ff_vk_exec_wait(FFVulkanContext *s, FFVkExecContext *e)
Definition vulkan.c:649
int ff_vk_shader_add_push_const(FFVulkanShader *shd, int offset, int size, VkShaderStageFlagBits stage)
Add/update push constants for execution.
Definition vulkan.c:1631
void ff_vk_uninit(FFVulkanContext *s)
Frees main context.
Definition vulkan.c:2779
int ff_vk_init(FFVulkanContext *s, void *log_parent, AVBufferRef *device_ref, AVBufferRef *frames_ref)
Initializes the AVClass, in case this context is not used as the main user's context.
Definition vulkan.c:2795
void ff_vk_free_buf(FFVulkanContext *s, FFVkBuffer *buf)
Definition vulkan.c:1418
void ff_vk_frame_barrier(FFVulkanContext *s, FFVkExecContext *e, AVFrame *pic, VkImageMemoryBarrier2 *bar, int *nb_bar, VkPipelineStageFlags2 src_stage, VkPipelineStageFlags2 dst_stage, VkAccessFlagBits2 new_access, VkImageLayout new_layout, uint32_t new_qf)
Definition vulkan.c:2216
int ff_vk_exec_start(FFVulkanContext *s, FFVkExecContext *e)
Start/submit/wait an execution.
Definition vulkan.c:664
int ff_vk_create_imageviews(FFVulkanContext *s, FFVkExecContext *e, VkImageView views[AV_NUM_DATA_POINTERS], AVFrame *f, enum FFVkShaderRepFormat rep_fmt)
Create an imageview and add it as a dependency to an execution.
Definition vulkan.c:2147
void ff_vk_shader_free(FFVulkanContext *s, FFVulkanShader *shd)
Free a shader.
Definition vulkan.c:2757
int ff_vk_shader_register_exec(FFVulkanContext *s, FFVkExecPool *pool, FFVulkanShader *shd)
Register a shader with an exec pool.
Definition vulkan.c:2555
FFVkExecContext * ff_vk_exec_get(FFVulkanContext *s, FFVkExecPool *pool)
Retrieve an execution pool.
Definition vulkan.c:626
int ff_vk_exec_submit(FFVulkanContext *s, FFVkExecContext *e)
Definition vulkan.c:987
void ff_vk_exec_bind_shader(FFVulkanContext *s, FFVkExecContext *e, const FFVulkanShader *shd)
Bind a shader.
Definition vulkan.c:2739
int ff_vk_shader_update_desc_buffer(FFVulkanContext *s, FFVkExecContext *e, FFVulkanShader *shd, int set, int bind, int elem, FFVkBuffer *buf, VkDeviceSize offset, VkDeviceSize len, VkFormat fmt)
Update a descriptor in a buffer with a buffer.
Definition vulkan.c:2703
AVVulkanDeviceQueueFamily * ff_vk_qf_find(FFVulkanContext *s, VkQueueFlagBits dev_family, VkVideoCodecOperationFlagBitsKHR vid_ops)
Chooses an appropriate QF.
Definition vulkan.c:320
int ff_vk_shader_link(FFVulkanContext *s, FFVulkanShader *shd, const char *spirv, size_t spirv_len, const char *entrypoint)
Link a shader into an executable.
Definition vulkan.c:2425
int ff_vk_exec_add_dep_frame(FFVulkanContext *s, FFVkExecContext *e, AVFrame *f, VkPipelineStageFlagBits2 wait_stage, VkPipelineStageFlagBits2 signal_stage)
Definition vulkan.c:885
void ff_vk_shader_update_push_const(FFVulkanContext *s, FFVkExecContext *e, FFVulkanShader *shd, VkShaderStageFlagBits stage, int offset, size_t size, void *src)
Update push constant in a shader.
Definition vulkan.c:2729
@ SWS_FILTER_SCALE
14-bit coefficients are picked to fit comfortably within int16_t for efficient SIMD processing (e....
Definition filters.h:40
static const struct @257111027162314367033347246032313251342043035002 planes[]
uint8_t w
Definition llvidencdsp.c:39
#define FFMAX(a, b)
Definition macros.h:47
#define FFALIGN(x, a)
Definition macros.h:78
Memory handling functions.
static const char * obj
Definition mscl.c:57
const char data[16]
Definition mxf.c:149
uint8_t interlaced
Definition mxfenc.c:2336
UID uid
Definition mxfenc.c:2488
IDirect3DDxgiInterfaceAccess _COM_Outptr_ void ** p
const SwsOp * ff_sws_op_list_input(const SwsOpList *ops)
Returns the input operation for a given op list, or NULL if there is none (e.g.
Definition ops.c:708
int ff_sws_rw_op_planes(const SwsOp *op)
Return the number of planes involved in a read/write operation.
Definition ops.c:133
const SwsOp * ff_sws_op_list_output(const SwsOpList *ops)
Returns the output operation for a given op list, or NULL if there is none.
Definition ops.c:717
SwsOpType
Copyright (C) 2025 Niklas Haas.
Definition ops.h:36
@ SWS_OP_RSHIFT
Definition ops.h:49
@ SWS_OP_SWIZZLE
Definition ops.h:43
@ SWS_OP_LSHIFT
Definition ops.h:48
@ SWS_OP_FILTER_V
Definition ops.h:64
@ SWS_OP_SCALE
Definition ops.h:56
@ SWS_OP_FILTER_H
Definition ops.h:63
@ SWS_OP_WRITE
Definition ops.h:41
@ SWS_OP_READ
Definition ops.h:40
@ SWS_OP_CLEAR
Definition ops.h:52
@ SWS_OP_MIN
Definition ops.h:54
@ SWS_OP_UNPACK
Definition ops.h:46
@ SWS_OP_LINEAR
Definition ops.h:57
@ SWS_OP_PACK
Definition ops.h:47
@ SWS_OP_DITHER
Definition ops.h:60
@ SWS_OP_MAX
Definition ops.h:55
@ SWS_OP_CONVERT
Definition ops.h:53
@ SWS_RW_PALETTE
Definition ops.h:109
@ SWS_RW_PLANAR
Note: 1-component reads are either SWS_RW_PLANAR or SWS_RW_PACKED, depending on the underlying interp...
Definition ops.h:107
@ SWS_RW_PACKED
Definition ops.h:108
@ AV_PIX_FMT_VULKAN
Vulkan hardware images.
Definition pixfmt.h:379
#define AV_PIX_FMT_X2BGR10
Definition pixfmt.h:620
void av_refstruct_unref(void *objp)
Decrement the reference count of the underlying object and automatically free the object if there are...
Definition refstruct.c:121
void * av_refstruct_ref(void *obj)
Create a new reference to an object managed via this API, i.e.
Definition refstruct.c:141
static void * av_refstruct_alloc_ext(size_t size, unsigned flags, void *opaque, void(*free_cb)(AVRefStructOpaque opaque, void *obj))
A wrapper around av_refstruct_alloc_ext_c() for the common case of a non-const qualified opaque.
Definition refstruct.h:94
#define FF_ARRAY_ELEMS(a)
static int spi_OpTypeFunction(SPICtx *spi, int return_type_id, const int *args, int nb_args)
Definition spvasm.h:498
static void spi_init(SPICtx *spi, uint8_t *spv_buf, int buf_len)
Definition spvasm.h:86
static int spi_OpVariable(SPICtx *spi, int var_id, int ptr_type_id, SpvStorageClass storage_class, int initializer_id)
Definition spvasm.h:537
static int spi_OpConstantUInt(SPICtx *spi, int type_id, uint32_t val)
Definition spvasm.h:565
static void spi_OpCapability(SPICtx *spi, SpvCapability capability)
Definition spvasm.h:119
static void spi_OpMemoryModel(SPICtx *spi, SpvAddressingModel addressing_model, SpvMemoryModel memory_model)
Definition spvasm.h:125
#define spi_OpCompositeExtract(spi, res_type, src,...)
Definition spvasm.h:319
#define spi_OpExtInst(spi, res_type, instr_id, set_id,...)
Definition spvasm.h:348
static void spi_OpFunctionEnd(SPICtx *spi)
Definition spvasm.h:532
static void spi_OpReturn(SPICtx *spi)
Definition spvasm.h:527
static int spi_OpImageRead(SPICtx *spi, int result_type_id, int img_id, int pos_id, SpvImageOperandsMask image_operands)
Definition spvasm.h:656
static int spi_OpTypeArray(SPICtx *spi, int element_type_id, int id, int length_id)
Definition spvasm.h:470
#define spi_OpMemberDecorate(spi, type, target, deco,...)
Definition spvasm.h:360
#define spi_OpTypeStruct(spi, id,...)
Definition spvasm.h:352
#define spi_OpDecorate(spi, target, deco,...)
Definition spvasm.h:356
static void spi_OpBranchConditional(SPICtx *spi, int cond_id, int true_label, int false_label, uint32_t branch_weights)
Definition spvasm.h:642
static int spi_OpTypeImage(SPICtx *spi, int sampled_type_id, SpvDim dim, int depth, int arrayed, int ms, int sampled, SpvImageFormat image_format)
Definition spvasm.h:453
static int spi_OpConstantFloat(SPICtx *spi, int type_id, float val)
Definition spvasm.h:596
#define spi_OpConstantComposite(spi, res_type, src,...)
Definition spvasm.h:307
static int spi_OpLoad(SPICtx *spi, int result_type_id, int ptr_id, SpvMemoryAccessMask memory_access, int align)
Definition spvasm.h:608
static void spi_OpExecutionMode(SPICtx *spi, int entry_point_id, SpvExecutionMode mode, int *s, int nb_s)
Definition spvasm.h:405
static int spi_OpConstantInt(SPICtx *spi, int type_id, int val)
Definition spvasm.h:584
static int spi_end(SPICtx *spi)
Definition spvasm.h:100
static int spi_OpLabel(SPICtx *spi, int label_id)
Definition spvasm.h:520
#define spi_OpAccessChain(spi, res_type, ptr_id,...)
Definition spvasm.h:311
static int spi_OpUndef(SPICtx *spi, int type_id)
Definition spvasm.h:415
static int spi_OpTypeBool(SPICtx *spi)
Definition spvasm.h:430
static int spi_OpTypePointer(SPICtx *spi, SpvStorageClass storage_class, int type_id)
Definition spvasm.h:488
#define spi_OpCompositeConstruct(spi, res_type, src,...)
Definition spvasm.h:315
static void spi_OpFunction(SPICtx *spi, int fn_id, int result_type_id, SpvFunctionControlMask function_control, int function_type_id)
Definition spvasm.h:509
#define spi_OpCompositeInsert(spi, res_type, src1, src2,...)
Definition spvasm.h:368
#define spi_OpVectorShuffle(spi, res_type, src1, src2,...)
Definition spvasm.h:364
static int spi_reserve(SPICtx *spi, int len)
Definition spvasm.h:108
static int spi_get_id(SPICtx *spi)
Definition spvasm.h:133
static int spi_OpExtInstImport(SPICtx *spi, const char *name)
Definition spvasm.h:397
static void spi_OpSelectionMerge(SPICtx *spi, int merge_block, SpvSelectionControlMask selection_control)
Definition spvasm.h:634
static int spi_OpEntryPoint(SPICtx *spi, SpvExecutionModel execution_model, const char *name, const int *args, int nb_args)
Definition spvasm.h:372
static int spi_OpTypeVoid(SPICtx *spi)
Definition spvasm.h:423
static void spi_OpImageWrite(SPICtx *spi, int img_id, int pos_id, int src_id, SpvImageOperandsMask image_operands)
Definition spvasm.h:669
A reference to a data buffer.
Definition buffer.h:82
uint8_t * data
The data buffer.
Definition buffer.h:90
This structure describes decoded (raw) audio or video data.
Definition frame.h:479
64-bit Rational number (pair of numerator and denominator).
Definition rational64.h:52
int64_t num
Numerator.
Definition rational64.h:53
int64_t den
Denominator.
Definition rational64.h:54
VkCommandBuffer buf
Definition vulkan.h:142
Copyright (C) 2026 Lynne.
Definition ops.h:27
int precompiled
Definition vulkan.h:215
uint32_t lg_size[3]
Definition vulkan.h:219
int off
Definition spvasm.h:55
int id
Definition spvasm.h:59
Main external API structure.
Definition swscale.h:227
unsigned flags
Bitmask of SWS_*.
Definition swscale.h:238
AVRational64 * matrix
Definition ops.h:185
int size_log2
Definition ops.h:187
Represents a computed filter kernel.
Definition filters.h:85
size_t num_weights
Definition filters.h:98
int * weights
The computed look-up table (LUT).
Definition filters.h:97
int filter_size
The number of source texels to convolve over for each row.
Definition filters.h:89
int * offsets
The computed source pixel positions for each row of the filter.
Definition filters.h:105
enum AVPixelFormat format
Definition format.h:81
int interlaced
Definition format.h:79
int field
Definition format.h:80
Represents a view into a single field of frame data.
Definition format.h:242
SwsFormat dst
Definition graph.h:165
Helper struct for representing a list of operations.
Definition ops.h:297
SwsFormat dst
Definition ops.h:302
uint8_t plane_src[4]
Definition ops.h:305
uint8_t plane_dst[4]
Definition ops.h:305
SwsOp * ops
Definition ops.h:298
int num_ops
Definition ops.h:299
SwsFormat src
Definition ops.h:302
Definition ops.h:241
SwsPixelType type
Definition ops.h:243
Represents a single filter pass in the scaling graph.
Definition graph.h:85
void * priv
Definition graph.h:121
const SwsGraph * graph
Definition graph.h:86
FFVkBuffer data_bufs[MAX_DATA_BUFS]
Definition ops.c:89
enum FFVkShaderRepFormat src_rep
Definition ops.c:91
enum FFVkShaderRepFormat dst_rep
Definition ops.c:92
FFVulkanOpsCtx * s
Definition ops.c:86
int interlaced
Definition ops.c:93
int nb_data_bufs
Definition ops.c:90
FFVulkanShader shd
Definition ops.c:88
FFVkExecPool e
Definition ops.c:87
static SwsInternal * sws_internal(const SwsContext *sws)
#define av_free(p)
#define av_mallocz(s)
#define av_log(a,...)
static uint8_t tmp[40]
Definition aes_ctr.c:52
void(* filter)(uint8_t *src, ptrdiff_t stride, int qscale)
Definition h263dsp.c:29
#define src
Definition vp8dsp.c:248
int size
RefStruct is an API for creating reference-counted objects with minimal overhead.
Definition refstruct.h:58
SwsPixelType
Definition uops.h:40
@ SWS_PIXEL_F32
Definition uops.h:45
#define SWS_COMP_TEST(mask, X)
Definition uops.h:123
#define img
static const uint16_t dither[8][8]
Definition vf_gradfun.c:46
static void process(NormalizeContext *s, AVFrame *in, AVFrame *out)
int len
static double c[64]
AVBufferRef * ff_sws_vk_device_ref(SwsContext *sws)
Returns the Vulkan device reference associated with sws, or NULL if Vulkan has not been initialized f...
Definition ops.c:74
static int compile(SwsContext *sws, const SwsOpList *ops, SwsCompiledOp *out)
Definition ops.c:1281
#define MAX_DATA_BUFS
Definition ops.c:83
static int create_filter_buf(FFVulkanOpsCtx *s, VulkanPriv *p, const SwsFilterWeights *wd, FFVkBuffer *buf)
Definition ops.c:172
#define MAX_DITHER_BUFS
Definition ops.c:81
static void ff_sws_vk_uninit(AVRefStructOpaque opaque, void *obj)
Copyright (C) 2026 Lynne.
Definition ops.c:34
static void free_fn(void *priv)
Definition ops.c:161
static void process(const SwsFrame *dst, const SwsFrame *src, int y, int h, const SwsPass *pass)
Definition ops.c:96
static int create_dither_buf(FFVulkanOpsCtx *s, VulkanPriv *p, const SwsDitherOp *dd, FFVkBuffer *buf)
Definition ops.c:207
#define MAX_FILT_BUFS
Definition ops.c:82
int ff_sws_vk_init(SwsContext *sws, AVBufferRef *dev_ref)
Definition ops.c:41
static int create_bufs(FFVulkanOpsCtx *s, VulkanPriv *p, const SwsOpList *ops)
Definition ops.c:242
FFVkShaderRepFormat
Returns the format to use for images in shaders.
Definition vulkan.h:438
@ FF_VK_REP_UINT
Definition vulkan.h:446
@ FF_VK_REP_FLOAT
Definition vulkan.h:442
static int ff_vk_unmap_buffer(FFVulkanContext *s, FFVkBuffer *buf, int flush)
Definition vulkan.h:653
static int ff_vk_map_buffer(FFVulkanContext *s, FFVkBuffer *buf, uint8_t **mem, int invalidate)
Definition vulkan.h:646