FFmpeg
Loading...
Searching...
No Matches
uops_tmpl.c
Go to the documentation of this file.
1/**
2 * Copyright (C) 2026 Niklas Haas
3 *
4 * This file is part of FFmpeg.
5 *
6 * FFmpeg is free software; you can redistribute it and/or
7 * modify it under the terms of the GNU Lesser General Public
8 * License as published by the Free Software Foundation; either
9 * version 2.1 of the License, or (at your option) any later version.
10 *
11 * FFmpeg is distributed in the hope that it will be useful,
12 * but WITHOUT ANY WARRANTY; without even the implied warranty of
13 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
14 * Lesser General Public License for more details.
15 *
16 * You should have received a copy of the GNU Lesser General Public
17 * License along with FFmpeg; if not, write to the Free Software
18 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
19 */
20
21#include <libavutil/bswap.h>
22
23#include "uops_tmpl.h"
24
25#ifndef BIT_DEPTH
26# define BIT_DEPTH 8
27#endif
28
29#if IS_FLOAT && BIT_DEPTH == 32
30# define PIXEL_TYPE SWS_PIXEL_F32
31# define pixel_t float
32# define inter_t float
33# define uinter_t float
34# define vec3_t v3f32_t
35# define PX F32
36# define px f32
37#elif BIT_DEPTH == 32
38# define PIXEL_MAX 0xFFFFFFFFu
39# define PIXEL_SWAP av_bswap32
40# define pixel_t uint32_t
41# define inter_t int64_t
42# define uinter_t uint64_t
43# define PX U32
44# define px u32
45#elif BIT_DEPTH == 16
46# define PIXEL_MAX 0xFFFFu
47# define PIXEL_SWAP av_bswap16
48# define pixel_t uint16_t
49# define inter_t int64_t
50# define uinter_t uint64_t
51# define PX U16
52# define px u16
53#elif BIT_DEPTH == 8
54# define PIXEL_MAX 0xFFu
55# define pixel_t uint8_t
56# define inter_t int32_t
57# define uinter_t uint32_t
58# define PX U8
59# define px u8
60#else
61# error Invalid BIT_DEPTH
62#endif
63
64/*********************************
65 * Generic read/write operations *
66 *********************************/
67
68DECL_READ(read_planar, const SwsCompMask mask)
69{
71 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
72 if (X) x[i] = in0[i];
73 if (Y) y[i] = in1[i];
74 if (Z) z[i] = in2[i];
75 if (W) w[i] = in3[i];
76 }
77
78 if (X) iter->in[0] += SIZEOF_BLOCK;
79 if (Y) iter->in[1] += SIZEOF_BLOCK;
80 if (Z) iter->in[2] += SIZEOF_BLOCK;
81 if (W) iter->in[3] += SIZEOF_BLOCK;
82
83 CONTINUE(x, y, z, w);
84}
85
86DECL_READ(read_packed, const SwsCompMask mask)
87{
88 const int elems = W ? 4 : Z ? 3 : Y ? 2 : 1;
89
91 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
92 if (X) x[i] = in0[elems * i + 0];
93 if (Y) y[i] = in0[elems * i + 1];
94 if (Z) z[i] = in0[elems * i + 2];
95 if (W) w[i] = in0[elems * i + 3];
96 }
97
98 iter->in[0] += SIZEOF_BLOCK * elems;
99 CONTINUE(x, y, z, w);
100}
101
102DECL_WRITE(write_planar, const SwsCompMask mask)
103{
105 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
106 if (X) out0[i] = x[i];
107 if (Y) out1[i] = y[i];
108 if (Z) out2[i] = z[i];
109 if (W) out3[i] = w[i];
110 }
111
112 if (X) iter->out[0] += SIZEOF_BLOCK;
113 if (Y) iter->out[1] += SIZEOF_BLOCK;
114 if (Z) iter->out[2] += SIZEOF_BLOCK;
115 if (W) iter->out[3] += SIZEOF_BLOCK;
116}
117
118DECL_WRITE(write_packed, const SwsCompMask mask)
119{
120 const int elems = W ? 4 : Z ? 3 : Y ? 2 : 1;
121
123 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
124 if (X) out0[elems * i + 0] = x[i];
125 if (Y) out0[elems * i + 1] = y[i];
126 if (Z) out0[elems * i + 2] = z[i];
127 if (W) out0[elems * i + 3] = w[i];
128 }
129
130 iter->out[0] += SIZEOF_BLOCK * elems;
131}
132
133#if BIT_DEPTH == 8
134
136{
138
140 for (int i = 0; i < SWS_BLOCK_SIZE; i += 8) {
141 const pixel_t val = ((const pixel_t *) in0)[i >> 3];
142 x[i + 0] = (val >> 7) & 1;
143 x[i + 1] = (val >> 6) & 1;
144 x[i + 2] = (val >> 5) & 1;
145 x[i + 3] = (val >> 4) & 1;
146 x[i + 4] = (val >> 3) & 1;
147 x[i + 5] = (val >> 2) & 1;
148 x[i + 6] = (val >> 1) & 1;
149 x[i + 7] = (val >> 0) & 1;
150 }
151
152 iter->in[0] += SIZEOF_BLOCK >> 3;
153 CONTINUE(x, y, z, w);
154}
155
156DECL_READ(read_nibble, const SwsCompMask mask)
157{
159
161 for (int i = 0; i < SWS_BLOCK_SIZE; i += 2) {
162 const pixel_t val = in0[i >> 1];
163 x[i + 0] = val >> 4; /* high nibble */
164 x[i + 1] = val & 0xF; /* low nibble */
165 }
166
167 iter->in[0] += SIZEOF_BLOCK >> 1;
168 CONTINUE(x, y, z, w);
169}
170
171DECL_READ(read_palette, const SwsCompMask mask)
172{
174
176 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
177 const pixel_t index = in0[i];
178 const pixel_t *value = &in1[index * 4];
179 x[i] = value[0];
180 y[i] = value[1];
181 z[i] = value[2];
182 w[i] = value[3];
183 }
184
185 iter->in[0] += SIZEOF_BLOCK;
186 CONTINUE(x, y, z, w);
187}
188
189DECL_WRITE(write_bit, const SwsCompMask mask)
190{
192
194 for (int i = 0; i < SWS_BLOCK_SIZE; i += 8) {
195 out0[i >> 3] = x[i + 0] << 7 |
196 x[i + 1] << 6 |
197 x[i + 2] << 5 |
198 x[i + 3] << 4 |
199 x[i + 4] << 3 |
200 x[i + 5] << 2 |
201 x[i + 6] << 1 |
202 x[i + 7];
203 }
204
205 iter->out[0] += SIZEOF_BLOCK >> 3;
206}
207
208DECL_WRITE(write_nibble, const SwsCompMask mask)
209{
211
213 for (int i = 0; i < SWS_BLOCK_SIZE; i += 2)
214 out0[i >> 1] = x[i] << 4 | x[i + 1];
215
216 iter->out[0] += SIZEOF_BLOCK >> 1;
217}
218
219#endif /* BIT_DEPTH == 8 */
220
221SWS_FOR(PX, READ_PLANAR, DECL_IMPL_READ, read_planar)
222SWS_FOR(PX, READ_PACKED, DECL_IMPL_READ, read_packed)
223SWS_FOR(PX, READ_NIBBLE, DECL_IMPL_READ, read_nibble)
224SWS_FOR(PX, READ_BIT, DECL_IMPL_READ, read_bit)
225SWS_FOR(PX, READ_PALETTE, DECL_IMPL_READ, read_palette)
226SWS_FOR(PX, WRITE_PLANAR, DECL_IMPL_WRITE, write_planar)
227SWS_FOR(PX, WRITE_PACKED, DECL_IMPL_WRITE, write_packed)
228SWS_FOR(PX, WRITE_NIBBLE, DECL_IMPL_WRITE, write_nibble)
229SWS_FOR(PX, WRITE_BIT, DECL_IMPL_WRITE, write_bit)
230
231SWS_FOR_STRUCT(PX, READ_PLANAR, DECL_ENTRY)
232SWS_FOR_STRUCT(PX, READ_PACKED, DECL_ENTRY)
233SWS_FOR_STRUCT(PX, READ_NIBBLE, DECL_ENTRY)
234SWS_FOR_STRUCT(PX, READ_BIT, DECL_ENTRY)
235SWS_FOR_STRUCT(PX, READ_PALETTE, DECL_ENTRY)
236SWS_FOR_STRUCT(PX, WRITE_PLANAR, DECL_ENTRY)
237SWS_FOR_STRUCT(PX, WRITE_PACKED, DECL_ENTRY)
238SWS_FOR_STRUCT(PX, WRITE_NIBBLE, DECL_ENTRY)
239SWS_FOR_STRUCT(PX, WRITE_BIT, DECL_ENTRY)
240
241/*****************************
242 * Scaling / filtering reads *
243 *****************************/
244
246{
247 if (params->uop->par.filter.type != SWS_PIXEL_F32)
248 return AVERROR(ENOTSUP);
249
250 const SwsFilterWeights *filter = params->uop->data.kernel;
251 static_assert(sizeof(out->priv.ptr) <= sizeof(int32_t[2]),
252 ">8 byte pointers not supported");
253
254 /* Pre-convert weights to float */
255 float *weights = av_calloc(filter->num_weights, sizeof(float));
256 if (!weights)
257 return AVERROR(ENOMEM);
258
259 for (int i = 0; i < filter->num_weights; i++)
260 weights[i] = (float) filter->weights[i] / SWS_FILTER_SCALE;
261
262 out->priv.ptr = weights;
263 out->priv.i32[2] = filter->filter_size;
264 out->free = ff_op_priv_free;
265 return 0;
266}
267
268/* Fully general vertical planar filter case */
269DECL_READ(read_planar_fv, const SwsCompMask mask, const SwsPixelType type)
270{
272 const SwsOpExec *exec = iter->exec;
273 const float *restrict weights = impl->priv.ptr;
274 const int filter_size = impl->priv.i32[2];
275 weights += filter_size * iter->y;
276
277 block_t xs, ys, zs, ws;
278 if (X) memset(&xs.f32, 0, sizeof(xs.f32));
279 if (Y) memset(&ys.f32, 0, sizeof(ys.f32));
280 if (Z) memset(&zs.f32, 0, sizeof(zs.f32));
281 if (W) memset(&ws.f32, 0, sizeof(ws.f32));
282
283 for (int j = 0; j < filter_size; j++) {
284 const float weight = weights[j];
285
287 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
288 if (X) xs.f32[i] += weight * in0[i];
289 if (Y) ys.f32[i] += weight * in1[i];
290 if (Z) zs.f32[i] += weight * in2[i];
291 if (W) ws.f32[i] += weight * in3[i];
292 }
293
294 if (X) in0 = bump_ptr(in0, exec->in_stride[0]);
295 if (Y) in1 = bump_ptr(in1, exec->in_stride[1]);
296 if (Z) in2 = bump_ptr(in2, exec->in_stride[2]);
297 if (W) in3 = bump_ptr(in3, exec->in_stride[3]);
298 }
299
300 if (X) iter->in[0] += SIZEOF_BLOCK;
301 if (Y) iter->in[1] += SIZEOF_BLOCK;
302 if (Z) iter->in[2] += SIZEOF_BLOCK;
303 if (W) iter->in[3] += SIZEOF_BLOCK;
304
305 CONTINUE(&xs, &ys, &zs, &ws);
306}
307
309{
310 if (params->uop->par.filter.type != SWS_PIXEL_F32)
311 return AVERROR(ENOTSUP);
312
313 SwsFilterWeights *filter = params->uop->data.kernel;
314 out->priv.i32[2] = filter->filter_size;
315
316 /* The horizontal filter uop reads weights for a full SWS_BLOCK_SIZE
317 * outputs per block (see read_planar_fh), so the weights array must be
318 * padded up to the block-aligned output count. Pad the tail with zero
319 * weights, which contribute nothing to the accumulated sums. */
320 const size_t padded_w = FFALIGN(filter->dst_size, SWS_BLOCK_SIZE);
321 if (padded_w == filter->dst_size) {
322 out->priv.ptr = av_refstruct_ref(filter->weights);
323 out->free = ff_op_priv_unref;
324 return 0;
325 }
326
327 int *weights = av_calloc(padded_w * filter->filter_size, sizeof(*weights));
328 if (!weights)
329 return AVERROR(ENOMEM);
330 memcpy(weights, filter->weights, filter->num_weights * sizeof(*weights));
331 out->priv.ptr = weights;
332 out->free = ff_op_priv_free;
333 return 0;
334}
335
336/* Fully general horizontal planar filter case */
337DECL_READ(read_planar_fh, const SwsCompMask mask, const SwsPixelType type)
338{
340 const SwsOpExec *exec = iter->exec;
341 const int *restrict weights = impl->priv.ptr;
342 const int filter_size = impl->priv.i32[2];
343 const float scale = 1.0f / SWS_FILTER_SCALE;
344 const int xpos = iter->x;
345 weights += filter_size * iter->x;
346
347 block_t xs, ys, zs, ws;
348 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
349 const int offset = exec->in_offset_x[xpos + i];
350 pixel_t *start0 = bump_ptr(in0, offset);
351 pixel_t *start1 = bump_ptr(in1, offset);
352 pixel_t *start2 = bump_ptr(in2, offset);
353 pixel_t *start3 = bump_ptr(in3, offset);
354
355 inter_t sx = 0, sy = 0, sz = 0, sw = 0;
356 for (int j = 0; j < filter_size; j++) {
357 const int weight = weights[j];
358 if (X) sx += weight * start0[j];
359 if (Y) sy += weight * start1[j];
360 if (Z) sz += weight * start2[j];
361 if (W) sw += weight * start3[j];
362 }
363
364 if (X) xs.f32[i] = (float) sx * scale;
365 if (Y) ys.f32[i] = (float) sy * scale;
366 if (Z) zs.f32[i] = (float) sz * scale;
367 if (W) ws.f32[i] = (float) sw * scale;
368
369 weights += filter_size;
370 }
371
372 CONTINUE(&xs, &ys, &zs, &ws);
373}
374
375SWS_FOR(PX, READ_PLANAR_FV, DECL_IMPL_READ, read_planar_fv)
376SWS_FOR(PX, READ_PLANAR_FH, DECL_IMPL_READ, read_planar_fh)
377SWS_FOR_STRUCT(PX, READ_PLANAR_FV, DECL_ENTRY, .setup = fn(setup_filter_v) )
378SWS_FOR_STRUCT(PX, READ_PLANAR_FH, DECL_ENTRY, .setup = fn(setup_filter_h) )
379
380/***************************
381 * Permutation and copying *
382 ***************************/
383
384DECL_FUNC(permute, const SwsCompMask mask, int num_moves,
385 int8_t d0, int8_t d1, int8_t d2, int8_t d3, int8_t d4, int8_t d5,
386 int8_t s0, int8_t s1, int8_t s2, int8_t s3, int8_t s4, int8_t s5)
387{
388 const int8_t dst[SWS_UOP_MOVE_MAX] = { d0, d1, d2, d3, d4, d5 };
389 const int8_t src[SWS_UOP_MOVE_MAX] = { s0, s1, s2, s3, s4, s5 };
390
391 pixel_t *ptr[5] = { NULL, x, y, z, w };
392 for (int n = 0; n < num_moves; n++)
393 ptr[dst[n] + 1] = ptr[src[n] + 1];
394
395 /* The unneeded registers may still alias the used ones, so point them
396 * back at the stack to avoid collisions */
397 block_t xx, yy, zz, ww;
398 CONTINUE(X ? ptr[1] : xx.px,
399 Y ? ptr[2] : yy.px,
400 Z ? ptr[3] : zz.px,
401 W ? ptr[4] : ww.px);
402}
403
404DECL_FUNC(copy, const SwsCompMask mask, int num_moves,
405 int8_t d0, int8_t d1, int8_t d2, int8_t d3, int8_t d4, int8_t d5,
406 int8_t s0, int8_t s1, int8_t s2, int8_t s3, int8_t s4, int8_t s5)
407{
408 const size_t block_size = SWS_BLOCK_SIZE * sizeof(pixel_t);
409 const int8_t dst[SWS_UOP_MOVE_MAX] = { d0, d1, d2, d3, d4, d5 };
410 const int8_t src[SWS_UOP_MOVE_MAX] = { s0, s1, s2, s3, s4, s5 };
411
412 block_t data[5];
413 memcpy(&data[1].px, x, block_size);
414 memcpy(&data[2].px, y, block_size);
415 memcpy(&data[3].px, z, block_size);
416 memcpy(&data[4].px, w, block_size);
417
418 for (int n = 0; n < num_moves; n++)
419 data[dst[n] + 1] = data[src[n] + 1];
420
421 memcpy(x, &data[1].px, block_size);
422 memcpy(y, &data[2].px, block_size);
423 memcpy(z, &data[3].px, block_size);
424 memcpy(w, &data[4].px, block_size);
425
426 CONTINUE(x, y, z, w);
427}
428
429SWS_FOR(PX, PERMUTE, DECL_IMPL, permute)
431SWS_FOR_STRUCT(PX, PERMUTE, DECL_ENTRY)
433
434/*********************
435 * Format conversion *
436 *********************/
437
438#define DECL_CAST(DST, dst) \
439 DECL_FUNC(to_##dst, const SwsCompMask mask) \
440 { \
441 block_t xx, yy, zz, ww; \
442 \
443 SWS_LOOP \
444 for (int i = 0; i < SWS_BLOCK_SIZE; i++) { \
445 if (X) xx.dst[i] = x[i]; \
446 if (Y) yy.dst[i] = y[i]; \
447 if (Z) zz.dst[i] = z[i]; \
448 if (W) ww.dst[i] = w[i]; \
449 } \
450 \
451 CONTINUE(&xx, &yy, &zz, &ww); \
452 } \
453 \
454 SWS_FOR(PX, TO_##DST, DECL_IMPL, to_##dst) \
455 SWS_FOR_STRUCT(PX, TO_##DST, DECL_ENTRY)
456
457DECL_CAST(U8, u8)
458DECL_CAST(U16, u16)
459DECL_CAST(U32, u32)
460DECL_CAST(F32, f32)
461
462/********************
463 * Bit manipulation *
464 ********************/
465
466#if !IS_FLOAT
467DECL_FUNC(lshift, const SwsCompMask mask, const uint8_t amount)
468{
470 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
471 if (X) x[i] <<= amount;
472 if (Y) y[i] <<= amount;
473 if (Z) z[i] <<= amount;
474 if (W) w[i] <<= amount;
475 }
476
477 CONTINUE(x, y, z, w);
478}
479
480DECL_FUNC(rshift, const SwsCompMask mask, const uint8_t amount)
481{
483 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
484 if (X) x[i] >>= amount;
485 if (Y) y[i] >>= amount;
486 if (Z) z[i] >>= amount;
487 if (W) w[i] >>= amount;
488 }
489
490 CONTINUE(x, y, z, w);
491}
492#endif
493
494SWS_FOR(PX, LSHIFT, DECL_IMPL, lshift)
495SWS_FOR(PX, RSHIFT, DECL_IMPL, rshift)
496
497SWS_FOR_STRUCT(PX, LSHIFT, DECL_ENTRY)
499
500#ifdef PIXEL_SWAP
501DECL_FUNC(swap_bytes, const SwsCompMask mask)
502{
504 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
505 if (X) x[i] = PIXEL_SWAP(x[i]);
506 if (Y) y[i] = PIXEL_SWAP(y[i]);
507 if (Z) z[i] = PIXEL_SWAP(z[i]);
508 if (W) w[i] = PIXEL_SWAP(w[i]);
509 }
510
511 CONTINUE(x, y, z, w);
512}
513#endif /* PIXEL_SWAP */
514
515SWS_FOR(PX, SWAP_BYTES, DECL_IMPL, swap_bytes)
516SWS_FOR_STRUCT(PX, SWAP_BYTES, DECL_ENTRY)
517
518#ifdef PIXEL_MAX
519DECL_FUNC(expand_bit, const SwsCompMask mask)
520{
522 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
523 if (X) x[i] = x[i] ? PIXEL_MAX : 0;
524 if (Y) y[i] = y[i] ? PIXEL_MAX : 0;
525 if (Z) z[i] = z[i] ? PIXEL_MAX : 0;
526 if (W) w[i] = w[i] ? PIXEL_MAX : 0;
527 }
528
529 CONTINUE(x, y, z, w);
530}
531#endif
532
533#if BIT_DEPTH == 8
534DECL_FUNC(expand_pair, const SwsCompMask mask)
535{
536 block_t x16, y16, z16, w16;
537
539 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
540 if (X) x16.u16[i] = x[i] << 8 | x[i];
541 if (Y) y16.u16[i] = y[i] << 8 | y[i];
542 if (Z) z16.u16[i] = z[i] << 8 | z[i];
543 if (W) w16.u16[i] = w[i] << 8 | w[i];
544 }
545
546 CONTINUE(&x16, &y16, &z16, &w16);
547}
548
549DECL_FUNC(expand_quad, const SwsCompMask mask)
550{
551 block_t x32, y32, z32, w32;
552
554 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
555 if (X) x32.u32[i] = (uint32_t) x[i] << 24 | x[i] << 16 | x[i] << 8 | x[i];
556 if (Y) y32.u32[i] = (uint32_t) y[i] << 24 | y[i] << 16 | y[i] << 8 | y[i];
557 if (Z) z32.u32[i] = (uint32_t) z[i] << 24 | z[i] << 16 | z[i] << 8 | z[i];
558 if (W) w32.u32[i] = (uint32_t) w[i] << 24 | w[i] << 16 | w[i] << 8 | w[i];
559 }
560
561 CONTINUE(&x32, &y32, &z32, &w32);
562}
563#endif /* BIT_DEPTH == 8 */
564
565SWS_FOR(PX, EXPAND_BIT, DECL_IMPL, expand_bit)
566SWS_FOR(PX, EXPAND_PAIR, DECL_IMPL, expand_pair)
567SWS_FOR(PX, EXPAND_QUAD, DECL_IMPL, expand_quad)
568SWS_FOR_STRUCT(PX, EXPAND_BIT, DECL_ENTRY)
569SWS_FOR_STRUCT(PX, EXPAND_PAIR, DECL_ENTRY)
570SWS_FOR_STRUCT(PX, EXPAND_QUAD, DECL_ENTRY)
571
572/*************************
573 * Packing and unpacking *
574 ************************/
575
576#if !IS_FLOAT
578 const uint8_t bx, const uint8_t by,
579 const uint8_t bz, const uint8_t bw)
580{
581 const uint8_t sx = bw + bz + by;
582 const uint8_t sy = bw + bz;
583 const uint8_t sz = bw;
584 const uint8_t sw = 0;
585
586 const pixel_t mx = (1 << bx) - 1;
587 const pixel_t my = (1 << by) - 1;
588 const pixel_t mz = (1 << bz) - 1;
589 const pixel_t mw = (1 << bw) - 1;
590
592 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
593 const pixel_t val = x[i];
594 if (X) x[i] = (val >> sx) & mx;
595 if (Y) y[i] = (val >> sy) & my;
596 if (Z) z[i] = (val >> sz) & mz;
597 if (W) w[i] = (val >> sw) & mw;
598 }
599
600 CONTINUE(x, y, z, w);
601}
602
604 const uint8_t bx, const uint8_t by,
605 const uint8_t bz, const uint8_t bw)
606{
607 const uint8_t sx = bw + bz + by;
608 const uint8_t sy = bw + bz;
609 const uint8_t sz = bw;
610 const uint8_t sw = 0;
611
613 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
614 pixel_t val = 0;
615 if (X) val |= x[i] << sx;
616 if (Y) val |= y[i] << sy;
617 if (Z) val |= z[i] << sz;
618 if (W) val |= w[i] << sw;
619 x[i] = val;
620 }
621
622 CONTINUE(x, y, z, w);
623}
624#endif /* !IS_FLOAT */
625
626SWS_FOR(PX, UNPACK, DECL_IMPL, unpack)
627SWS_FOR(PX, PACK, DECL_IMPL, pack)
628SWS_FOR_STRUCT(PX, UNPACK, DECL_ENTRY)
629SWS_FOR_STRUCT(PX, PACK, DECL_ENTRY)
630
631/***********************
632 * Pixel data clearing *
633 ***********************/
634
635#ifdef PIXEL_MAX
636DECL_FUNC(clear, const SwsCompMask mask, const SwsCompMask one,
637 const SwsCompMask zero)
638{
639 #define ONE(N) SWS_COMP_TEST(one, N)
640 #define ZERO(N) SWS_COMP_TEST(zero, N)
641 const pixel_t cx = ONE(0) ? PIXEL_MAX : ZERO(0) ? 0 : impl->priv.px[0];
642 const pixel_t cy = ONE(1) ? PIXEL_MAX : ZERO(1) ? 0 : impl->priv.px[1];
643 const pixel_t cz = ONE(2) ? PIXEL_MAX : ZERO(2) ? 0 : impl->priv.px[2];
644 const pixel_t cw = ONE(3) ? PIXEL_MAX : ZERO(3) ? 0 : impl->priv.px[3];
645
647 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
648 if (X) x[i] = cx;
649 if (Y) y[i] = cy;
650 if (Z) z[i] = cz;
651 if (W) w[i] = cw;
652 }
653
654 CONTINUE(x, y, z, w);
655}
656#endif
657
658SWS_FOR(PX, CLEAR, DECL_IMPL, clear)
660
661/*************************
662 * Arithmetic operations *
663 *************************/
664
666{
667 const pixel_t scale = impl->priv.px[0];
668
670 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
671 if (X) x[i] *= scale;
672 if (Y) y[i] *= scale;
673 if (Z) z[i] *= scale;
674 if (W) w[i] *= scale;
675 }
676
677 CONTINUE(x, y, z, w);
678}
679
681{
683 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
684 if (X) x[i] += impl->priv.px[0];
685 if (Y) y[i] += impl->priv.px[1];
686 if (Z) z[i] += impl->priv.px[2];
687 if (W) w[i] += impl->priv.px[3];
688 }
689
690 CONTINUE(x, y, z, w);
691}
692
694{
696 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
697 if (X) x[i] = FFMIN(x[i], impl->priv.px[0]);
698 if (Y) y[i] = FFMIN(y[i], impl->priv.px[1]);
699 if (Z) z[i] = FFMIN(z[i], impl->priv.px[2]);
700 if (W) w[i] = FFMIN(w[i], impl->priv.px[3]);
701 }
702
703 CONTINUE(x, y, z, w);
704}
705
707{
709 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
710 if (X) x[i] = FFMAX(x[i], impl->priv.px[0]);
711 if (Y) y[i] = FFMAX(y[i], impl->priv.px[1]);
712 if (Z) z[i] = FFMAX(z[i], impl->priv.px[2]);
713 if (W) w[i] = FFMAX(w[i], impl->priv.px[3]);
714 }
715
716 CONTINUE(x, y, z, w);
717}
718
720SWS_FOR(PX, ADD, DECL_IMPL, add)
727
728/*************
729 * Dithering *
730 *************/
731
733{
734 const SwsUOp *uop = params->uop;
735 const SwsDitherUOp *dither = &uop->par.dither;
736 const int size = 1 << dither->size_log2;
737 if (size >= SWS_BLOCK_SIZE) {
738 /* No extra padding needed */
739 out->priv.ptr = av_refstruct_ref(uop->data.ptr);
740 out->free = ff_op_priv_unref;
741 return 0;
742 }
743
744 const int stride = FFMAX(size, SWS_BLOCK_SIZE);
746 pixel_t *matrix = av_malloc(sizeof(pixel_t) * height * stride);
747 if (!matrix)
748 return AVERROR(ENOMEM);
749 out->priv.ptr = matrix;
750 out->free = ff_op_priv_free;
751
752 /* Pad to multiple of block size. We don't need extra padding for the
753 * height because ff_sws_dither_height() already includes any padding
754 * necessary for the y_offset */
755 for (int y = 0; y < height; y++) {
756 pixel_t *row = &matrix[y * stride];
757 for (int x = 0; x < size; x++)
758 row[x] = uop->data.ptr[y * size + x].px;
759 for (int x = size; x < stride; x++)
760 row[x] = row[x % size];
761 }
762
763 return 0;
764}
765
767 const uint8_t off0, const uint8_t off1,
768 const uint8_t off2, const uint8_t off3,
769 const uint8_t size_log2)
770{
771 const int size = 1 << size_log2;
772 const int stride = FFMAX(size, SWS_BLOCK_SIZE);
773
774 const pixel_t *matrix = impl->priv.ptr;
775 matrix += (iter->y & (size - 1)) * stride;
776 matrix += (iter->x & (size - 1)) & ~(SWS_BLOCK_SIZE - 1);
777
778 const pixel_t *const row0 = &matrix[off0 * stride];
779 const pixel_t *const row1 = &matrix[off1 * stride];
780 const pixel_t *const row2 = &matrix[off2 * stride];
781 const pixel_t *const row3 = &matrix[off3 * stride];
782
784 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
785 if (X) x[i] += row0[i];
786 if (Y) y[i] += row1[i];
787 if (Z) z[i] += row2[i];
788 if (W) w[i] += row3[i];
789 }
790
791 CONTINUE(x, y, z, w);
792}
793
794SWS_FOR(PX, DITHER, DECL_IMPL, dither)
795SWS_FOR_STRUCT(PX, DITHER, DECL_ENTRY, .setup = fn(setup_dither) )
796
797/*********************
798 * Linear operations *
799 *********************/
800
801typedef struct {
802 /* Stored in split form for convenience */
803 pixel_t m[4][4];
804 pixel_t k[4];
805} fn(LinCoeffs);
806
808{
809 const SwsUOp *uop = params->uop;
810 fn(LinCoeffs) c;
811
812 for (int i = 0; i < 4; i++) {
813 for (int j = 0; j < 4; j++)
814 c.m[i][j] = uop->data.mat4x5[i][j].px;
815 c.k[i] = uop->data.mat4x5[i][4].px;
816 }
817
818 out->priv.ptr = av_memdup(&c, sizeof(c));
819 out->free = ff_op_priv_free;
820 return out->priv.ptr ? 0 : AVERROR(ENOMEM);
821}
822
823/**
824 * Fully general case for a 5x5 linear affine transformation. Should never be
825 * called without constant `mask`. This function will compile down to the
826 * appropriately optimized version for the required subset of operations when
827 * called with a constant mask.
828 */
829DECL_FUNC(linear, const SwsCompMask mask, const uint32_t one, const uint32_t zero)
830{
831 const fn(LinCoeffs) c = *(const fn(LinCoeffs) *) impl->priv.ptr;
832
834 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
835 const pixel_t xx = x[i];
836 const pixel_t yy = y[i];
837 const pixel_t zz = z[i];
838 const pixel_t ww = w[i];
839
840#define LIN_VAL(I, J, val) \
841 ((one & SWS_MASK(I, J)) ? (val) : (uinter_t) c.m[I][J] * (val))
842
843#define LIN_ROW(I, var) do { \
844 pixel_t tmp = (zero & SWS_MASK(I, 4)) ? 0 : c.k[I]; \
845 if (!(zero & SWS_MASK(I, 0))) tmp += LIN_VAL(I, 0, xx); \
846 if (!(zero & SWS_MASK(I, 1))) tmp += LIN_VAL(I, 1, yy); \
847 if (!(zero & SWS_MASK(I, 2))) tmp += LIN_VAL(I, 2, zz); \
848 if (!(zero & SWS_MASK(I, 3))) tmp += LIN_VAL(I, 3, ww); \
849 var[i] = tmp; \
850} while (0)
851
852 if (X) LIN_ROW(0, x);
853 if (Y) LIN_ROW(1, y);
854 if (Z) LIN_ROW(2, z);
855 if (W) LIN_ROW(3, w);
856 }
857
858 CONTINUE(x, y, z, w);
859}
860
863
864/******************
865 * Look-up tables *
866 ******************/
867
868DECL_SETUP(setup_lut3d, params, out)
869{
870 const SwsLut3D *lut = params->uop->data.lut3d;
871 out->priv.ptr = (void *) av_refstruct_ref_c(lut);
872 out->free = ff_op_priv_unref;
873 return 0;
874}
875
876#if IS_FLOAT
877av_always_inline static vec3_t fn(vec3)(v3u16_t v)
878{
879 return (vec3_t) { v.x, v.y, v.z };
880}
881
882#define lerp(a, b, w) ((a) + (w) * ((pixel_t) (b) - (a)))
883
884av_always_inline static
885vec3_t fn(lerp3)(vec3_t a, vec3_t b, pixel_t w)
886{
887 return (vec3_t) {
888 lerp(a.x, b.x, w),
889 lerp(a.y, b.y, w),
890 lerp(a.z, b.z, w),
891 };
892}
893
894av_always_inline static
895vec3_t fn(lut3d_static)(const SwsLut3D *restrict lut3d, vec3_t rgb)
896{
897 const int r_base = (int) rgb.x;
898 const int g_base = (int) rgb.y;
899 const int b_base = (int) rgb.z;
900
901 int off0 = (r_base < INPUT_LUT_SIZE - 1);
902 int off1 = (g_base < INPUT_LUT_SIZE - 1) * INPUT_LUT_SIZE;
903 int off2 = (b_base < INPUT_LUT_SIZE - 1) * INPUT_LUT_SIZE * INPUT_LUT_SIZE;
904 pixel_t f0 = rgb.x - r_base;
905 pixel_t f1 = rgb.y - g_base;
906 pixel_t f2 = rgb.z - b_base;
907
908 /* Sort offsets descending by relative weight */
909 if (f0 < f1) {
910 FFSWAP(pixel_t, f0, f1);
911 FFSWAP(int, off0, off1);
912 }
913 if (f0 < f2) {
914 FFSWAP(pixel_t, f0, f2);
915 FFSWAP(int, off0, off2);
916 }
917 if (f1 < f2) {
918 FFSWAP(pixel_t, f1, f2);
919 FFSWAP(int, off1, off2);
920 }
921
922 /* Tetrahedral interpolation */
923 const pixel_t w0 = 1 - f0;
924 const pixel_t w1 = f0 - f1;
925 const pixel_t w2 = f1 - f2;
926 const pixel_t w3 = f2;
927
928 const v3u16_t *restrict base = &lut3d->input[b_base][g_base][r_base];
929 const vec3_t v0 = fn(vec3)(base[0]);
930 const vec3_t v1 = fn(vec3)(base[off0]);
931 const vec3_t v2 = fn(vec3)(base[off0 + off1]);
932 const vec3_t v3 = fn(vec3)(base[off0 + off1 + off2]);
933
934 return (vec3_t) {
935 w0 * v0.x + w1 * v1.x + w2 * v2.x + w3 * v3.x,
936 w0 * v0.y + w1 * v1.y + w2 * v2.y + w3 * v3.y,
937 w0 * v0.z + w1 * v1.z + w2 * v2.z + w3 * v3.z,
938 };
939}
940
941av_always_inline static
942vec3_t fn(lut3d_dynamic)(const SwsLut3D *restrict lut3d, vec3_t rgb)
943{
944 rgb.x *= (TONE_LUT_SIZE - 1) / (pixel_t) UINT16_MAX;
945
946 /* Linear interpolation */
947 const int Ix = (int) rgb.x;
948 const pixel_t If = rgb.x - Ix;
949
950 const v2u16_t a = lut3d->tone_map[Ix];
951 const v2u16_t b = lut3d->tone_map[Ix + 1];
952
953 const pixel_t k = lerp(a.y, b.y, If);
954 const pixel_t bias = (1 << 15) - k;
955 const pixel_t scale = k / (pixel_t) (1 << 15);
956
957 rgb.x = lerp(a.x, b.x, If);
958 rgb.y = bias + scale * rgb.y;
959 rgb.z = bias + scale * rgb.z;
960
961 /* Re-scale to output LUT size */
962 rgb.x *= (OUTPUT_LUT_SIZE_I - 1) / (pixel_t) UINT16_MAX;
963 rgb.y *= (OUTPUT_LUT_SIZE_PT - 1) / (pixel_t) UINT16_MAX;
964 rgb.z *= (OUTPUT_LUT_SIZE_PT - 1) / (pixel_t) UINT16_MAX;
965
966 /* Trilinear interpolation */
967 const int lo0 = (int) rgb.x;
968 const int lo1 = (int) rgb.y;
969 const int lo2 = (int) rgb.z;
970
971 const int hi0 = FFMIN(lo0 + 1, OUTPUT_LUT_SIZE_I - 1);
972 const int hi1 = FFMIN(lo1 + 1, OUTPUT_LUT_SIZE_PT - 1);
973 const int hi2 = FFMIN(lo2 + 1, OUTPUT_LUT_SIZE_PT - 1);
974
975 const pixel_t w0 = rgb.x - lo0;
976 const vec3_t c000 = fn(vec3)(lut3d->output[lo2][lo1][lo0]);
977 const vec3_t c001 = fn(vec3)(lut3d->output[lo2][lo1][hi0]);
978 const vec3_t c00 = fn(lerp3)(c000, c001, w0);
979 const vec3_t c010 = fn(vec3)(lut3d->output[lo2][hi1][lo0]);
980 const vec3_t c011 = fn(vec3)(lut3d->output[lo2][hi1][hi0]);
981 const vec3_t c01 = fn(lerp3)(c010, c011, w0);
982 const vec3_t c100 = fn(vec3)(lut3d->output[hi2][lo1][lo0]);
983 const vec3_t c101 = fn(vec3)(lut3d->output[hi2][lo1][hi0]);
984 const vec3_t c10 = fn(lerp3)(c100, c101, w0);
985 const vec3_t c110 = fn(vec3)(lut3d->output[hi2][hi1][lo0]);
986 const vec3_t c111 = fn(vec3)(lut3d->output[hi2][hi1][hi0]);
987 const vec3_t c11 = fn(lerp3)(c110, c111, w0);
988
989 const pixel_t w1 = rgb.y - lo1;
990 const vec3_t c0 = fn(lerp3)(c00, c01, w1);
991 const vec3_t c1 = fn(lerp3)(c10, c11, w1);
992
993 const pixel_t w2 = rgb.z - lo2;
994 return fn(lerp3)(c0, c1, w2);
995}
996
997DECL_FUNC(lut3d, const SwsCompMask mask, const int dynamic)
998{
999 const SwsLut3D *restrict lut3d = impl->priv.ptr;
1000
1001 SWS_LOOP
1002 for (int i = 0; i < SWS_BLOCK_SIZE; i++) {
1003 vec3_t c = { x[i], y[i], z[i] };
1004 c = fn(lut3d_static)(lut3d, c);
1005 if (dynamic)
1006 c = fn(lut3d_dynamic)(lut3d, c);
1007
1008 x[i] = c.x;
1009 y[i] = c.y;
1010 z[i] = c.z;
1011 }
1012
1013 CONTINUE(x, y, z, w);
1014}
1015#endif /* IS_FLOAT */
1016
1017SWS_FOR(PX, LUT_3D, DECL_IMPL, lut3d)
1018SWS_FOR_STRUCT(PX, LUT_3D, DECL_ENTRY, .setup = fn(setup_lut3d) )
1019
1020#undef PIXEL_MAX
1021#undef PIXEL_SWAP
1022#undef pixel_t
1023#undef inter_t
1024#undef uinter_t
1025#undef vec3_t
1026#undef PX
1027#undef px
#define fn(a)
#define ZERO
uint8_t ptrdiff_t const uint8_t ptrdiff_t int intptr_t intptr_t my
Definition dsp.h:57
uint8_t ptrdiff_t const uint8_t ptrdiff_t int intptr_t mx
Definition dsp.h:57
uint8_t ptrdiff_t const uint8_t ptrdiff_t int intptr_t intptr_t int int16_t * dst
Definition dsp.h:87
SwsAArch64OpImplParams params
Definition ops.c:51
static double val(void *priv, double ch)
Definition aeval.c:77
static double mz(int i, double w0, double r, double alpha)
Definition af_atilt.c:55
static FILE * out
int32_t
#define av_assert2(cond)
assert() equivalent, that does lie in speed critical code.
Definition avassert.h:68
#define COPY(src, name)
static unsigned int BS_FUNC read_bit(BSCTX *bc)
Return one bit from the buffer.
#define MAX
Definition blend_modes.c:46
#define Y
Definition boxblur.h:37
byte swapping routines
#define i(width, name, range_min, range_max)
Definition cbs_h264.c:63
#define xs(width, name, var, subs,...)
Definition cbs_vp9.c:222
#define RSHIFT(a, b)
Definition common.h:56
#define NULL
Definition coverity.c:32
#define min(a, b)
#define max(a, b)
#define SCALE(c)
Definition dcadata.c:7338
#define ADD(a, b)
static void permute(int16_t dst[64], const int16_t src[64], enum idct_permutation_type perm_type)
Definition dct.c:158
static int unpack(const uint8_t *src, const uint8_t *src_end, uint8_t *dst, int width, int height)
Unpack buffer.
Definition eatgv.c:73
double value
Definition eval.c:102
#define X
Definition f_ebur128.c:157
#define AVERROR(e)
Definition error.h:45
void * av_memdup(const void *p, size_t size)
Duplicate a buffer with av_malloc().
Definition mem.c:408
int index
Definition gxfenc.c:90
int a
static const int weights[]
Definition hevc_pel.c:32
#define b
Definition input.c:43
static int linear(InterplayACMContext *s, unsigned ind, unsigned col)
static int zero(InterplayACMContext *s, unsigned ind, unsigned col)
static void scale(int *out, const int *in, const int w, const int h, const int shift)
Definition intra.c:278
#define W(a, i, v)
Definition jpegls.h:119
uint32_t type
Definition jpegmpfenc.c:80
#define ONE
Definition jrevdct.c:137
unsigned offset
Definition libaomenc.c:763
#define av_always_inline
Definition attributes.h:72
@ SWS_FILTER_SCALE
14-bit coefficients are picked to fit comfortably within int16_t for efficient SIMD processing (e....
Definition filters.h:40
@ TONE_LUT_SIZE
Definition lut3d.h:40
@ OUTPUT_LUT_SIZE_I
Definition lut3d.h:46
@ INPUT_LUT_SIZE
Definition lut3d.h:36
@ OUTPUT_LUT_SIZE_PT
Definition lut3d.h:47
uint8_t w
Definition llvidencdsp.c:39
static const uint16_t mask[17]
Definition lzw.c:38
#define FFSWAP(type, a, b)
Definition macros.h:52
#define FFMIN(a, b)
Definition macros.h:49
#define FFMAX(a, b)
Definition macros.h:47
#define FFALIGN(x, a)
Definition macros.h:78
void * av_calloc(size_t nmemb, size_t size)
Definition mem.c:370
static const uint64_t c1
Definition murmur3.c:52
const char data[16]
Definition mxf.c:149
int ff_sws_setup_vec4(const SwsImplParams *params, SwsImplResult *out)
Definition ops_chain.c:144
int ff_sws_setup_scalar(const SwsImplParams *params, SwsImplResult *out)
Definition ops_chain.c:129
static void ff_op_priv_unref(SwsOpPriv *priv)
Definition ops_chain.h:141
static void ff_op_priv_free(SwsOpPriv *priv)
Definition ops_chain.h:136
#define av_malloc(s)
Definition ops_static.c:52
static double lerp(double a, double b, double x)
Definition perlin.c:75
#define MIN(a, b)
const void * av_refstruct_ref_c(const void *obj)
Analog of av_refstruct_ref(), but for constant objects.
Definition refstruct.c:150
void * av_refstruct_ref(void *obj)
Create a new reference to an object managed via this API, i.e.
Definition refstruct.c:141
const h264_weight_func weight
Represents a computed filter kernel.
Definition filters.h:85
Append a set of operations for applying a gamut/tone mapping 3D LUT to the pixels.
Definition lut3d.h:50
Copyright (C) 2026 Niklas Haas.
int32_t * in_offset_x
Pixel offset map; for horizontal scaling, in bytes.
ptrdiff_t in_stride[4]
Definition uops.h:292
SwsUOpParams par
Definition uops.h:297
SwsPixel * ptr
Definition uops.h:302
union SwsUOp::@242237116251216327057105100216205033300341206345 data
SwsPixel mat4x5[4][5]
Definition uops.h:305
pixel_t m[4][4]
Definition uops_tmpl.c:803
pixel_t k[4]
Definition uops_tmpl.c:804
Definition rpzaenc.c:60
#define stride
void(* filter)(uint8_t *src, ptrdiff_t stride, int qscale)
Definition h263dsp.c:29
#define src
Definition vp8dsp.c:248
#define height
Definition dsp.h:89
#define PIXEL_MAX
Definition tiny_ssim.c:40
int size
SwsDitherUOp dither
Definition uops.h:288
uint32_t u32[SWS_BLOCK_SIZE]
Definition uops_tmpl.h:47
float f32[SWS_BLOCK_SIZE]
Definition uops_tmpl.h:48
uint16_t u16[SWS_BLOCK_SIZE]
Definition uops_tmpl.h:46
int ff_sws_dither_height(const SwsDitherUOp *dither)
Computes (1 << size_log2) + MAX(y_offset).
Definition uops.c:203
SwsPixelType
Definition uops.h:40
@ SWS_PIXEL_F32
Definition uops.h:45
#define SWS_UOP_MOVE_MAX
Definition uops.h:234
#define SWS_COMP_ELEMS(N)
Definition uops.h:125
uint8_t SwsCompMask
Bit-mask of components.
Definition uops.h:118
#define SWS_FOR(TYPE, UOP, MACRO,...)
Definition uops_macros.h:17
#define SWS_FOR_STRUCT(TYPE, UOP, MACRO,...)
Definition uops_macros.h:19
#define DECL_CAST(DST, dst)
Definition uops_tmpl.c:438
#define LIN_ROW(I, var)
#define DECL_IMPL_WRITE(...)
Definition uops_tmpl.h:138
#define DECL_WRITE(NAME,...)
Definition uops_tmpl.h:104
#define DECL_IMPL_READ(...)
Definition uops_tmpl.h:133
#define CONTINUE(...)
Definition uops_tmpl.h:112
#define Z
Definition uops_tmpl.h:88
#define DECL_FUNC(NAME,...)
Definition uops_tmpl.h:92
#define bump_ptr(ptr, bump)
Definition uops_tmpl.h:83
#define DECL_IMPL(FUNC, NAME, TYPE, UOP,...)
Definition uops_tmpl.h:124
#define SIZEOF_BLOCK
Definition uops_tmpl.h:51
#define SWS_BLOCK_SIZE
Copyright (C) 2026 Niklas Haas.
Definition uops_tmpl.h:41
#define DECL_READ(NAME,...)
Definition uops_tmpl.h:99
#define SWS_LOOP
Definition uops_tmpl.h:73
#define DECL_ENTRY(SETUP, NAME,...)
Definition uops_tmpl.h:144
#define DECL_SETUP(NAME, PARAMS, OUT)
Definition uops_tmpl.h:119
static const uint16_t dither[8][8]
Definition vf_gradfun.c:46
#define LINEAR
static void copy(const float *p1, float *p2, const int length)
uint8_t base
Definition vp3data.h:128
static int bias(int x, int c)
Definition vqcdec.c:115
static double c[64]
#define CLEAR(destin)
Definition wavpackenc.c:50
static int setup_linear(const SwsImplParams *params, SwsImplResult *out)
Definition ops.c:297
static int setup_filter_v(const SwsImplParams *params, SwsImplResult *out)
Definition ops.c:46
static int setup_dither(const SwsImplParams *params, SwsImplResult *out)
Definition ops.c:273
static int setup_filter_h(const SwsImplParams *params, SwsImplResult *out)
Definition ops.c:76