FFmpeg
Loading...
Searching...
No Matches
ops.c
Go to the documentation of this file.
1/**
2 * Copyright (C) 2025-2026 Niklas Haas
3 *
4 * This file is part of FFmpeg.
5 *
6 * FFmpeg is free software; you can redistribute it and/or
7 * modify it under the terms of the GNU Lesser General Public
8 * License as published by the Free Software Foundation; either
9 * version 2.1 of the License, or (at your option) any later version.
10 *
11 * FFmpeg is distributed in the hope that it will be useful,
12 * but WITHOUT ANY WARRANTY; without even the implied warranty of
13 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
14 * Lesser General Public License for more details.
15 *
16 * You should have received a copy of the GNU Lesser General Public
17 * License along with FFmpeg; if not, write to the Free Software
18 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
19 */
20
21#include <float.h>
22
23#include "libavutil/avassert.h"
24#include "libavutil/mem.h"
25#include "libavutil/x86/cpu.h"
26
27#include "../ops_chain.h"
28#include "../uops.h"
29#include "../uops_macros.h"
30
32{
33 const SwsUOp *uop = params->uop;
34
35 /* 3-component packed reads/writes process one extra garbage word */
36 if (uop->mask == SWS_COMP_ELEMS(3)) {
37 switch (uop->uop) {
38 case SWS_UOP_READ_PACKED: out->over_read[0] = sizeof(uint32_t); break;
39 case SWS_UOP_WRITE_PACKED: out->over_write[0] = sizeof(uint32_t); break;
40 }
41 }
42
43 return 0;
44}
45
47{
48 const SwsFilterWeights *filter = params->uop->data.kernel;
49 static_assert(sizeof(out->priv.ptr) <= sizeof(int32_t[2]),
50 ">8 byte pointers not supported");
51
52 /* Pre-convert weights to float */
53 float *weights = av_calloc(filter->num_weights, sizeof(float));
54 if (!weights)
55 return AVERROR(ENOMEM);
56
57 for (int i = 0; i < filter->num_weights; i++)
58 weights[i] = (float) filter->weights[i] / SWS_FILTER_SCALE;
59
60 out->priv.ptr = weights;
61 out->priv.uptr[1] = filter->filter_size;
62 out->free = ff_op_priv_free;
63 return 0;
64}
65
66static int hscale_sizeof_weight(const SwsUOp *uop)
67{
68 switch (uop->type) {
69 case SWS_PIXEL_U8: return sizeof(int16_t);
70 case SWS_PIXEL_U16: return sizeof(int16_t);
71 case SWS_PIXEL_F32: return sizeof(float);
72 default: return 0;
73 }
74}
75
77{
78 const SwsUOp *uop = params->uop;
79 const SwsFilterWeights *filter = uop->data.kernel;
80
81 /**
82 * `vpgatherdd` gathers 32 bits at a time; so if we're filtering a smaller
83 * size, we need to gather 2/4 taps simultaneously and unroll the inner
84 * loop over several packed samples.
85 */
86 const int pixel_size = ff_sws_pixel_type_size(uop->type);
87 const int taps_align = sizeof(int32_t) / pixel_size;
88 const int filter_size = filter->filter_size;
89 const int block_size = params->table->block_size;
90 const size_t aligned_size = FFALIGN(filter_size, taps_align);
91 const size_t line_size = FFALIGN(filter->dst_size, block_size);
92 av_assert1(FFALIGN(line_size, taps_align) == line_size);
93 if (aligned_size > INT_MAX)
94 return AVERROR(EINVAL);
95
96 union {
97 void *ptr;
98 int16_t *i16;
99 float *f32;
100 } weights;
101
102 const int sizeof_weight = hscale_sizeof_weight(uop);
103 weights.ptr = av_calloc(line_size, sizeof_weight * aligned_size);
104 if (!weights.ptr)
105 return AVERROR(ENOMEM);
106
107 /**
108 * Transpose filter weights to group (aligned) taps by block
109 */
110 const int mmsize = block_size * 2;
111 const int gather_size = mmsize / sizeof(int32_t); /* pixels per vpgatherdd */
112 for (size_t x = 0; x < line_size; x += block_size) {
113 const int elems = FFMIN(block_size, filter->dst_size - x);
114 for (int j = 0; j < filter_size; j++) {
115 const int jb = j & ~(taps_align - 1);
116 const int ji = j - jb;
117 const size_t idx_base = x * aligned_size + jb * block_size + ji;
118 for (int i = 0; i < elems; i++) {
119 const int w = filter->weights[(x + i) * filter_size + j];
120 size_t idx = idx_base;
121 if (uop->type == SWS_PIXEL_U8) {
122 /* Interleave the pixels within each lane, i.e.:
123 * [a0 a1 a2 a3 | b0 b1 b2 b3 ] pixels 0-1, taps 0-3 (lane 0)
124 * [e0 e1 e2 e3 | f0 f1 f2 f3 ] pixels 4-5, taps 0-3 (lane 1)
125 * [c0 c1 c2 c3 | d0 d1 d2 d3 ] pixels 2-3, taps 0-3 (lane 0)
126 * [g0 g1 g2 g3 | h0 h1 h2 h3 ] pixels 6-7, taps 0-3 (lane 1)
127 * [i0 i1 i2 i3 | j0 j1 j2 j3 ] pixels 8-9, taps 0-3 (lane 0)
128 * ...
129 * [o0 o1 o2 o3 | p0 p1 p2 p3 ] pixels 14-15, taps 0-3 (lane 1)
130 * (repeat for taps 4-7, etc.)
131 */
132 const int gather_base = i & ~(gather_size - 1);
133 const int gather_pos = i - gather_base;
134 const int lane_idx = gather_pos >> 2;
135 const int pos_in_lane = gather_pos & 3;
136 idx += gather_base * 4 /* which gather (m0 or m1) */
137 + (pos_in_lane >> 1) * (mmsize / 2) /* lo/hi unpack */
138 + lane_idx * 8 /* 8 ints per lane */
139 + (pos_in_lane & 1) * 4; /* 4 taps per pair */
140 } else {
141 idx += i * taps_align;
142 }
143
144 switch (uop->type) {
145 case SWS_PIXEL_U8: weights.i16[idx] = w; break;
146 case SWS_PIXEL_U16: weights.i16[idx] = w; break;
147 case SWS_PIXEL_F32: weights.f32[idx] = w; break;
148 }
149 }
150 }
151 }
152
153 out->priv.ptr = weights.ptr;
154 out->priv.uptr[1] = aligned_size;
155 out->free = ff_op_priv_free;
156
157 for (int i = 0; i < 4; i++) {
158 if (uop->mask & SWS_COMP(i))
159 out->over_read[i] = (aligned_size - filter_size) * pixel_size;
160 }
161 return 0;
162}
163
165{
166 SwsContext *ctx = params->ctx;
167 const SwsUOp *uop = params->uop;
168 if ((ctx->flags & SWS_BITEXACT) && uop->type == SWS_PIXEL_F32)
169 return false; /* different accumulation order due to 4x4 transpose */
170
171 const int cpu_flags = av_get_cpu_flags();
173 return true; /* always prefer over gathers if gathers are slow */
174
175 /**
176 * Otherwise, prefer it above a certain filter size. Empirically, this
177 * kernel seems to be faster whenever the reference/gather kernel crosses
178 * a breakpoint for the number of gathers needed, but this filter doesn't.
179 *
180 * Tested on a Lunar Lake (Intel Core Ultra 7 258V) system.
181 */
182 const SwsFilterWeights *filter = uop->data.kernel;
183 return uop->type == SWS_PIXEL_U8 && filter->filter_size > 12 ||
184 uop->type == SWS_PIXEL_U16 && filter->filter_size > 4 ||
185 uop->type == SWS_PIXEL_F32 && filter->filter_size > 1;
186}
187
189{
190 const SwsUOp *uop = params->uop;
191 const SwsFilterWeights *filter = uop->data.kernel;
192 const int pixel_size = ff_sws_pixel_type_size(uop->type);
193 const int sizeof_weights = hscale_sizeof_weight(uop);
194 const int block_size = params->table->block_size;
195 const int taps_align = 16 / sizeof_weights; /* taps per iteration (XMM) */
196 const int pixels_align = 4; /* pixels per iteration */
197 const int filter_size = filter->filter_size;
198 const size_t aligned_size = FFALIGN(filter_size, taps_align);
199 const int line_size = FFALIGN(filter->dst_size, block_size);
200 av_assert1(FFALIGN(line_size, pixels_align) == line_size);
201
202 union {
203 void *ptr;
204 int16_t *i16;
205 float *f32;
206 } weights;
207
208 weights.ptr = av_calloc(line_size, aligned_size * sizeof_weights);
209 if (!weights.ptr)
210 return AVERROR(ENOMEM);
211
212 /**
213 * Desired memory layout: [w][taps][pixels_align][taps_align]
214 *
215 * Example with taps_align=8, pixels_align=4:
216 * [a0, a1, ... a7] weights for pixel 0, taps 0..7
217 * [b0, b1, ... b7] weights for pixel 1, taps 0..7
218 * [c0, c1, ... c7] weights for pixel 2, taps 0..7
219 * [d0, d1, ... d7] weights for pixel 3, taps 0..7
220 * [a8, a9, ... a15] weights for pixel 0, taps 8..15
221 * ...
222 * repeat for all taps, then move on to pixels 4..7, etc.
223 */
224 for (int x = 0; x < filter->dst_size; x++) {
225 for (int j = 0; j < filter_size; j++) {
226 const int xb = x & ~(pixels_align - 1);
227 const int jb = j & ~(taps_align - 1);
228 const int xi = x - xb, ji = j - jb;
229 const int w = filter->weights[x * filter_size + j];
230 const int idx = xb * aligned_size + jb * pixels_align + xi * taps_align + ji;
231
232 switch (uop->type) {
233 case SWS_PIXEL_U8: weights.i16[idx] = w; break;
234 case SWS_PIXEL_U16: weights.i16[idx] = w; break;
235 case SWS_PIXEL_F32: weights.f32[idx] = w; break;
236 }
237 }
238 }
239
240 out->priv.ptr = weights.ptr;
241 out->priv.uptr[1] = aligned_size * sizeof_weights;
242 out->free = ff_op_priv_free;
243
244 for (int i = 0; i < 4; i++) {
245 if (uop->mask & SWS_COMP(i))
246 out->over_read[i] = (aligned_size - filter_size) * pixel_size;
247 }
248 return 0;
249}
250
252{
253 const SwsUOp *uop = params->uop;
254 switch (uop->type) {
255 case SWS_PIXEL_U8: out->priv.u16[0] = uop->data.scalar.u8; break; /* for pmullw */
256 case SWS_PIXEL_U16: out->priv.u16[0] = uop->data.scalar.u16; break;
257 case SWS_PIXEL_U32: out->priv.u32[0] = uop->data.scalar.u32; break;
258 case SWS_PIXEL_F32: out->priv.f32[0] = uop->data.scalar.f32; break;
259 default: return AVERROR(EINVAL);
260 }
261
262 return 0;
263}
264
266{
267 const SwsUOp *uop = params->uop;
268 for (int i = 0; i < 4; i++)
269 out->priv.u32[i] = uop->data.vec4[i].u32;
270 return 0;
271}
272
274{
275 out->priv.ptr = av_refstruct_ref(params->uop->data.ptr);
276 out->free = ff_op_priv_unref;
277 return 0;
278}
279
280static void splat_lane(void *dst, SwsPixelType type, SwsPixel px)
281{
282 switch (ff_sws_pixel_type_size(type)) {
283 case 1:
284 memset(dst, px.u8, 16);
285 break;
286 case 2:
287 for (int i = 0; i < 8; i++)
288 ((uint16_t *) dst)[i] = px.u16;
289 break;
290 case 4:
291 for (int i = 0; i < 4; i++)
292 ((uint32_t *) dst)[i] = px.u32;
293 break;
294 }
295}
296
298{
299 const SwsUOp *uop = params->uop;
300 if (uop->type == SWS_PIXEL_F32) {
301 out->priv.ptr = av_memdup(uop->data.mat4x5, sizeof(uop->data.mat4x5));
302 out->free = ff_op_priv_free;
303 return out->priv.ptr ? 0 : AVERROR(ENOMEM);
304 }
305
306 uint8_t *mat = av_malloc(4 * 5 * 16); /* one lane per component */
307 if (!mat)
308 return AVERROR(ENOMEM);
309 out->priv.ptr = mat;
310 out->free = ff_op_priv_free;
311
312 for (int i = 0; i < 4; i++) {
313 for (int j = 0; j < 5; j++) {
314 SwsPixel px = uop->data.mat4x5[i][j];
315 SwsPixelType type = uop->type;
316 if (type == SWS_PIXEL_U8) {
317 type = SWS_PIXEL_U16; /* for pmullw */
318 px.u16 = (px.u8 << 8) | px.u8;
319 }
320
321 splat_lane(mat, type, px);
322 mat += 16;
323 }
324 }
325 return 0;
326}
327
328static bool uop_is_type_invariant(const SwsUOpType uop)
329{
330 switch (uop) {
333 case SWS_UOP_CLEAR:
334 return true;
335 default:
336 return false;
337 }
338}
339
340#define REF_ENTRY(EXT, NAME, ...) &uop_##NAME##EXT,
341#define DECL_ENTRY(EXT, CHECK, SETUP, NAME, ...) \
342 void ff_##NAME##EXT(void); \
343 static const SwsUOpEntry uop_##NAME##EXT = { \
344 .func = (SwsFuncPtr) ff_##NAME##EXT, \
345 .check = CHECK, \
346 .setup = SETUP, \
347 __VA_ARGS__, \
348 };
349
350/* Define all UOPs except conversion ops and type-invariant ops */
351#define DECL_OPS_COMMON(EXT, TYPE) \
352SWS_FOR_STRUCT(TYPE, READ_PACKED, DECL_ENTRY, EXT, NULL, setup_rw_packed) \
353SWS_FOR_STRUCT(TYPE, READ_NIBBLE, DECL_ENTRY, EXT, NULL, NULL) \
354SWS_FOR_STRUCT(TYPE, READ_BIT, DECL_ENTRY, EXT, NULL, NULL) \
355SWS_FOR_STRUCT(TYPE, READ_PALETTE, DECL_ENTRY, EXT, NULL, NULL) \
356SWS_FOR_STRUCT(TYPE, WRITE_PACKED, DECL_ENTRY, EXT, NULL, setup_rw_packed) \
357SWS_FOR_STRUCT(TYPE, WRITE_NIBBLE, DECL_ENTRY, EXT, NULL, NULL) \
358SWS_FOR_STRUCT(TYPE, WRITE_BIT, DECL_ENTRY, EXT, NULL, NULL) \
359SWS_FOR_STRUCT(TYPE, SWAP_BYTES, DECL_ENTRY, EXT, NULL, NULL) \
360SWS_FOR_STRUCT(TYPE, EXPAND_BIT, DECL_ENTRY, EXT, NULL, NULL) \
361SWS_FOR_STRUCT(TYPE, PERMUTE, DECL_ENTRY, EXT, NULL, NULL) \
362SWS_FOR_STRUCT(TYPE, COPY, DECL_ENTRY, EXT, NULL, NULL) \
363SWS_FOR_STRUCT(TYPE, SCALE, DECL_ENTRY, EXT, NULL, setup_scale) \
364SWS_FOR_STRUCT(TYPE, ADD, DECL_ENTRY, EXT, NULL, ff_sws_setup_vec4) \
365SWS_FOR_STRUCT(TYPE, MIN, DECL_ENTRY, EXT, NULL, ff_sws_setup_vec4) \
366SWS_FOR_STRUCT(TYPE, MAX, DECL_ENTRY, EXT, NULL, ff_sws_setup_vec4) \
367SWS_FOR_STRUCT(TYPE, UNPACK, DECL_ENTRY, EXT, NULL, NULL) \
368SWS_FOR_STRUCT(TYPE, PACK, DECL_ENTRY, EXT, NULL, NULL) \
369SWS_FOR_STRUCT(TYPE, LSHIFT, DECL_ENTRY, EXT, NULL, NULL) \
370SWS_FOR_STRUCT(TYPE, RSHIFT, DECL_ENTRY, EXT, NULL, NULL) \
371SWS_FOR_STRUCT(TYPE, LINEAR, DECL_ENTRY, EXT, NULL, setup_linear) \
372SWS_FOR_STRUCT(TYPE, LINEAR_FMA, DECL_ENTRY, EXT, NULL, setup_linear) \
373SWS_FOR_STRUCT(TYPE, DITHER, DECL_ENTRY, EXT, NULL, setup_dither) \
374/* end of macro */
375
376#define REF_OPS_COMMON(EXT, TYPE) \
377 SWS_FOR(TYPE, READ_PACKED, REF_ENTRY, EXT) \
378 SWS_FOR(TYPE, READ_NIBBLE, REF_ENTRY, EXT) \
379 SWS_FOR(TYPE, READ_BIT, REF_ENTRY, EXT) \
380 SWS_FOR(TYPE, READ_PALETTE, REF_ENTRY, EXT) \
381 SWS_FOR(TYPE, WRITE_PACKED, REF_ENTRY, EXT) \
382 SWS_FOR(TYPE, WRITE_NIBBLE, REF_ENTRY, EXT) \
383 SWS_FOR(TYPE, WRITE_BIT, REF_ENTRY, EXT) \
384 SWS_FOR(TYPE, SWAP_BYTES, REF_ENTRY, EXT) \
385 SWS_FOR(TYPE, EXPAND_BIT, REF_ENTRY, EXT) \
386 SWS_FOR(TYPE, PERMUTE, REF_ENTRY, EXT) \
387 SWS_FOR(TYPE, COPY, REF_ENTRY, EXT) \
388 SWS_FOR(TYPE, SCALE, REF_ENTRY, EXT) \
389 SWS_FOR(TYPE, ADD, REF_ENTRY, EXT) \
390 SWS_FOR(TYPE, MIN, REF_ENTRY, EXT) \
391 SWS_FOR(TYPE, MAX, REF_ENTRY, EXT) \
392 SWS_FOR(TYPE, UNPACK, REF_ENTRY, EXT) \
393 SWS_FOR(TYPE, PACK, REF_ENTRY, EXT) \
394 SWS_FOR(TYPE, LSHIFT, REF_ENTRY, EXT) \
395 SWS_FOR(TYPE, RSHIFT, REF_ENTRY, EXT) \
396 SWS_FOR(TYPE, LINEAR, REF_ENTRY, EXT) \
397 SWS_FOR(TYPE, LINEAR_FMA, REF_ENTRY, EXT) \
398 SWS_FOR(TYPE, DITHER, REF_ENTRY, EXT) \
399 /* end of macro */
400
401#define DECL_TABLE_U8(EXT, SIZE, FLAG) \
402DECL_OPS_COMMON(EXT, U8) \
403SWS_FOR_STRUCT(U8, READ_PLANAR, DECL_ENTRY, EXT, NULL, NULL) \
404SWS_FOR_STRUCT(U8, WRITE_PLANAR, DECL_ENTRY, EXT, NULL, NULL) \
405SWS_FOR_STRUCT(U8, CLEAR, DECL_ENTRY, EXT, NULL, setup_clear) \
406 \
407static const SwsUOpTable uops_u8##EXT = { \
408 .cpu_flags = AV_CPU_FLAG_##FLAG, \
409 .block_size = SIZE, \
410 .entries = { \
411 REF_OPS_COMMON(EXT, U8) \
412 SWS_FOR(U8, READ_PLANAR, REF_ENTRY, EXT) \
413 SWS_FOR(U8, WRITE_PLANAR, REF_ENTRY, EXT) \
414 SWS_FOR(U8, CLEAR, REF_ENTRY, EXT) \
415 NULL \
416 }, \
417};
418
419#define DECL_TABLE_U16(EXT, SIZE, FLAG) \
420DECL_OPS_COMMON(EXT, U16) \
421SWS_FOR_STRUCT(U8, TO_U16, DECL_ENTRY, EXT, NULL, NULL) \
422SWS_FOR_STRUCT(U16, TO_U8, DECL_ENTRY, EXT, NULL, NULL) \
423SWS_FOR_STRUCT(U8, EXPAND_PAIR, DECL_ENTRY, EXT, NULL, NULL) \
424 \
425static const SwsUOpTable uops_u16##EXT = { \
426 .cpu_flags = AV_CPU_FLAG_##FLAG, \
427 .block_size = SIZE, \
428 .entries = { \
429 REF_OPS_COMMON(EXT, U16) \
430 SWS_FOR(U8, TO_U16, REF_ENTRY, EXT) \
431 SWS_FOR(U16, TO_U8, REF_ENTRY, EXT) \
432 SWS_FOR(U8, EXPAND_PAIR, REF_ENTRY, EXT) \
433 NULL \
434 }, \
435};
436
437#define DECL_TABLE_U32(EXT, SIZE, FLAG) \
438DECL_OPS_COMMON(EXT, U32) \
439SWS_FOR_STRUCT(U8, TO_U32, DECL_ENTRY, EXT, NULL, NULL) \
440SWS_FOR_STRUCT(U32, TO_U8, DECL_ENTRY, EXT, NULL, NULL) \
441SWS_FOR_STRUCT(U16, TO_U32, DECL_ENTRY, EXT, NULL, NULL) \
442SWS_FOR_STRUCT(U32, TO_U16, DECL_ENTRY, EXT, NULL, NULL) \
443SWS_FOR_STRUCT(U8, EXPAND_QUAD, DECL_ENTRY, EXT, NULL, NULL) \
444 \
445static const SwsUOpTable uops_u32##EXT = { \
446 .cpu_flags = AV_CPU_FLAG_##FLAG, \
447 .block_size = SIZE, \
448 .entries = { \
449 REF_OPS_COMMON(EXT, U32) \
450 SWS_FOR(U8, TO_U32, REF_ENTRY, EXT) \
451 SWS_FOR(U32, TO_U8, REF_ENTRY, EXT) \
452 SWS_FOR(U16, TO_U32, REF_ENTRY, EXT) \
453 SWS_FOR(U32, TO_U16, REF_ENTRY, EXT) \
454 SWS_FOR(U8, EXPAND_QUAD, REF_ENTRY, EXT) \
455 NULL \
456 }, \
457};
458
459#define DECL_TABLE_F32(EXT, SIZE, FLAG) \
460DECL_OPS_COMMON(EXT, F32) \
461SWS_FOR_STRUCT(U8, TO_F32, DECL_ENTRY, EXT, NULL, NULL) \
462SWS_FOR_STRUCT(F32, TO_U8, DECL_ENTRY, EXT, NULL, NULL) \
463SWS_FOR_STRUCT(U16, TO_F32, DECL_ENTRY, EXT, NULL, NULL) \
464SWS_FOR_STRUCT(F32, TO_U16, DECL_ENTRY, EXT, NULL, NULL) \
465SWS_FOR_STRUCT(U8, READ_PLANAR_FH, DECL_ENTRY, EXT, NULL, setup_filter_h) \
466SWS_FOR_STRUCT(U16, READ_PLANAR_FH, DECL_ENTRY, EXT, NULL, setup_filter_h) \
467SWS_FOR_STRUCT(F32, READ_PLANAR_FH, DECL_ENTRY, EXT, NULL, setup_filter_h) \
468SWS_FOR_STRUCT(U8, READ_PLANAR_FH, DECL_ENTRY, _4x4##EXT, \
469 check_filter_h_4x4, setup_filter_h_4x4) \
470SWS_FOR_STRUCT(U16, READ_PLANAR_FH, DECL_ENTRY, _4x4##EXT, \
471 check_filter_h_4x4, setup_filter_h_4x4) \
472SWS_FOR_STRUCT(F32, READ_PLANAR_FH, DECL_ENTRY, _4x4##EXT, \
473 check_filter_h_4x4, setup_filter_h_4x4) \
474SWS_FOR_STRUCT(U8, READ_PLANAR_FV, DECL_ENTRY, EXT, NULL, setup_filter_v) \
475SWS_FOR_STRUCT(U16, READ_PLANAR_FV, DECL_ENTRY, EXT, NULL, setup_filter_v) \
476SWS_FOR_STRUCT(F32, READ_PLANAR_FV, DECL_ENTRY, EXT, NULL, setup_filter_v) \
477SWS_FOR_STRUCT(U8, READ_PLANAR_FV_FMA, DECL_ENTRY, EXT, NULL, setup_filter_v) \
478SWS_FOR_STRUCT(U16, READ_PLANAR_FV_FMA, DECL_ENTRY, EXT, NULL, setup_filter_v) \
479SWS_FOR_STRUCT(F32, READ_PLANAR_FV_FMA, DECL_ENTRY, EXT, NULL, setup_filter_v) \
480 \
481static const SwsUOpTable uops_f32##EXT = { \
482 .cpu_flags = AV_CPU_FLAG_##FLAG, \
483 .block_size = SIZE, \
484 .entries = { \
485 REF_OPS_COMMON(EXT, F32) \
486 SWS_FOR(U8, TO_F32, REF_ENTRY, EXT) \
487 SWS_FOR(F32, TO_U8, REF_ENTRY, EXT) \
488 SWS_FOR(U16, TO_F32, REF_ENTRY, EXT) \
489 SWS_FOR(F32, TO_U16, REF_ENTRY, EXT) \
490 SWS_FOR(U8, READ_PLANAR_FH, REF_ENTRY, _4x4##EXT) \
491 SWS_FOR(U16, READ_PLANAR_FH, REF_ENTRY, _4x4##EXT) \
492 SWS_FOR(F32, READ_PLANAR_FH, REF_ENTRY, _4x4##EXT) \
493 SWS_FOR(U8, READ_PLANAR_FH, REF_ENTRY, EXT) \
494 SWS_FOR(U16, READ_PLANAR_FH, REF_ENTRY, EXT) \
495 SWS_FOR(F32, READ_PLANAR_FH, REF_ENTRY, EXT) \
496 SWS_FOR(U8, READ_PLANAR_FV, REF_ENTRY, EXT) \
497 SWS_FOR(U16, READ_PLANAR_FV, REF_ENTRY, EXT) \
498 SWS_FOR(F32, READ_PLANAR_FV, REF_ENTRY, EXT) \
499 SWS_FOR(U8, READ_PLANAR_FV_FMA, REF_ENTRY, EXT) \
500 SWS_FOR(U16, READ_PLANAR_FV_FMA, REF_ENTRY, EXT) \
501 SWS_FOR(F32, READ_PLANAR_FV_FMA, REF_ENTRY, EXT) \
502 NULL \
503 }, \
504};
505
506DECL_TABLE_U8( _m1_sse4, 16, SSE4)
507DECL_TABLE_U8( _m1_avx2, 32, AVX2)
508DECL_TABLE_U8( _m2_sse4, 32, SSE4)
509DECL_TABLE_U8( _m2_avx2, 64, AVX2)
510DECL_TABLE_U16(_m1_avx2, 16, AVX2)
511DECL_TABLE_U16(_m2_avx2, 32, AVX2)
512DECL_TABLE_U32(_m2_avx2, 16, AVX2)
513DECL_TABLE_F32(_m2_avx2, 16, AVX2)
514
515static const SwsUOpTable *const tables[] = {
516 &uops_u8_m1_sse4,
517 &uops_u8_m1_avx2, /* order before _m2_sse4 */
518 &uops_u8_m2_sse4,
519 &uops_u8_m2_avx2,
520 &uops_u16_m1_avx2,
521 &uops_u16_m2_avx2,
522 &uops_u32_m2_avx2,
523 &uops_f32_m2_avx2,
524};
525
526SWS_DECL_FUNC(ff_sws_process1_x86);
527SWS_DECL_FUNC(ff_sws_process2_x86);
528SWS_DECL_FUNC(ff_sws_process3_x86);
529SWS_DECL_FUNC(ff_sws_process4_x86);
530
531/* Declare packed shuffle functions */
532SWS_FOR_STRUCT(U8, RW_SHUFFLE, DECL_ENTRY, _sse4, NULL, NULL)
533SWS_FOR_STRUCT(U8, RW_SHUFFLE, DECL_ENTRY, _avx2, NULL, NULL)
534SWS_FOR_STRUCT(U8, RW_SHUFFLE, DECL_ENTRY, _avx512, NULL, NULL)
535SWS_FOR_STRUCT(U8, RW_SHUFFLE, DECL_ENTRY, _avx512icl, NULL, NULL)
536
537static int get_mmsize(void)
538{
539 const int cpu_flags = av_get_cpu_flags();
541 return 64;
542 else if (EXTERNAL_AVX2(cpu_flags))
543 return 32;
544 else if (EXTERNAL_SSE4(cpu_flags))
545 return 16;
546 else
547 return AVERROR(ENOTSUP);
548}
549
550static int movsize(const int bytes, const int mmsize)
551{
552 return bytes <= 4 ? 4 : /* movd */
553 bytes <= 8 ? 8 : /* movq */
554 bytes <= 16 ? 16 : /* xmm movu */
555 bytes <= 32 ? 32 : /* ymm movu */
556 mmsize; /* zmm movu */
557}
558
559static int translate_shuffle(const SwsUOp *uop, int mmsize, SwsCompiledOp *out)
560{
561 /* We can't shuffle across lanes, so restrict the vector size to XMM
562 * whenever the read/write size would be a subset of the full vector,
563 * unless we have access to AVX-512 ICL vpermb */
564 const SwsShuffleUOp *par = &uop->par.shuffle;
565 const int lane_aligned = par->read_size == par->write_size &&
566 16 % par->read_size == 0;
567 if (!lane_aligned && !EXTERNAL_AVX512ICL(av_get_cpu_flags()))
568 mmsize = 16;
569
570 /* Generate the shuffle mask */
571 const int mask_size = lane_aligned ? 16 : mmsize;
572 int8_t *mask = av_malloc(mask_size);
573 if (!mask)
574 return AVERROR(ENOMEM);
575
576 const int groups = ff_sws_shuffle_mask(uop, mask, mask_size);
577 if (groups < 0) {
578 av_free(mask);
579 return groups;
580 }
581
582 const int read_chunk = groups * par->read_size;
583 const int write_chunk = groups * par->write_size;
584 const int num_lanes = lane_aligned ? mmsize / 16 : 1;
585 const int in_total = num_lanes * read_chunk;
586 const int out_total = num_lanes * write_chunk;
587 *out = (SwsCompiledOp) {
588 .priv = mask,
589 .free = av_free,
590 .slice_align = 1,
591 .block_size = groups * uop->data.shuffle.pixels * num_lanes,
592 .over_read = { movsize(in_total, mmsize) - in_total },
593 .over_write = { movsize(out_total, mmsize) - out_total },
594 };
595
596#define ASSIGN_SHUFFLE_FUNC(CPU, EXT, NAME, ...) \
597do { \
598 const SwsUOpEntry *entry = &uop_##NAME##EXT; \
599 if (!memcmp(&uop->par, &entry->par, sizeof(uop->par))) { \
600 out->func = (SwsOpFunc) entry->func; \
601 out->cpu_flags = AV_CPU_FLAG_##CPU; \
602 } \
603} while (0);
604
605 switch (mmsize) {
606 case 16: SWS_FOR(U8, RW_SHUFFLE, ASSIGN_SHUFFLE_FUNC, SSE4, _sse4); break;
607 case 32: SWS_FOR(U8, RW_SHUFFLE, ASSIGN_SHUFFLE_FUNC, AVX2, _avx2); break;
608 case 64:
609 if (lane_aligned) {
610 SWS_FOR(U8, RW_SHUFFLE, ASSIGN_SHUFFLE_FUNC, AVX512, _avx512);
611 } else { /* vpermb variant */
612 SWS_FOR(U8, RW_SHUFFLE, ASSIGN_SHUFFLE_FUNC, AVX512ICL, _avx512icl);
613 }
614 break;
615 }
616
617 if (!out->func) {
618 av_free(mask);
619 return AVERROR(ENOTSUP);
620 }
621
622 return 0;
623}
624
625/* Expand pixel value to 32-bits by repeating as necessary */
626static uint32_t expand32(const SwsPixelType type, const SwsPixel value)
627{
628 switch (type) {
629 case SWS_PIXEL_U8: return value.u8 * 0x01010101u;
630 case SWS_PIXEL_U16: return value.u16 * 0x00010001u;
631 case SWS_PIXEL_U32: return value.u32;
632 case SWS_PIXEL_F32: return value.u32; /* reinterpret */
633 default: return 0;
634 }
635}
636
637static void normalize_clear(SwsUOp *uop)
638{
639 for (int i = 0; i < 4; i++)
640 uop->data.vec4[i].u32 = expand32(uop->type, uop->data.vec4[i]);
641}
642
644{
645 int ret, mmsize = get_mmsize();
646 if (mmsize < 0)
647 return mmsize;
648
649 if (uops->num_ops == 1 && uops->ops[0].uop == SWS_UOP_RW_SHUFFLE) {
650 const SwsUOp *uop = &uops->ops[0];
651 ret = translate_shuffle(uop, mmsize, out);
652 if (ret >= 0) {
654 ff_sws_uop_name(uop, name);
655 av_log(ctx, AV_LOG_VERBOSE, "Using x86 packed shuffle fast path: %s\n", name);
656 }
657 return ret;
658 }
659
661 if (!chain)
662 return AVERROR(ENOMEM);
663
664 *out = (SwsCompiledOp) {
665 /* Use at most two full YMM regs during the widest precision section */
666 .block_size = 2 * FFMIN(mmsize, 32) / uops->pixel_size_max,
667 .slice_align = 1,
669 .priv = chain,
670 };
671
672 for (int i = 0; i < uops->num_ops; i++) {
673 SwsUOp *uop = &uops->ops[i];
674 int op_block_size = out->block_size;
675
676 if (uop_is_type_invariant(uop->uop)) {
677 if (uop->uop == SWS_UOP_CLEAR)
678 normalize_clear(uop);
679 op_block_size *= ff_sws_pixel_type_size(uop->type);
680 uop->type = SWS_PIXEL_U8;
681 }
682
684 op_block_size, chain);
685 if (ret < 0)
686 goto fail;
687 }
688
689 switch (av_popcount(uops->planes_in | uops->planes_out)) {
690 case 1: out->func = ff_sws_process1_x86; break;
691 case 2: out->func = ff_sws_process2_x86; break;
692 case 3: out->func = ff_sws_process3_x86; break;
693 case 4: out->func = ff_sws_process4_x86; break;
694 }
695
696 if (ret < 0) {
698 return ret;
699 }
700
701 out->cpu_flags = chain->cpu_flags;
702 memcpy(out->over_read, chain->over_read, sizeof(out->over_read));
703 memcpy(out->over_write, chain->over_write, sizeof(out->over_write));
704
705 av_log(ctx, AV_LOG_DEBUG, "Compiled micro-ops:\n");
706 for (int i = 0; i < uops->num_ops; i++) {
708 ff_sws_uop_name(&uops->ops[i], name);
709 av_log(ctx, AV_LOG_DEBUG, " %s\n", name);
710 }
711
712 return 0;
713
714fail:
716 return ret;
717}
718
720{
721 const int cpu_flags = av_get_cpu_flags();
722 const int mmsize = get_mmsize();
723 if (mmsize < 0)
724 return mmsize;
725
732
734 if (!uops)
735 return AVERROR(ENOMEM);
736
737 int ret = ff_sws_ops_translate(ctx, ops, flags, uops);
738 if (ret < 0)
739 goto fail;
740
741 ret = compile_uops_x86(ctx, uops, out);
742
743fail:
745 return ret;
746}
747
749 .name = "x86",
750 .flags = SWS_BACKEND_X86,
751 .compile = compile_x86,
752 .compile_uops = compile_uops_x86,
753 .hw_format = AV_PIX_FMT_NONE,
754};
uint8_t ptrdiff_t const uint8_t ptrdiff_t int intptr_t intptr_t int int16_t * dst
Definition dsp.h:87
SwsAArch64OpImplParams params
Definition ops.c:51
static FILE * out
static AVFormatContext * ctx
int32_t
simple assert() macros that are a bit more flexible than ISO C assert().
#define av_assert1(cond)
assert() equivalent, that does not lie in speed critical code.
Definition avassert.h:58
#define flags(name, subs,...)
Definition cbs_h264.c:74
#define i(width, name, range_min, range_max)
Definition cbs_h264.c:63
#define xi(width, name, var, range_min, range_max, subs,...)
Definition cbs_h264.c:115
#define av_popcount
Definition common.h:154
#define NULL
Definition coverity.c:32
static int read_chunk(AVFormatContext *s)
Definition dhav.c:173
double value
Definition eval.c:102
static const OptionGroupDef groups[]
#define fail
Definition test.h:479
#define AVERROR(e)
Definition error.h:45
#define AV_LOG_DEBUG
Stuff which is only useful for libav* developers.
Definition log.h:231
#define AV_LOG_VERBOSE
Detailed information.
Definition log.h:226
void * av_memdup(const void *p, size_t size)
Duplicate a buffer with av_malloc().
Definition mem.c:408
@ SWS_BACKEND_X86
Chained x86 SIMD kernels.
Definition swscale.h:118
@ SWS_BITEXACT
Definition swscale.h:178
static const int weights[]
Definition hevc_pel.c:32
uint32_t type
Definition jpegmpfenc.c:80
static atomic_int cpu_flags
Definition cpu.c:56
int av_get_cpu_flags(void)
Return the flags which specify extensions supported by the CPU.
Definition cpu.c:109
#define AV_CPU_FLAG_SLOW_GATHER
CPU has slow gathers.
Definition cpu.h:62
#define EXTERNAL_AVX512ICL(flags)
Definition cpu.h:78
#define EXTERNAL_FMA3(flags)
Definition cpu.h:68
#define EXTERNAL_SSE4(flags)
Definition cpu.h:62
#define EXTERNAL_AVX512(flags)
Definition cpu.h:77
#define EXTERNAL_AVX2(flags)
Definition cpu.h:72
@ SWS_FILTER_SCALE
14-bit coefficients are picked to fit comfortably within int16_t for efficient SIMD processing (e....
Definition filters.h:40
uint8_t w
Definition llvidencdsp.c:39
static const uint16_t mask[17]
Definition lzw.c:38
#define FFMIN(a, b)
Definition macros.h:49
#define FFALIGN(x, a)
Definition macros.h:78
void * av_calloc(size_t nmemb, size_t size)
Definition mem.c:370
Memory handling functions.
const SwsOpBackend backend_x86
Definition ops.c:748
SwsOpChain * ff_sws_op_chain_alloc(void)
Copyright (C) 2025 Niklas Haas.
Definition ops_chain.c:26
int ff_sws_uop_lookup(SwsContext *ctx, const SwsUOpTable *const tables[], int num_tables, const SwsUOp *uop, const int block_size, SwsOpChain *chain)
"Compile" a single uop by looking it up in a list of fixed size uop tables, in decreasing order of pr...
Definition ops_chain.c:60
void ff_sws_op_chain_free_cb(void *ptr)
Definition ops_chain.c:31
static void ff_sws_op_chain_free(SwsOpChain *chain)
Definition ops_chain.h:96
static void ff_op_priv_unref(SwsOpPriv *priv)
Definition ops_chain.h:141
static void ff_op_priv_free(SwsOpPriv *priv)
Definition ops_chain.h:136
#define SWS_DECL_FUNC(NAME)
int ff_sws_shuffle_mask(const SwsUOp *uop, int8_t shuffle[], int size)
Compute a shuffle mask for pshufb-style ASM functions, by repeating the shuffle pattern for as many g...
#define av_malloc(s)
Definition ops_static.c:52
@ AV_PIX_FMT_NONE
Definition pixfmt.h:72
const char * name
Definition qsvenc.c:142
void * av_refstruct_ref(void *obj)
Create a new reference to an object managed via this API, i.e.
Definition refstruct.c:141
#define FF_ARRAY_ELEMS(a)
Main external API structure.
Definition swscale.h:227
Represents a computed filter kernel.
Definition filters.h:85
Compiled "chain" of operations, which can be dispatched efficiently.
Definition ops_chain.h:84
int over_read[4]
Definition ops_chain.h:90
int over_write[4]
Definition ops_chain.h:91
int cpu_flags
Definition ops_chain.h:89
Helper struct for representing a list of operations.
Definition ops.h:297
uint8_t pixels
Definition uops.h:221
uint8_t write_size
Definition uops.h:216
uint8_t read_size
Definition uops.h:215
SwsUOp * ops
Definition uops.h:329
int pixel_size_max
Definition uops.h:335
int num_ops
Definition uops.h:330
SwsCompMask planes_in
Definition uops.h:333
SwsCompMask planes_out
Definition uops.h:334
Copyright (C) 2025 Niklas Haas.
Definition ops_chain.h:146
Definition uops.h:292
SwsPixel scalar
Definition uops.h:303
SwsCompMask mask
Definition uops.h:296
SwsUOpType uop
Definition uops.h:295
SwsUOpParams par
Definition uops.h:297
SwsFilterWeights * kernel
Definition uops.h:301
union SwsUOp::@242237116251216327057105100216205033300341206345 data
SwsPixelType type
Definition uops.h:294
SwsShuffleMask shuffle
Definition uops.h:306
SwsPixel mat4x5[4][5]
Definition uops.h:305
SwsPixel vec4[4]
Definition uops.h:304
#define av_free(p)
#define av_log(a,...)
void(* filter)(uint8_t *src, ptrdiff_t stride, int qscale)
Definition h263dsp.c:29
static const uint8_t *const tables[]
float f32
Definition uops.h:85
uint32_t u32
Definition uops.h:84
uint8_t u8
Definition uops.h:82
uint16_t u16
Definition uops.h:83
SwsShuffleUOp shuffle
Definition uops.h:281
int ff_sws_ops_translate(SwsContext *ctx, const SwsOpList *ops, SwsUOpFlags flags, SwsUOpList *uops)
Translate a list of operations down to micro-ops, which can be further optimized and then directly ex...
Definition uops.c:618
void ff_sws_uop_name(const SwsUOp *op, char buf[SWS_UOP_NAME_MAX])
Definition uops.c:52
SwsUOpList * ff_sws_uop_list_alloc(void)
Definition uops.c:174
void ff_sws_uop_list_free(SwsUOpList **p_ops)
Definition uops.c:160
uint32_t SwsUOpFlags
Definition uops.h:147
SwsPixelType
Definition uops.h:40
@ SWS_PIXEL_F32
Definition uops.h:45
@ SWS_PIXEL_U32
Definition uops.h:44
@ SWS_PIXEL_U16
Definition uops.h:43
@ SWS_PIXEL_U8
Definition uops.h:42
#define SWS_COMP(X)
Definition uops.h:122
#define SWS_COMP_ELEMS(N)
Definition uops.h:125
SwsUOpType
Definition uops.h:157
@ SWS_UOP_READ_PLANAR
Definition uops.h:161
@ SWS_UOP_WRITE_PLANAR
Definition uops.h:170
@ SWS_UOP_WRITE_PACKED
Definition uops.h:171
@ SWS_UOP_READ_PACKED
Definition uops.h:165
@ SWS_UOP_CLEAR
Definition uops.h:203
@ SWS_UOP_RW_SHUFFLE
Definition uops.h:176
@ SWS_UOP_FLAG_FMA
Definition uops.h:150
@ SWS_UOP_FLAG_PSHUFB
Definition uops.h:151
@ SWS_UOP_FLAG_EXPAND_BIT
Definition uops.h:152
@ SWS_UOP_FLAG_ADD
Definition uops.h:154
@ SWS_UOP_FLAG_READ_PALETTE
Definition uops.h:153
#define SWS_UOP_NAME_MAX
Generate a unique name for a SwsUOp.
Definition uops.h:325
static av_const int ff_sws_pixel_type_size(SwsPixelType type)
Definition uops.h:51
#define SWS_FOR(TYPE, UOP, MACRO,...)
Definition uops_macros.h:17
#define SWS_FOR_STRUCT(TYPE, UOP, MACRO,...)
Definition uops_macros.h:19
#define DECL_ENTRY(SETUP, NAME,...)
Definition uops_tmpl.h:144
static int setup_clear(const SwsImplParams *params, SwsImplResult *out)
Definition ops.c:265
static int movsize(const int bytes, const int mmsize)
Definition ops.c:550
#define ASSIGN_SHUFFLE_FUNC(CPU, EXT, NAME,...)
static int hscale_sizeof_weight(const SwsUOp *uop)
Definition ops.c:66
static int setup_filter_h_4x4(const SwsImplParams *params, SwsImplResult *out)
Definition ops.c:188
static bool uop_is_type_invariant(const SwsUOpType uop)
Definition ops.c:328
static int compile_uops_x86(SwsContext *ctx, const SwsUOpList *uops, SwsCompiledOp *out)
Definition ops.c:643
static uint32_t expand32(const SwsPixelType type, const SwsPixel value)
Definition ops.c:626
static void splat_lane(void *dst, SwsPixelType type, SwsPixel px)
Definition ops.c:280
static int compile_x86(SwsContext *ctx, const SwsOpList *ops, SwsCompiledOp *out)
Definition ops.c:719
#define DECL_TABLE_U8(EXT, SIZE, FLAG)
Definition ops.c:401
#define DECL_TABLE_U16(EXT, SIZE, FLAG)
Definition ops.c:419
static int setup_linear(const SwsImplParams *params, SwsImplResult *out)
Definition ops.c:297
static int setup_scale(const SwsImplParams *params, SwsImplResult *out)
Definition ops.c:251
static int setup_filter_v(const SwsImplParams *params, SwsImplResult *out)
Definition ops.c:46
#define DECL_TABLE_U32(EXT, SIZE, FLAG)
Definition ops.c:437
static void normalize_clear(SwsUOp *uop)
Definition ops.c:637
static bool check_filter_h_4x4(const SwsImplParams *params)
Definition ops.c:164
static int setup_dither(const SwsImplParams *params, SwsImplResult *out)
Definition ops.c:273
static int translate_shuffle(const SwsUOp *uop, int mmsize, SwsCompiledOp *out)
Definition ops.c:559
static int setup_filter_h(const SwsImplParams *params, SwsImplResult *out)
Definition ops.c:76
#define DECL_TABLE_F32(EXT, SIZE, FLAG)
Definition ops.c:459
static int setup_rw_packed(const SwsImplParams *params, SwsImplResult *out)
Copyright (C) 2025-2026 Niklas Haas.
Definition ops.c:31
static int get_mmsize(void)
Definition ops.c:537