Commit 05ecb86e31 for ffmpeg
commit 05ecb86e31e89ad82ca084793971680cf2ef58b2
Author: Ramiro Polla <ramiro.polla@gmail.com>
Date: Tue Sep 29 02:51:47 2026 +0200
swscale/aarch64/ops: port to uops
Translate the SwsOpList to uops with ff_sws_ops_translate() and build
the aarch64 implementation parameters directly from the resulting
SwsUOps, instead of mapping each SwsOp to an uop in the backend.
This removes the SwsOp-specific logic from ops_impl_conv.c (read/write
mode selection, swizzle to move decomposition, clear/linear/dither
parameter extraction), and uses the generic ff_sws_setup_vec4() and
ff_sws_setup_scalar() helpers for clear, min, max and scale. The dither
matrix, including the extra rows for over-reading with y offsets, is
now prepared by the uops layer, so the backend only takes a reference
to it.
The backend also implements compile_uops, so it is now covered by the
sw_ops checkasm test.
One pair of copy entries in ops_entries.c changes because the uops
layer decomposes that swizzle into a different (and better) sequence
of moves.
Sponsored-by: Sovereign Tech Fund
Signed-off-by: Ramiro Polla <ramiro.polla@gmail.com>
diff --git a/libswscale/aarch64/ops.c b/libswscale/aarch64/ops.c
index 96058a282f..b02312b4b7 100644
--- a/libswscale/aarch64/ops.c
+++ b/libswscale/aarch64/ops.c
@@ -66,7 +66,7 @@ static SwsFuncPtr aarch64_lookup(const SwsAArch64OpImplParams *p)
/*********************************************************************/
static int aarch64_setup_linear(const SwsAArch64OpImplParams *p,
- const SwsOp *op, SwsImplResult *res)
+ const SwsUOp *uop, SwsImplResult *res)
{
/**
* Compute number of full vector registers needed to pack all non-zero
@@ -87,7 +87,7 @@ static int aarch64_setup_linear(const SwsAArch64OpImplParams *p,
for (int j = 0; j < 5; j++) {
const int jj = (j == 0) ? 4 : (j - 1);
if (!(p->par.lin.zero & SWS_MASK(i, jj)))
- coeffs[i_coeff++] = (float) op->lin.m[i][jj].num / op->lin.m[i][jj].den;
+ coeffs[i_coeff++] = uop->data.mat4[i][jj].f32;
}
}
@@ -98,103 +98,54 @@ static int aarch64_setup_linear(const SwsAArch64OpImplParams *p,
}
/*********************************************************************/
-static int aarch64_setup_dither(const SwsAArch64OpImplParams *p,
- const SwsOp *op, SwsImplResult *res)
+static int aarch64_setup_dither(const SwsUOp *uop, SwsImplResult *res)
{
- /**
- * The input dither matrix is (1 << size_log2)² pixels large. It is
- * periodic, so the x and y offsets should be masked to fit inside
- * (1 << size_log2).
- * The width of the matrix is assumed to be at least 8, which matches
- * the maximum block_size for aarch64 asmgen when f32 operations
- * (i.e., dithering) are used. This guarantees that the x offset is
- * aligned and that reading block_size elements does not extend past
- * the end of the row. The x offset doesn't change between components,
- * so it is only required to be masked once.
- * The y offset, on the other hand, may change per component, and
- * would therefore need to be masked for every y_offset value. To
- * simplify the execution, we over-allocate the number of rows of
- * the output dither matrix by the largest y_offset value. This way,
- * we only need to mask y offset once, and can safely increment the
- * dither matrix pointer by fixed offsets for every y_offset change.
- */
-
- /* Find the largest y_offset value. */
- const int size = 1 << op->dither.size_log2;
- const int8_t *off = op->dither.y_offset;
- int max_offset = 0;
- for (int i = 0; i < 4; i++) {
- if (off[i] >= 0)
- max_offset = FFMAX(max_offset, off[i] & (size - 1));
- }
-
- /* Allocate (size + max_offset) rows to allow over-reading the matrix. */
- const int stride = size * sizeof(float);
- const int num_rows = size + max_offset;
- float *matrix = av_malloc(num_rows * stride);
- if (!matrix)
- return AVERROR(ENOMEM);
-
- for (int i = 0; i < size * size; i++)
- matrix[i] = (float) op->dither.matrix[i].num / op->dither.matrix[i].den;
-
- memcpy(&matrix[size * size], matrix, max_offset * stride);
-
- res->priv.ptr = matrix;
- res->free = ff_op_priv_free;
-
+ res->priv.ptr = av_refstruct_ref(uop->data.ptr);
+ res->free = ff_op_priv_unref;
return 0;
}
/*********************************************************************/
-static int aarch64_setup(const SwsOpList *ops, int block_size, int n,
- const SwsAArch64OpImplParams *p, SwsImplResult *out)
+static int aarch64_setup(const SwsUOp *uop, const SwsAArch64OpImplParams *p,
+ SwsImplResult *out)
{
- const SwsOp *op = &ops->ops[n];
- switch (op->op) {
- case SWS_OP_READ:
+ switch (uop->uop) {
+ case SWS_UOP_READ_BIT:
/* Negative shift values to perform right shift using ushl. */
- if (op->rw.frac == 3) {
- out->priv = (SwsOpPriv) {
- .u8 = {
- -7, -6, -5, -4, -3, -2, -1, 0,
- -7, -6, -5, -4, -3, -2, -1, 0,
- }
- };
- }
+ out->priv = (SwsOpPriv) {
+ .u8 = {
+ -7, -6, -5, -4, -3, -2, -1, 0,
+ -7, -6, -5, -4, -3, -2, -1, 0,
+ }
+ };
break;
- case SWS_OP_WRITE:
+ case SWS_UOP_WRITE_BIT:
/* Shift values for ushl. */
- if (op->rw.frac == 3) {
- out->priv = (SwsOpPriv) {
- .u8 = {
- 7, 6, 5, 4, 3, 2, 1, 0,
- 7, 6, 5, 4, 3, 2, 1, 0,
- }
- };
- }
+ out->priv = (SwsOpPriv) {
+ .u8 = {
+ 7, 6, 5, 4, 3, 2, 1, 0,
+ 7, 6, 5, 4, 3, 2, 1, 0,
+ }
+ };
break;
- case SWS_OP_CLEAR:
- ff_sws_setup_clear(&(const SwsImplParams) { .op = op }, out);
- break;
- case SWS_OP_MIN:
- case SWS_OP_MAX:
- ff_sws_setup_clamp(&(const SwsImplParams) { .op = op }, out);
- break;
- case SWS_OP_SCALE:
- ff_sws_setup_scale(&(const SwsImplParams) { .op = op }, out);
- break;
- case SWS_OP_LINEAR:
- return aarch64_setup_linear(p, op, out);
- case SWS_OP_DITHER:
- return aarch64_setup_dither(p, op, out);
+ case SWS_UOP_CLEAR:
+ case SWS_UOP_MIN:
+ case SWS_UOP_MAX:
+ return ff_sws_setup_vec4(&(const SwsImplParams) { .uop = uop }, out);
+ case SWS_UOP_SCALE:
+ return ff_sws_setup_scalar(&(const SwsImplParams) { .uop = uop }, out);
+ case SWS_UOP_LINEAR:
+ case SWS_UOP_LINEAR_FMA:
+ return aarch64_setup_linear(p, uop, out);
+ case SWS_UOP_DITHER:
+ return aarch64_setup_dither(uop, out);
}
return 0;
}
/*********************************************************************/
-static int aarch64_compile(SwsContext *ctx, const SwsOpList *ops,
- SwsCompiledOp *out)
+static int aarch64_compile_uops(SwsContext *ctx, const SwsUOpList *uops,
+ SwsCompiledOp *out)
{
int ret;
@@ -203,7 +154,7 @@ static int aarch64_compile(SwsContext *ctx, const SwsOpList *ops,
return AVERROR(ENOTSUP);
/* Use at most two full vregs during the widest precision section */
- int block_size = (ff_sws_op_list_max_size(ops) == 4) ? 8 : 16;
+ int block_size = (uops->pixel_size_max == 4) ? 8 : 16;
SwsOpChain *chain = ff_sws_op_chain_alloc();
if (!chain)
@@ -218,18 +169,16 @@ static int aarch64_compile(SwsContext *ctx, const SwsOpList *ops,
};
/* Look up kernel functions. */
- for (int i = 0; i < ops->num_ops; i++) {
+ for (int i = 0; i < uops->num_ops; i++) {
SwsAArch64OpImplParams params = { 0 };
- ret = convert_to_aarch64_impl(ctx, ops, i, block_size, ¶ms);
- if (ret < 0)
- goto error;
+ convert_to_aarch64_impl(&uops->ops[i], block_size, ¶ms);
SwsFuncPtr func = aarch64_lookup(¶ms);
if (!func) {
ret = AVERROR(ENOTSUP);
goto error;
}
SwsImplResult res = { 0 };
- ret = aarch64_setup(ops, block_size, i, ¶ms, &res);
+ ret = aarch64_setup(&uops->ops[i], ¶ms, &res);
if (ret < 0)
goto error;
ret = ff_sws_op_chain_append(chain, func, res.free, &res.priv);
@@ -243,12 +192,8 @@ static int aarch64_compile(SwsContext *ctx, const SwsOpList *ops,
void ff_sws_process_0111_neon(void);
void ff_sws_process_1111_neon(void);
- const SwsOp *read = ff_sws_op_list_input(ops);
- const SwsOp *write = ff_sws_op_list_output(ops);
- const int read_planes = read ? ff_sws_rw_op_planes(read) : 0;
- const int write_planes = ff_sws_rw_op_planes(write);
SwsOpFunc process_func = NULL;
- switch (FFMAX(read_planes, write_planes)) {
+ switch (av_popcount(uops->planes_in | uops->planes_out)) {
case 1: process_func = (SwsOpFunc) ff_sws_process_0001_neon; break;
case 2: process_func = (SwsOpFunc) ff_sws_process_0011_neon; break;
case 3: process_func = (SwsOpFunc) ff_sws_process_0111_neon; break;
@@ -258,16 +203,38 @@ static int aarch64_compile(SwsContext *ctx, const SwsOpList *ops,
out->func = process_func;
out->cpu_flags = chain->cpu_flags;
+ return 0;
+
error:
+ ff_sws_op_chain_free(chain);
+ return ret;
+}
+
+/*********************************************************************/
+static int aarch64_compile(SwsContext *ctx, const SwsOpList *ops,
+ SwsCompiledOp *out)
+{
+ SwsUOpList *uops = ff_sws_uop_list_alloc();
+ if (!uops)
+ return AVERROR(ENOMEM);
+
+ const SwsUOpFlags flags = (ctx->flags & SWS_BITEXACT) ? 0 : SWS_UOP_FLAG_FMA;
+ int ret = ff_sws_ops_translate(ctx, ops, flags, uops);
if (ret < 0)
- ff_sws_op_chain_free(chain);
+ goto error;
+
+ ret = aarch64_compile_uops(ctx, uops, out);
+
+error:
+ ff_sws_uop_list_free(&uops);
return ret;
}
/*********************************************************************/
const SwsOpBackend backend_aarch64 = {
- .name = "aarch64",
- .flags = SWS_BACKEND_AARCH64,
- .compile = aarch64_compile,
- .hw_format = AV_PIX_FMT_NONE,
+ .name = "aarch64",
+ .flags = SWS_BACKEND_AARCH64,
+ .compile = aarch64_compile,
+ .compile_uops = aarch64_compile_uops,
+ .hw_format = AV_PIX_FMT_NONE,
};
diff --git a/libswscale/aarch64/ops_asmgen.c b/libswscale/aarch64/ops_asmgen.c
index 6068fc327a..f83e487bc7 100644
--- a/libswscale/aarch64/ops_asmgen.c
+++ b/libswscale/aarch64/ops_asmgen.c
@@ -938,8 +938,23 @@ static void asmgen_op_dither(SwsAArch64Context *s, const SwsAArch64OpImplParams
RasmOp y64 = a64op_x(s->y);
/**
- * For a description of the matrix buffer layout, read the comments
- * in aarch64_setup_dither() in aarch64/ops.c.
+ * The dither matrix is (1 << size_log2)² pixels large. It is
+ * periodic, so the x and y offsets should be masked to fit inside
+ * (1 << size_log2). The matrix buffer is prepared by
+ * translate_dither_op() in libswscale/uops.c.
+ * The width of the matrix is assumed to be at least 8, which matches
+ * the maximum block_size for aarch64 asmgen when f32 operations
+ * (i.e., dithering) are used. This guarantees that the x offset is
+ * aligned and that reading block_size elements does not extend past
+ * the end of the row. The x offset doesn't change between components,
+ * so it is only required to be masked once.
+ * The y offset, on the other hand, may change per component, and
+ * would therefore need to be masked for every y_offset value. To
+ * avoid this, the matrix buffer is over-allocated by the largest
+ * y_offset value, with the extra rows repeating the first rows of
+ * the matrix. This way, we only need to mask the y offset once, and
+ * can safely increment the dither matrix pointer by fixed offsets
+ * for every y_offset change.
*/
/**
diff --git a/libswscale/aarch64/ops_entries.c b/libswscale/aarch64/ops_entries.c
index 6bb9db1613..cd42184715 100644
--- a/libswscale/aarch64/ops_entries.c
+++ b/libswscale/aarch64/ops_entries.c
@@ -159,8 +159,8 @@ ENTRY(ff_sws_copy_000000103120_32_u8_1110_neon, { .uop = SWS_UOP_COPY,
ENTRY(ff_sws_copy_000000302010_8_u8_1110_neon, { .uop = SWS_UOP_COPY, .block_size = 8, .type = SWS_PIXEL_U8, .mask = 0xe, .par.move = { .num_moves = 3, .dst = {1, 2, 3, 0, 0, 0}, .src = {0, 0, 0, 0, 0, 0} } })
ENTRY(ff_sws_copy_000000302010_16_u8_1110_neon, { .uop = SWS_UOP_COPY, .block_size = 16, .type = SWS_PIXEL_U8, .mask = 0xe, .par.move = { .num_moves = 3, .dst = {1, 2, 3, 0, 0, 0}, .src = {0, 0, 0, 0, 0, 0} } })
ENTRY(ff_sws_copy_000000302010_32_u8_1110_neon, { .uop = SWS_UOP_COPY, .block_size = 32, .type = SWS_PIXEL_U8, .mask = 0xe, .par.move = { .num_moves = 3, .dst = {1, 2, 3, 0, 0, 0}, .src = {0, 0, 0, 0, 0, 0} } })
-ENTRY(ff_sws_copy_001f01f03020_8_u8_1111_neon, { .uop = SWS_UOP_COPY, .block_size = 8, .type = SWS_PIXEL_U8, .mask = 0xf, .par.move = { .num_moves = 5, .dst = {2, 3, -1, 0, 1, 0}, .src = {0, 0, 0, 1, -1, 0} } })
-ENTRY(ff_sws_copy_001f01f03020_16_u8_1111_neon, { .uop = SWS_UOP_COPY, .block_size = 16, .type = SWS_PIXEL_U8, .mask = 0xf, .par.move = { .num_moves = 5, .dst = {2, 3, -1, 0, 1, 0}, .src = {0, 0, 0, 1, -1, 0} } })
+ENTRY(ff_sws_copy_000032120120_8_u8_1111_neon, { .uop = SWS_UOP_COPY, .block_size = 8, .type = SWS_PIXEL_U8, .mask = 0xf, .par.move = { .num_moves = 4, .dst = {2, 0, 1, 3, 0, 0}, .src = {0, 1, 2, 2, 0, 0} } })
+ENTRY(ff_sws_copy_000032120120_16_u8_1111_neon, { .uop = SWS_UOP_COPY, .block_size = 16, .type = SWS_PIXEL_U8, .mask = 0xf, .par.move = { .num_moves = 4, .dst = {2, 0, 1, 3, 0, 0}, .src = {0, 1, 2, 2, 0, 0} } })
ENTRY(ff_sws_swap_bytes_8_u16_0001_neon, { .uop = SWS_UOP_SWAP_BYTES, .block_size = 8, .type = SWS_PIXEL_U16, .mask = 0x1 })
ENTRY(ff_sws_swap_bytes_8_u16_0010_neon, { .uop = SWS_UOP_SWAP_BYTES, .block_size = 8, .type = SWS_PIXEL_U16, .mask = 0x2 })
ENTRY(ff_sws_swap_bytes_8_u16_0011_neon, { .uop = SWS_UOP_SWAP_BYTES, .block_size = 8, .type = SWS_PIXEL_U16, .mask = 0x3 })
diff --git a/libswscale/aarch64/ops_impl.h b/libswscale/aarch64/ops_impl.h
index 304b68e44c..560021e4ce 100644
--- a/libswscale/aarch64/ops_impl.h
+++ b/libswscale/aarch64/ops_impl.h
@@ -41,7 +41,7 @@ static inline uint16_t nibble_mask(SwsCompMask mask)
/**
* SwsAArch64OpImplParams describes the parameters for an SwsUOpType
- * operation. It consists of simplified parameters from the SwsOp structure,
+ * operation. It consists of simplified parameters from the SwsUOp structure,
* with the purpose of being straight-forward to implement and execute.
*/
typedef struct SwsAArch64OpImplParams {
diff --git a/libswscale/aarch64/ops_impl_conv.c b/libswscale/aarch64/ops_impl_conv.c
index 9869c29608..4f8e6e3437 100644
--- a/libswscale/aarch64/ops_impl_conv.c
+++ b/libswscale/aarch64/ops_impl_conv.c
@@ -24,297 +24,47 @@
*/
#include "libavutil/error.h"
-#include "libavutil/rational.h"
-#include "libswscale/ops.h"
#include "ops_impl.h"
-static void swizzle_emit(SwsAArch64OpImplParams *out, uint8_t dst, uint8_t src)
-{
- int idx = out->par.move.num_moves++;
- out->par.move.dst[idx] = dst;
- out->par.move.src[idx] = src;
-}
-
-static void convert_swizzle_to_moves(const SwsOp *op, SwsAArch64OpImplParams *out)
-{
- SwsSwizzleOp swizzle = {
- .in = {
- op->swizzle.in[0],
- op->swizzle.in[1],
- op->swizzle.in[2],
- op->swizzle.in[3],
- }
- };
-
- /* Compute used vectors (src and dst) */
- uint8_t src_used[4] = { 0 };
- bool done[4] = { true, true, true, true };
- LOOP(out->mask, dst) {
- uint8_t src = swizzle.in[dst];
- src_used[src]++;
- done[dst] = false;
- }
-
- /* First perform unobstructed copies. */
- for (bool progress = true; progress; ) {
- progress = false;
- for (int dst = 0; dst < 4; dst++) {
- if (done[dst] || src_used[dst])
- continue;
- uint8_t src = swizzle.in[dst];
- swizzle_emit(out, dst, src);
- src_used[src]--;
- done[dst] = true;
- progress = true;
- }
- }
-
- /* Then swap and rotate remaining operations. */
- for (int dst = 0; dst < 4; dst++) {
- if (done[dst])
- continue;
-
- swizzle_emit(out, -1, dst);
-
- uint8_t cur_dst = dst;
- uint8_t src = swizzle.in[cur_dst];
- while (src != dst) {
- swizzle_emit(out, cur_dst, src);
- done[cur_dst] = true;
- cur_dst = src;
- src = swizzle.in[cur_dst];
- }
-
- swizzle_emit(out, cur_dst, -1);
- done[cur_dst] = true;
- }
-}
-
/**
- * Convert SwsOp to a SwsAArch64OpImplParams. Read the comments regarding
+ * Convert SwsUOp to a SwsAArch64OpImplParams. Read the comments regarding
* SwsAArch64OpImplParams in ops_impl.h for more information.
*/
-static int convert_to_aarch64_impl(SwsContext *ctx, const SwsOpList *ops, int n,
- int block_size, SwsAArch64OpImplParams *out)
+static void convert_to_aarch64_impl(const SwsUOp *uop, int block_size,
+ SwsAArch64OpImplParams *out)
{
- const SwsOp *op = &ops->ops[n];
-
+ out->uop = uop->uop;
+ out->mask = uop->mask;
+ out->type = uop->type;
out->block_size = block_size;
+ out->par = uop->par;
/**
- * Most SwsOp work on fields described by SWS_OP_NEEDED().
- * The few that don't will override this field later.
+ * Deduplicate params to prevent identical CPS functions from being
+ * instantiated multiple times under different names.
*/
- out->mask = 0;
- for (int i = 0; i < 4; i++) {
- if (SWS_OP_NEEDED(op, i))
- out->mask |= SWS_COMP(i);
- }
-
- out->type = op->type;
-
- /* Map SwsOpType to SwsUOpType */
- switch (op->op) {
- case SWS_OP_READ:
- if (op->rw.filter.op)
- return AVERROR(ENOTSUP);
- /**
- * The different types of read operations have been split into
- * their own SwsUOpType to simplify the implementation.
- */
- if (op->rw.frac == 1)
- out->uop = SWS_UOP_READ_NIBBLE;
- else if (op->rw.frac == 3)
- out->uop = SWS_UOP_READ_BIT;
- else if (op->rw.mode == SWS_RW_PACKED && op->rw.elems > 1)
- out->uop = SWS_UOP_READ_PACKED;
- else if (op->rw.mode == SWS_RW_PACKED || op->rw.mode == SWS_RW_PLANAR)
- out->uop = SWS_UOP_READ_PLANAR;
- else
- return AVERROR(ENOTSUP);
- break;
- case SWS_OP_WRITE:
- if (op->rw.filter.op)
- return AVERROR(ENOTSUP);
- /**
- * The different types of write operations have been split into
- * their own SwsUOpType to simplify the implementation.
- */
- if (op->rw.frac == 1)
- out->uop = SWS_UOP_WRITE_NIBBLE;
- else if (op->rw.frac == 3)
- out->uop = SWS_UOP_WRITE_BIT;
- else if (op->rw.mode == SWS_RW_PACKED && op->rw.elems > 1)
- out->uop = SWS_UOP_WRITE_PACKED;
- else if (op->rw.mode == SWS_RW_PACKED || op->rw.mode == SWS_RW_PLANAR)
- out->uop = SWS_UOP_WRITE_PLANAR;
- else
- return AVERROR(ENOTSUP);
- break;
- case SWS_OP_SWAP_BYTES: out->uop = SWS_UOP_SWAP_BYTES; break;
- case SWS_OP_SWIZZLE: {
- /**
- * Detect whether copies are needed or if a simple permute is
- * enough.
- */
- out->uop = SWS_UOP_PERMUTE;
- SwsCompMask seen = 0;
- LOOP(out->mask, i) {
- uint8_t src = op->swizzle.in[i];
- if (seen & SWS_COMP(src)) {
- out->uop = SWS_UOP_COPY;
- break;
- }
- seen |= SWS_COMP(src);
- }
- break;
- }
- case SWS_OP_UNPACK: out->uop = SWS_UOP_UNPACK; break;
- case SWS_OP_PACK: out->uop = SWS_UOP_PACK; break;
- case SWS_OP_LSHIFT: out->uop = SWS_UOP_LSHIFT; break;
- case SWS_OP_RSHIFT: out->uop = SWS_UOP_RSHIFT; break;
- case SWS_OP_CLEAR: out->uop = SWS_UOP_CLEAR; break;
- case SWS_OP_CONVERT:
- if (op->convert.expand) {
- switch (op->convert.to) {
- case SWS_PIXEL_U16: out->uop = SWS_UOP_EXPAND_PAIR; break;
- case SWS_PIXEL_U32: out->uop = SWS_UOP_EXPAND_QUAD; break;
- }
- } else {
- switch (op->convert.to) {
- case SWS_PIXEL_U8: out->uop = SWS_UOP_TO_U8; break;
- case SWS_PIXEL_U16: out->uop = SWS_UOP_TO_U16; break;
- case SWS_PIXEL_U32: out->uop = SWS_UOP_TO_U32; break;
- case SWS_PIXEL_F32: out->uop = SWS_UOP_TO_F32; break;
- }
- }
- break;
- case SWS_OP_MIN:
- case SWS_OP_MAX:
- out->uop = (op->op == SWS_OP_MIN) ? SWS_UOP_MIN : SWS_UOP_MAX;
- out->mask &= ff_sws_comp_mask_q4(op->clamp.limit);
- break;
- case SWS_OP_SCALE: out->uop = SWS_UOP_SCALE; break;
- case SWS_OP_LINEAR:
- out->uop = (ctx->flags & SWS_BITEXACT)
- ? SWS_UOP_LINEAR
- : SWS_UOP_LINEAR_FMA;
- break;
- case SWS_OP_DITHER: out->uop = SWS_UOP_DITHER; break;
- default:
- return AVERROR(ENOTSUP);
- }
-
switch (out->uop) {
- case SWS_UOP_READ_BIT:
- case SWS_UOP_READ_NIBBLE:
- case SWS_UOP_READ_PACKED:
- case SWS_UOP_READ_PLANAR:
- case SWS_UOP_WRITE_BIT:
- case SWS_UOP_WRITE_NIBBLE:
- case SWS_UOP_WRITE_PACKED:
- case SWS_UOP_WRITE_PLANAR:
- switch (op->rw.elems) {
- case 1: out->mask = SWS_COMP_ELEMS(1); break;
- case 2: out->mask = SWS_COMP_ELEMS(2); break;
- case 3: out->mask = SWS_COMP_ELEMS(3); break;
- case 4: out->mask = SWS_COMP_ELEMS(4); break;
- };
- break;
case SWS_UOP_PERMUTE:
- case SWS_UOP_COPY:
+ case SWS_UOP_COPY: {
/* Recompute mask taking identity swizzle into account */
out->mask = 0;
- for (int i = 0; i < 4; i++) {
- if (SWS_OP_NEEDED(op, i) && op->swizzle.in[i] != i)
- out->mask |= SWS_COMP(i);
+ for (int i = 0; i < out->par.move.num_moves; i++) {
+ int dst = out->par.move.dst[i];
+ if (dst >= 0)
+ out->mask |= SWS_COMP(dst);
}
- convert_swizzle_to_moves(op, out);
+
/* The element size and type don't matter. */
- out->block_size = block_size * ff_sws_pixel_type_size(op->type);
+ out->block_size = block_size * ff_sws_pixel_type_size(out->type);
out->type = SWS_PIXEL_U8;
break;
- case SWS_UOP_UNPACK:
- for (int i = 0; i < 4; i++)
- out->par.pack.pattern[i] = op->pack.pattern[i];
- break;
- case SWS_UOP_PACK:
- out->mask = 0;
- for (int i = 0; i < 4 && op->pack.pattern[i]; i++)
- out->mask |= SWS_COMP(i);
- for (int i = 0; i < 4; i++)
- out->par.pack.pattern[i] = op->pack.pattern[i];
- break;
- case SWS_UOP_LSHIFT:
- case SWS_UOP_RSHIFT:
- out->par.shift.amount = op->shift.amount;
- break;
- case SWS_UOP_CLEAR:
- out->mask = 0;
- for (int i = 0; i < 4; i++) {
- if (op->clear.mask & SWS_COMP(i)) {
- out->mask |= SWS_COMP(i);
- if (op->clear.value[i].num == 0) {
- out->par.clear.zero |= SWS_COMP(i);
- } else {
- uint32_t val = op->clear.value[i].num / op->clear.value[i].den;
- if ((op->type == SWS_PIXEL_U8 && val == UINT8_MAX) ||
- (op->type == SWS_PIXEL_U16 && val == UINT16_MAX) ||
- (op->type == SWS_PIXEL_U32 && val == UINT32_MAX))
- out->par.clear.one |= SWS_COMP(i);
- }
- }
- }
- break;
- case SWS_UOP_LINEAR:
+ }
case SWS_UOP_LINEAR_FMA:
- out->mask = 0;
- const uint32_t lin_mask = ff_sws_linear_mask(&op->lin);
- for (int i = 0; i < 4; i++) {
- if (!SWS_OP_NEEDED(op, i) || !(lin_mask & SWS_MASK_ROW(i))) {
- for (int j = 0; j < 5; j++)
- out->par.lin.zero |= SWS_MASK(i, j);
- continue;
- }
- out->mask |= SWS_COMP(i);
- for (int j = 0; j < 5; j++) {
- const AVRational64 k = op->lin.m[i][j];
- if (j < 4 && k.num == k.den)
- out->par.lin.one |= SWS_MASK(i, j);
- else if (k.num == 0)
- out->par.lin.zero |= SWS_MASK(i, j);
- }
- }
+ /* par.lin.exact is currently unused by asmgen_op_linear(). */
+ out->par.lin.exact = 0;
break;
- case SWS_UOP_DITHER:
- out->mask = SWS_COMP_MASK(op->dither.y_offset[0] >= 0,
- op->dither.y_offset[1] >= 0,
- op->dither.y_offset[2] >= 0,
- op->dither.y_offset[3] >= 0);
- LOOP(out->mask, i) {
- out->par.dither.y_offset[i] = op->dither.y_offset[i];
- }
- out->par.dither.size_log2 = op->dither.size_log2;
- break;
- }
-
- switch (out->uop) {
- case SWS_UOP_READ_BIT:
- case SWS_UOP_READ_NIBBLE:
- case SWS_UOP_READ_PACKED:
- case SWS_UOP_READ_PLANAR:
- case SWS_UOP_WRITE_BIT:
- case SWS_UOP_WRITE_NIBBLE:
- case SWS_UOP_WRITE_PACKED:
- case SWS_UOP_WRITE_PLANAR:
- case SWS_UOP_SWAP_BYTES:
- case SWS_UOP_CLEAR:
- /* Only the element size matters, not the type. */
- if (out->type == SWS_PIXEL_F32)
- out->type = SWS_PIXEL_U32;
+ default:
break;
}
-
- return 0;
}
diff --git a/libswscale/tests/sws_ops_aarch64.c b/libswscale/tests/sws_ops_aarch64.c
index 2155319a33..6ce1ab67f4 100644
--- a/libswscale/tests/sws_ops_aarch64.c
+++ b/libswscale/tests/sws_ops_aarch64.c
@@ -197,16 +197,25 @@ static int collect_ops_compile(SwsContext *ctx, const SwsOpList *ops,
struct AVTreeNode **root = (struct AVTreeNode **) ctx->opaque;
int ret;
+ SwsUOpList *uops = ff_sws_uop_list_alloc();
+ if (!uops)
+ return AVERROR(ENOMEM);
+
+ const SwsUOpFlags flags = (ctx->flags & SWS_BITEXACT) ? 0 : SWS_UOP_FLAG_FMA;
+ ret = ff_sws_ops_translate(ctx, ops, flags, uops);
+ if (ret == AVERROR(ENOTSUP)) {
+ ret = 0;
+ goto end;
+ }
+ if (ret < 0)
+ goto end;
+
/* Use at most two full vregs during the widest precision section */
- int block_size = (ff_sws_op_list_max_size(ops) == 4) ? 8 : 16;
+ int block_size = (uops->pixel_size_max == 4) ? 8 : 16;
- for (int i = 0; i < ops->num_ops; i++) {
+ for (int i = 0; i < uops->num_ops; i++) {
SwsAArch64OpImplParams params = { 0 };
- ret = convert_to_aarch64_impl(ctx, ops, i, block_size, ¶ms);
- if (ret == AVERROR(ENOTSUP))
- continue;
- if (ret < 0)
- goto end;
+ convert_to_aarch64_impl(&uops->ops[i], block_size, ¶ms);
ret = aarch64_collect_op(¶ms, root);
if (ret < 0)
goto end;
@@ -226,6 +235,7 @@ static int collect_ops_compile(SwsContext *ctx, const SwsOpList *ops,
ret = 0;
end:
+ ff_sws_uop_list_free(&uops);
return ret;
}