Commit 706660a8a5 for ffmpeg
commit 706660a8a5db07f0816e142b4e89dbcd64a9b70f
Author: Niklas Haas <git@haasn.dev>
Date: Fri Aug 14 10:45:50 2026 +0200
swscale/ops_optimizer: move convert->expand promotion to uops layer
This changes a lot of the op lists, but without affecting the translated
micro-ops. Slightly modifies the Vulkan shader, but the SPIR-V compiler
should be smart enough to optimize that multiplication back into a bit
operation.
rgb24 16x16 -> rgb48le 16x16:
[ u8 +++X] SWS_OP_READ : 3 elem(s) packed >> 0
min: {0 0 0 _}, max: {255 255 255 _}
- [ u8 +++X] SWS_OP_CONVERT : u8 -> u16 (expand)
+ [ u8 +++X] SWS_OP_CONVERT : u8 -> u16
+ min: {0 0 0 _}, max: {255 255 255 _}
+ [u16 +++X] SWS_OP_SCALE : * 257
min: {0 0 0 _}, max: {65535 65535 65535 _}
[u16 XXXX] SWS_OP_WRITE : 3 elem(s) packed >> 0
(X = unused, z = byteswapped, + = exact, 0 = zero)
translated micro-ops:
u8_read_packed_xyz
u8_expand_pair_xyz
u16_write_packed_xyz
Sponsored-by: Sovereign Tech Fund
Signed-off-by: Niklas Haas <git@haasn.dev>
Signed-off-by: Ramiro Polla <ramiro.polla@gmail.com>
diff --git a/libswscale/ops_optimizer.c b/libswscale/ops_optimizer.c
index 3945de2034..ad2c151024 100644
--- a/libswscale/ops_optimizer.c
+++ b/libswscale/ops_optimizer.c
@@ -563,18 +563,6 @@ retry:
ff_sws_op_list_remove_at(ops, n + 1, 1);
goto retry;
}
-
- /* Conversion followed by integer expansion */
- if (next->op == SWS_OP_SCALE && !op->convert.expand &&
- ff_sws_pixel_type_is_int(op->type) &&
- ff_sws_pixel_type_is_int(op->convert.to) &&
- !ff_cmp_q64(next->scale.factor,
- ff_sws_pixel_expand(op->type, op->convert.to)))
- {
- op->convert.expand = true;
- ff_sws_op_list_remove_at(ops, n + 1, 1);
- goto retry;
- }
break;
case SWS_OP_MIN:
@@ -1025,21 +1013,21 @@ static int solve_shuffle(const SwsUOpList *const uops, SwsUOp *out)
return AVERROR(EINVAL);
}
-int ff_sws_uop_list_optimize(SwsContext *ctx, SwsUOpFlags flags, SwsUOpList *uops)
+static int is_integer_scale(const SwsUOp *op, int64_t val)
{
- /* Try promoting the entire uop list to a packed shuffle operation */
- if (flags & SWS_UOP_FLAG_PSHUFB) {
- SwsUOp shuffle;
- int ret = solve_shuffle(uops, &shuffle);
- if (ret >= 0) {
- ff_sws_uop_list_remove_at(uops, 0, uops->num_ops);
- return ff_sws_uop_list_append(uops, &shuffle);
- } else if (ret < 0 && ret != AVERROR(ENOTSUP)) {
- return ret;
- }
+ if (op->uop != SWS_UOP_SCALE)
+ return false;
+
+ switch (op->type) {
+ case SWS_PIXEL_U8: return op->data.scalar.u8 == val;
+ case SWS_PIXEL_U16: return op->data.scalar.u16 == val;
+ case SWS_PIXEL_U32: return op->data.scalar.u32 == val;
+ default: return false;
}
+}
-#if 0
+int ff_sws_uop_list_optimize(SwsContext *ctx, SwsUOpFlags flags, SwsUOpList *uops)
+{
static const SwsUOp dummy = {0};
retry:
@@ -1048,10 +1036,35 @@ retry:
SwsUOp *op = &uops->ops[i];
switch (op->uop) {
- /* placeholder */
+ case SWS_UOP_TO_U16:
+ if (is_integer_scale(next, 0x101)) {
+ op->uop = SWS_UOP_EXPAND_PAIR;
+ ff_sws_uop_list_remove_at(uops, i + 1, 1);
+ goto retry;
+ }
+ break;
+
+ case SWS_UOP_TO_U32:
+ if (is_integer_scale(next, 0x1010101)) {
+ op->uop = SWS_UOP_EXPAND_QUAD;
+ ff_sws_uop_list_remove_at(uops, i + 1, 1);
+ goto retry;
+ }
+ break;
+ }
+ }
+
+ /* Try promoting the entire uop list to a packed shuffle operation */
+ if (flags & SWS_UOP_FLAG_PSHUFB) {
+ SwsUOp shuffle;
+ int ret = solve_shuffle(uops, &shuffle);
+ if (ret >= 0) {
+ ff_sws_uop_list_remove_at(uops, 0, uops->num_ops);
+ return ff_sws_uop_list_append(uops, &shuffle);
+ } else if (ret < 0 && ret != AVERROR(ENOTSUP)) {
+ return ret;
}
}
-#endif
return 0;
}
diff --git a/tests/ref/fate/sws-ops-list b/tests/ref/fate/sws-ops-list
index aa7baa23f8..96acb787ca 100644
--- a/tests/ref/fate/sws-ops-list
+++ b/tests/ref/fate/sws-ops-list
@@ -1 +1 @@
-3dec8ca0edbf6a301f53b7782932c98f
+41b97cc2b41ce3afd672ee14ad9d544d