Commit 031c0a531a for ffmpeg
commit 031c0a531a7f21a3b2a454f88e4c5b9b3a6365bb
Author: Lynne <dev@lynne.ee>
Date: Sat Sep 26 16:36:09 2026 +0900
vulkan_ffv1: decode the largest slices first
Several waves share a SIMD and the issue arbiter favours the oldest, so
the slices that take the longest get the priority instead of the luck
of raster order, and the frame no longer waits on a heavy slice that
was started late. The host sorts the slices by their size and the
shader reads its slice index from that order.
Decoding a 6464x4852 16-bit RGB frame with 1024 slices on an RX 6900
XT, with the bitstream in VRAM, goes from 53.1/43.8/43.2 ms to
50.7/40.6/39.1 ms with context model 1/0/2.
diff --git a/libavcodec/vulkan/ffv1_dec.comp.glsl b/libavcodec/vulkan/ffv1_dec.comp.glsl
index c7559e6e15..33eeafb433 100644
--- a/libavcodec/vulkan/ffv1_dec.comp.glsl
+++ b/libavcodec/vulkan/ffv1_dec.comp.glsl
@@ -30,8 +30,8 @@
#include "common.glsl"
#include "ffv1_common.glsl"
-layout (set = 1, binding = 1, scalar) readonly buffer slice_offsets_buf {
- u32vec2 slice_offsets[];
+layout (set = 1, binding = 1, scalar) readonly buffer slice_order_buf {
+ uint32_t slice_order[];
};
layout (set = 1, binding = 2, scalar) writeonly buffer slice_status_buf {
uint32_t slice_status[];
@@ -378,7 +378,7 @@ void decode_slice(in SliceContext sc, uint slice_idx)
w >>= 1;
int bayer_h = sc.slice_dim.y >> 1;
sp.x >>= 1;
- sp.y = int(gl_WorkGroupID.y)*rgb_linecache;
+ sp.y = int(slice_idx / gl_NumWorkGroups.x)*rgb_linecache;
/* c_bits = bps + 1 (the +1 is for is_rgb). For PCM mode, all planes use
* raw bps. For non-PCM, gm uses bps (bps+1 before 4.8, which coded an
* extra bit); gd/b-gm/r-gm use bps+1. */
@@ -389,7 +389,7 @@ void decode_slice(in SliceContext sc, uint slice_idx)
} else
bits = u16vec4(c_bits - 1, c_bits - 1, c_bits - 1, c_bits - 1);
#elif defined(RGB)
- sp.y = int(gl_WorkGroupID.y)*rgb_linecache;
+ sp.y = int(slice_idx / gl_NumWorkGroups.x)*rgb_linecache;
#endif
#ifndef GOLOMB
@@ -471,7 +471,7 @@ void decode_slice(in SliceContext sc, uint slice_idx)
void main(void)
{
- uint slice_idx = gl_WorkGroupID.y*gl_NumWorkGroups.x + gl_WorkGroupID.x;
+ uint slice_idx = slice_order[gl_WorkGroupID.y*gl_NumWorkGroups.x + gl_WorkGroupID.x];
#ifdef GOLOMB
rc = slice_ctx[slice_idx].c;
diff --git a/libavcodec/vulkan_ffv1.c b/libavcodec/vulkan_ffv1.c
index 6f9527f17f..f9e5ce7449 100644
--- a/libavcodec/vulkan_ffv1.c
+++ b/libavcodec/vulkan_ffv1.c
@@ -76,6 +76,7 @@ typedef struct FFv1VulkanDecodePicture {
FFVkBuffer *slice_fltmap_buf;
FFVkBuffer *slice_feedback_buf;
uint32_t *slice_offset;
+ uint32_t slice_size[MAX_SLICES];
int slice_num;
int crc_checked;
@@ -176,7 +177,7 @@ static int vk_ffv1_start_frame(AVCodecContext *avctx,
err = ff_vk_get_pooled_buffer(&ctx->s, &fv->slice_feedback_pool,
&fp->slice_feedback_buf,
VK_BUFFER_USAGE_STORAGE_BUFFER_BIT,
- NULL, 2*(2*f->slice_count*sizeof(uint32_t)),
+ NULL, 5*f->slice_count*sizeof(uint32_t),
VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT |
VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT);
if (err < 0)
@@ -229,6 +230,9 @@ static int vk_ffv1_decode_slice(AVCodecContext *avctx,
FFVkBuffer *slice_offset = fp->slice_feedback_buf;
FFVkBuffer *slices_buf = vp->slices_buf;
+ if (fp->slice_num < MAX_SLICES)
+ fp->slice_size[fp->slice_num] = size;
+
if (slices_buf && slices_buf->host_ref) {
AV_WN32(slice_offset->mapped_mem + (2*fp->slice_num + 0)*sizeof(uint32_t),
data - slices_buf->mapped_mem);
@@ -252,6 +256,12 @@ static int vk_ffv1_decode_slice(AVCodecContext *avctx,
return 0;
}
+static int cmp_slice_order(const void *a, const void *b)
+{
+ uint64_t x = *(const uint64_t *)a, y = *(const uint64_t *)b;
+ return (x < y) - (x > y);
+}
+
static int vk_ffv1_end_frame(AVCodecContext *avctx)
{
int err;
@@ -514,6 +524,16 @@ static int vk_ffv1_end_frame(AVCodecContext *avctx)
nb_img_bar = 0;
nb_buf_bar = 0;
+ /* Decode the largest slices first: they get the oldest waves, which
+ * have issue priority when several waves share a SIMD */
+ uint64_t order[MAX_SLICES];
+ for (int i = 0; i < f->slice_count; i++)
+ order[i] = (uint64_t)(i < fp->slice_num ? fp->slice_size[i] : 0) << 32 | (UINT32_MAX - i);
+ qsort(order, f->slice_count, sizeof(*order), cmp_slice_order);
+ for (int i = 0; i < f->slice_count; i++)
+ AV_WN32(slice_feedback->mapped_mem + (4*f->slice_count + i)*sizeof(uint32_t),
+ UINT32_MAX - (uint32_t)order[i]);
+
/* Decode */
ff_vk_shader_update_desc_buffer(&ctx->s, exec, &fv->decode,
1, 0, 0,
@@ -523,7 +543,8 @@ static int vk_ffv1_end_frame(AVCodecContext *avctx)
ff_vk_shader_update_desc_buffer(&ctx->s, exec, &fv->decode,
1, 1, 0,
slice_feedback,
- 0, 2*f->slice_count*sizeof(uint32_t),
+ 4*f->slice_count*sizeof(uint32_t),
+ f->slice_count*sizeof(uint32_t),
VK_FORMAT_UNDEFINED);
ff_vk_shader_update_desc_buffer(&ctx->s, exec, &fv->decode,
1, 2, 0,