Commit 460da35c4c for ffmpeg

commit 460da35c4c72903e18a485f0bed49406aef87c83
Author: Lynne <dev@lynne.ee>
Date:   Sun Sep 27 01:38:06 2026 +0900

    ffv1enc_vulkan: encode the largest slices first

    Dispatch the slices in descending order of their size in the previous
    frame, as the decoder does with the actual sizes. Several waves share a
    SIMD and the issue arbiter favours the oldest, so the slices that take
    the longest get the priority, and the frame no longer waits on a heavy
    slice that was started late.

    The order is written by the host into a small buffer read by the encode
    shader, which now keys the RGB line cache rows by the slice row rather
    than the workgroup row. The first frame is encoded in raster order.

    The output is unchanged. Encoding a 6464x4852 16-bit RGB frame with
    1024 slices on an RX 6900 XT, with the frame in VRAM, goes from
    44.2/40.9 ms to 40.8/37.0 ms with context model 1/0. It was
    202.5/178.5 ms before this series.

diff --git a/libavcodec/ffv1enc_vulkan.c b/libavcodec/ffv1enc_vulkan.c
index 265ae9bf35..89bb4a50f7 100644
--- a/libavcodec/ffv1enc_vulkan.c
+++ b/libavcodec/ffv1enc_vulkan.c
@@ -102,8 +102,18 @@ typedef struct VulkanEncodeFFv1Context {
     uint32_t max_pixels_per_slice;
     int ppi;
     int chunks;
+
+    FFVkBuffer order_buf;
+    uint32_t prev_size[MAX_SLICES];
+    int have_prev;
 } VulkanEncodeFFv1Context;

+static int cmp_slice_order(const void *a, const void *b)
+{
+    uint64_t x = *(const uint64_t *)a, y = *(const uint64_t *)b;
+    return (x < y) - (x > y);
+}
+
 extern const char *ff_source_common_comp;
 extern const char *ff_source_rangecoder_comp;
 extern const char *ff_source_ffv1_vlc_comp;
@@ -672,6 +682,35 @@ static int vulkan_encode_ffv1_submit_frame(AVCodecContext *avctx,
                                         0, remap_data_size*f->slice_count,
                                         VK_FORMAT_UNDEFINED);

+    /* Encode the slices that were largest in the previous frame first: they
+     * get the oldest waves, which have issue priority when several waves
+     * share a SIMD */
+    {
+        uint64_t order[MAX_SLICES];
+        size_t order_off = fd->idx*f->max_slice_count*sizeof(uint32_t);
+        uint32_t *dst = (uint32_t *)(fv->order_buf.mapped_mem + order_off);
+        for (int i = 0; i < f->slice_count; i++)
+            order[i] = (uint64_t)(fv->have_prev ? fv->prev_size[i] : 0) << 32 |
+                       (UINT32_MAX - i);
+        qsort(order, f->slice_count, sizeof(*order), cmp_slice_order);
+        for (int i = 0; i < f->slice_count; i++)
+            dst[i] = UINT32_MAX - (uint32_t)order[i];
+        if (!(fv->order_buf.flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT)) {
+            VkMappedMemoryRange flush_data = {
+                .sType = VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE,
+                .memory = fv->order_buf.mem,
+                .offset = 0,
+                .size = VK_WHOLE_SIZE,
+            };
+            vk->FlushMappedMemoryRanges(fv->s.hwctx->act_dev, 1, &flush_data);
+        }
+        ff_vk_shader_update_desc_buffer(&fv->s, exec, &fv->enc,
+                                        1, 6, 0,
+                                        &fv->order_buf,
+                                        order_off, f->slice_count*sizeof(uint32_t),
+                                        VK_FORMAT_UNDEFINED);
+    }
+
     ff_vk_exec_bind_shader(&fv->s, exec, &fv->enc);
     ff_vk_shader_update_push_const(&fv->s, exec, &fv->enc,
                                    VK_SHADER_STAGE_COMPUTE_BIT,
@@ -729,19 +768,25 @@ static int get_packet(AVCodecContext *avctx, FFVkExecContext *exec,
     /* Make sure the encode + gather submission is done */
     ff_vk_exec_wait(&fv->s, exec);

-    /* Invalidate the packed size if needed */
-    size_t total_off = (fd->idx*(f->max_slice_count + 1) + f->slice_count)*sizeof(uint32_t);
+    /* Invalidate the slice sizes and the packed size if needed */
+    size_t res_off = fd->idx*(f->max_slice_count + 1)*sizeof(uint32_t);
+    size_t total_off = res_off + f->slice_count*sizeof(uint32_t);
     if (!(fv->results_buf.flags & VK_MEMORY_PROPERTY_HOST_COHERENT_BIT)) {
         VkMappedMemoryRange invalidate_data = {
             .sType = VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE,
             .memory = fv->results_buf.mem,
-            .offset = total_off,
-            .size = sizeof(uint32_t),
+            .offset = 0,
+            .size = VK_WHOLE_SIZE,
         };
         vk->InvalidateMappedMemoryRanges(fv->s.hwctx->act_dev,
                                          1, &invalidate_data);
     }

+    /* Keep the slice sizes for the next frame's dispatch order */
+    memcpy(fv->prev_size, fv->results_buf.mapped_mem + res_off,
+           f->slice_count*sizeof(uint32_t));
+    fv->have_prev = 1;
+
     pkt->size = AV_RN32(fv->results_buf.mapped_mem + total_off);
     av_log(avctx, AV_LOG_VERBOSE, "Encoded data: %iMiB\n", pkt->size / (1024*1024));

@@ -1113,9 +1158,12 @@ static int init_encode_shader(AVCodecContext *avctx, VkSpecializationInfo *sl)
             .type   = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER,
             .stages = VK_SHADER_STAGE_COMPUTE_BIT,
         },
+        { /* slice_order */
+            .type   = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER,
+            .stages = VK_SHADER_STAGE_COMPUTE_BIT,
+        },
     };
-    ff_vk_shader_add_descriptor_set(&fv->s, shd, desc_set,
-                                    4 + fv->is_rgb + !!f->remap_mode, 0);
+    ff_vk_shader_add_descriptor_set(&fv->s, shd, desc_set, 7, 0);

     if (f->bayer) {
         if (fv->ctx.ac == AC_GOLOMB_RICE)
@@ -1438,6 +1486,14 @@ static av_cold int vulkan_encode_ffv1_init(AVCodecContext *avctx)
                          VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT));
     RET(ff_vk_map_buffer(&fv->s, &fv->results_buf, NULL, 0));

+    RET(ff_vk_create_buf(&fv->s, &fv->order_buf,
+                         fv->async_depth*f->max_slice_count*sizeof(uint32_t),
+                         NULL, NULL,
+                         VK_BUFFER_USAGE_STORAGE_BUFFER_BIT,
+                         VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT |
+                         VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT));
+    RET(ff_vk_map_buffer(&fv->s, &fv->order_buf, NULL, 0));
+
 fail:
     return err;
 }
@@ -1475,6 +1531,7 @@ static av_cold int vulkan_encode_ffv1_close(AVCodecContext *avctx)
     av_refstruct_pool_uninit(&fv->remap_data_pool);

     ff_vk_free_buf(&fv->s, &fv->results_buf);
+    ff_vk_free_buf(&fv->s, &fv->order_buf);

     ff_vk_free_buf(&fv->s, &fv->consts_buf);

diff --git a/libavcodec/vulkan/ffv1_enc.comp.glsl b/libavcodec/vulkan/ffv1_enc.comp.glsl
index 999c47294e..177a9f6fe8 100644
--- a/libavcodec/vulkan/ffv1_enc.comp.glsl
+++ b/libavcodec/vulkan/ffv1_enc.comp.glsl
@@ -46,6 +46,9 @@ layout (set = 1, binding = 1, scalar) writeonly buffer slice_results_buf {
  * formats this avoids the fp16/fp32 conversion that would otherwise flush
  * denormals before we get to look at them. */
 layout (set = 1, binding = 3) uniform uimage2D src[];
+layout (set = 1, binding = 6, scalar) readonly buffer slice_order_buf {
+    uint32_t slice_order[];
+};
 #ifdef FLOAT
 layout (set = 1, binding = 5, scalar) readonly buffer fltmap_buf {
     uint fltmap[];
@@ -460,7 +463,7 @@ void encode_slice(in SliceContext sc, uint slice_idx)
     int bayer_w = sc.slice_dim.x >> 1;
     int bayer_h = sc.slice_dim.y >> 1;
     sp.x >>= 1;
-    sp.y = int(gl_WorkGroupID.y)*rgb_linecache;
+    sp.y = int(slice_idx / gl_NumWorkGroups.x)*rgb_linecache;
     /* c_bits = bps + 1 for is_rgb pixfmts (Bayer is treated as RGB). gm uses
      * raw bps; gd/b-gm/r-gm need an extra bit for the RCT difference. PCM
      * stores raw samples so all planes use bps. */
@@ -469,7 +472,7 @@ void encode_slice(in SliceContext sc, uint slice_idx)
     else
         bits = u16vec4(c_bits - 1, c_bits - 1, c_bits - 1, c_bits - 1);
 #elif defined(RGB)
-    sp.y = int(gl_WorkGroupID.y)*rgb_linecache;
+    sp.y = int(slice_idx / gl_NumWorkGroups.x)*rgb_linecache;
 #endif

 #ifndef GOLOMB
@@ -643,7 +646,7 @@ void finalize_slice(in uint slice_idx)

 void main(void)
 {
-    uint slice_idx = gl_WorkGroupID.y*gl_NumWorkGroups.x + gl_WorkGroupID.x;
+    uint slice_idx = slice_order[gl_WorkGroupID.y*gl_NumWorkGroups.x + gl_WorkGroupID.x];

 #ifdef GOLOMB
     if (gl_LocalInvocationID.x == 0)