Commit 35ee00d398 for ffmpeg

commit 35ee00d3986d96d721eb47c87a597fa606c6ae79
Author: Lynne <dev@lynne.ee>
Date:   Sun Sep 27 00:49:03 2026 +0900

    vulkan_ffv1: read the row-above quant inputs from shared memory

    The context base of each chunk of 32 samples is built from inputs 1, 2
    and 4 of the quant table, which every invocation read from the uniform
    buffer. When the slice starts, those inputs of the tables of the first
    plane and of the chroma planes are now copied to shared memory as
    int16, and the chunks read them from there. At 3 KiB, the copy leaves
    enough shared memory for every slice of a frame to stay resident. Any
    other plane reads the uniform buffer as before.

    Decoding a 6464x4852 16-bit RGB frame with 1024 slices on an RX 6900
    XT, with the bitstream in VRAM, goes from 50.7/40.6/39.1 ms to
    50.3/40.1/38.7 ms with context model 1/0/2. It was 177.1/146.6/142.9 ms
    before this series.

diff --git a/libavcodec/vulkan/ffv1_dec.comp.glsl b/libavcodec/vulkan/ffv1_dec.comp.glsl
index 33eeafb433..fbc587ce9d 100644
--- a/libavcodec/vulkan/ffv1_dec.comp.glsl
+++ b/libavcodec/vulkan/ffv1_dec.comp.glsl
@@ -72,6 +72,8 @@ void decode_line_pcm(ivec2 sp, int w, int y, int p)
     }
 }

+shared int16_t quant_top[2][3][MAX_QUANT_TABLE_SIZE];
+
 void decode_line(ivec2 sp, int w,
                  int y, int p, int bits, uint state_off,
                  uint8_t quant_table_idx, int run_index, bool ext)
@@ -103,10 +105,25 @@ void decode_line(ivec2 sp, int w,
     ivec2 qthr = quant_ballot ? quant_thresh[quant_table_idx][gl_LocalInvocationID.x] : ivec2(0);
     ivec2 qso = quant_ballot ? quant_scale_off[quant_table_idx] : ivec2(0);

+#ifdef BAYER
+    int slot = p == 0 ? 0 : p > 1 ? 1 : -1;
+#else
+    int slot = p == 0 ? 0 : p < 3 ? 1 : -1;
+#endif
+
     ivec4 tr = get_top(dec[p], sp, ivec2(min(1 + int(gl_LocalInvocationID.x), w - 1), y),
                        0, w, ext);
     for (int x = 0; x < w; x += 32) {
-        ivec3 tn = get_pred_top_quant(tr, quant_table_idx, ext);
+        ivec3 tn;
+        if (slot >= 0) {
+            int tb = quant_top[slot][0][(tr[0] - tr[1]) & MAX_QUANT_TABLE_MASK] +
+                     quant_top[slot][1][(tr[1] - tr[2]) & MAX_QUANT_TABLE_MASK];
+            if (ext)
+                tb += quant_top[slot][2][(tr[3] - tr[1]) & MAX_QUANT_TABLE_MASK];
+            tn = ivec3(tr[0], tr[1], tb);
+        } else {
+            tn = get_pred_top_quant(tr, quant_table_idx, ext);
+        }
         tn.z += qso.y;
         tr = get_top(dec[p], sp, ivec2(min(x + 33 + int(gl_LocalInvocationID.x), w - 1), y),
                      0, w, ext);
@@ -435,6 +452,14 @@ void decode_slice(in SliceContext sc, uint slice_idx)
 #ifdef GOLOMB
     slice_state_off >>= 3; // division by VLC_STATE_SIZE
     golomb_init();
+#else
+    for (uint i = gl_LocalInvocationID.x; i < 2*3*MAX_QUANT_TABLE_SIZE; i += gl_WorkGroupSize.x) {
+        uint t = i / (3*MAX_QUANT_TABLE_SIZE);
+        uint k = (i / MAX_QUANT_TABLE_SIZE) % 3;
+        uint e = i % MAX_QUANT_TABLE_SIZE;
+        quant_top[t][k][e] = int16_t(quant_table[sc.quant_table_idx[t]][k == 2 ? 4 : k + 1][e]);
+    }
+    barrier();
 #endif

 #ifdef BAYER