Commit 24e7332342 for ffmpeg
commit 24e7332342fdacfd091399b20bba3c7c0271bbb6
Author: Lynne <dev@lynne.ee>
Date: Sun Sep 27 01:36:03 2026 +0900
ffv1enc_vulkan: compute the slice CRC with all invocations
The CRC of a range coded slice was computed a byte at a time by a
single invocation once the slice was coded.
All invocations now compute it: each takes a segment of the slice, and
the CRCs of the segments are combined, from the first, with a GF(2)
matrix that advances a CRC past a segment of zero bytes, whose columns
are computed alongside. The table is read from shared memory.
The output is unchanged. Encoding a 6464x4852 16-bit RGB frame with
1024 slices on an RX 6900 XT, with the frame in VRAM, goes from
52.9/50.1 ms to 44.2/40.9 ms with context model 1/0.
diff --git a/libavcodec/vulkan/ffv1_enc.comp.glsl b/libavcodec/vulkan/ffv1_enc.comp.glsl
index bb311532dd..999c47294e 100644
--- a/libavcodec/vulkan/ffv1_enc.comp.glsl
+++ b/libavcodec/vulkan/ffv1_enc.comp.glsl
@@ -32,6 +32,7 @@
#define PB_UNALIGNED
#include "common.glsl"
#include "ffv1_common.glsl"
+#extension GL_KHR_shader_subgroup_arithmetic : require
layout (set = 0, binding = 2, scalar) uniform crc_ieee_buf {
uint32_t crc_ieee[256];
@@ -58,6 +59,7 @@ layout (set = 1, binding = 2, scalar) buffer slice_state_buf {
};
layout (constant_id = 19) const bool enc_ext = false;
+shared uint crc_tab[has_crc ? 256 : 1];
void encode_line_pcm(in SliceContext sc, readonly uimage2D img,
ivec2 sp, int y, uint p, uint comp)
@@ -560,11 +562,6 @@ void finalize_slice(in uint slice_idx)
{
#ifdef GOLOMB
uint32_t enc_len = hdr_len + flush_put_bits(pb);
-#else
- uint32_t enc_len = rac_terminate();
- if (gl_LocalInvocationID.x > 0)
- return;
-#endif
u8buf bs = u8buf(slice_data + rc.bs_start);
@@ -596,6 +593,52 @@ void finalize_slice(in uint slice_idx)
}
slice_results[slice_idx] = enc_len;
+#else
+ uint enc_len = rac_terminate();
+ uint lane = gl_SubgroupInvocationID;
+ u8buf bs = u8buf(slice_data + rc.bs_start);
+
+ if (lane < 3 + uint(has_crc))
+ bs[enc_len + lane].v = uint8_t(lane < 3 ? enc_len >> (16 - 8*lane) : 0);
+ enc_len += 3 + uint(has_crc);
+
+ if (has_crc) {
+ controlBarrier(gl_ScopeWorkgroup, gl_ScopeWorkgroup,
+ gl_StorageSemanticsBuffer, gl_SemanticsAcquireRelease);
+
+ uint seg = enc_len >> 5;
+ uint len0 = enc_len - 31*seg;
+ uint start = lane == 0 ? 0 : len0 + (lane - 1)*seg;
+ uint len = lane == 0 ? len0 : seg;
+ uint crc = lane == 0 ? crcref : 0;
+ uint z = 1u << lane;
+ for (uint i = 0; i < len0; i += 8) {
+ uint b[8];
+ [[unroll]] for (uint k = 0; k < 8; k++)
+ b[k] = i + k < len ? uint(bs[start + i + k].v) : 0;
+ [[unroll]] for (uint k = 0; k < 8; k++) {
+ if (i + k < len)
+ crc = crc_tab[(crc ^ b[k]) & 0xFF] ^ (crc >> 8);
+ if (i + k < seg)
+ z = crc_tab[z & 0xFF] ^ (z >> 8);
+ }
+ }
+
+ uint acc = subgroupBroadcast(crc, 0);
+ for (uint i = 1; i < 32; i++)
+ acc = subgroupXor(bitfieldExtract(acc, int(lane), 1) != 0 ? z : 0) ^
+ subgroupBroadcast(crc, i);
+ if (crcref != 0x00000000)
+ acc ^= 0x8CD88196;
+
+ if (lane < 4)
+ bs[enc_len + lane].v = uint8_t(acc >> (8*lane));
+ enc_len += 4;
+ }
+
+ if (lane == 0)
+ slice_results[slice_idx] = enc_len;
+#endif
}
void main(void)
@@ -607,6 +650,9 @@ void main(void)
rc = slice_ctx[slice_idx].c;
barrier();
#else
+ if (has_crc)
+ for (uint i = gl_LocalInvocationID.x; i < 256; i += gl_WorkGroupSize.x)
+ crc_tab[i] = crc_ieee[i];
rac_init_enc(slice_ctx[slice_idx].c);
#endif