This is an automated email from the git hooks/post-receive script. Git pushed a commit to branch master in repository ffmpeg.
commit 35ee00d3986d96d721eb47c87a597fa606c6ae79 Author: Lynne <[email protected]> AuthorDate: Sun Sep 27 00:49:03 2026 +0900 Commit: Lynne <[email protected]> CommitDate: Sat Oct 3 12:15:46 2026 +0900 vulkan_ffv1: read the row-above quant inputs from shared memory The context base of each chunk of 32 samples is built from inputs 1, 2 and 4 of the quant table, which every invocation read from the uniform buffer. When the slice starts, those inputs of the tables of the first plane and of the chroma planes are now copied to shared memory as int16, and the chunks read them from there. At 3 KiB, the copy leaves enough shared memory for every slice of a frame to stay resident. Any other plane reads the uniform buffer as before. Decoding a 6464x4852 16-bit RGB frame with 1024 slices on an RX 6900 XT, with the bitstream in VRAM, goes from 50.7/40.6/39.1 ms to 50.3/40.1/38.7 ms with context model 1/0/2. It was 177.1/146.6/142.9 ms before this series. --- libavcodec/vulkan/ffv1_dec.comp.glsl | 27 ++++++++++++++++++++++++++- 1 file changed, 26 insertions(+), 1 deletion(-) diff --git a/libavcodec/vulkan/ffv1_dec.comp.glsl b/libavcodec/vulkan/ffv1_dec.comp.glsl index 33eeafb433..fbc587ce9d 100644 --- a/libavcodec/vulkan/ffv1_dec.comp.glsl +++ b/libavcodec/vulkan/ffv1_dec.comp.glsl @@ -72,6 +72,8 @@ void decode_line_pcm(ivec2 sp, int w, int y, int p) } } +shared int16_t quant_top[2][3][MAX_QUANT_TABLE_SIZE]; + void decode_line(ivec2 sp, int w, int y, int p, int bits, uint state_off, uint8_t quant_table_idx, int run_index, bool ext) @@ -103,10 +105,25 @@ void decode_line(ivec2 sp, int w, ivec2 qthr = quant_ballot ? quant_thresh[quant_table_idx][gl_LocalInvocationID.x] : ivec2(0); ivec2 qso = quant_ballot ? quant_scale_off[quant_table_idx] : ivec2(0); +#ifdef BAYER + int slot = p == 0 ? 0 : p > 1 ? 1 : -1; +#else + int slot = p == 0 ? 0 : p < 3 ? 1 : -1; +#endif + ivec4 tr = get_top(dec[p], sp, ivec2(min(1 + int(gl_LocalInvocationID.x), w - 1), y), 0, w, ext); for (int x = 0; x < w; x += 32) { - ivec3 tn = get_pred_top_quant(tr, quant_table_idx, ext); + ivec3 tn; + if (slot >= 0) { + int tb = quant_top[slot][0][(tr[0] - tr[1]) & MAX_QUANT_TABLE_MASK] + + quant_top[slot][1][(tr[1] - tr[2]) & MAX_QUANT_TABLE_MASK]; + if (ext) + tb += quant_top[slot][2][(tr[3] - tr[1]) & MAX_QUANT_TABLE_MASK]; + tn = ivec3(tr[0], tr[1], tb); + } else { + tn = get_pred_top_quant(tr, quant_table_idx, ext); + } tn.z += qso.y; tr = get_top(dec[p], sp, ivec2(min(x + 33 + int(gl_LocalInvocationID.x), w - 1), y), 0, w, ext); @@ -435,6 +452,14 @@ void decode_slice(in SliceContext sc, uint slice_idx) #ifdef GOLOMB slice_state_off >>= 3; // division by VLC_STATE_SIZE golomb_init(); +#else + for (uint i = gl_LocalInvocationID.x; i < 2*3*MAX_QUANT_TABLE_SIZE; i += gl_WorkGroupSize.x) { + uint t = i / (3*MAX_QUANT_TABLE_SIZE); + uint k = (i / MAX_QUANT_TABLE_SIZE) % 3; + uint e = i % MAX_QUANT_TABLE_SIZE; + quant_top[t][k][e] = int16_t(quant_table[sc.quant_table_idx[t]][k == 2 ? 4 : k + 1][e]); + } + barrier(); #endif #ifdef BAYER -- To stop receiving notification emails like this one, please contact [email protected]. _______________________________________________ ffmpeg-cvslog mailing list -- [email protected] To unsubscribe send an email to [email protected]
