This is an automated email from the git hooks/post-receive script. Git pushed a commit to branch master in repository ffmpeg.
commit 9c852d9e23f13c836c75039aa762b6a181f6ce5f Author: Zuxy Meng <[email protected]> AuthorDate: Fri Sep 4 21:25:24 2026 -0700 Commit: Zuxy Meng <[email protected]> CommitDate: Thu Oct 1 17:49:31 2026 -0700 avcodec/arm|aarch64/h264dsp: Fix edge cases of bi-weight for neon S16 saturating sum must be computed dot-product-first. The NEON code accumulated the offset into the dot product with wrapping multiply-accumate, so large-magnitude sums wrapped instead of saturating. Accumulate the dot product from zero (wrapping is exact: the dot product cannot overflow S16 for 8-bit with log2_denom < 7), then add the offset with a saturating add. This fixes bi-weight checkasm test for NEON. Signed-off-by: Zuxy Meng <[email protected]> --- libavcodec/aarch64/h264dsp_neon.S | 36 ++++++++++++++++++++++-------------- libavcodec/arm/h264dsp_neon.S | 36 ++++++++++++++++++++++-------------- tests/checkasm/h264dsp.c | 2 +- 3 files changed, 45 insertions(+), 29 deletions(-) diff --git a/libavcodec/aarch64/h264dsp_neon.S b/libavcodec/aarch64/h264dsp_neon.S index c09db164b6..06c0dcbbf5 100644 --- a/libavcodec/aarch64/h264dsp_neon.S +++ b/libavcodec/aarch64/h264dsp_neon.S @@ -586,8 +586,8 @@ endfunc .macro biweight_16 macs, macd dup v0.16b, w5 dup v1.16b, w6 - mov v4.16b, v16.16b - mov v6.16b, v16.16b + movi v4.2d, #0 + movi v6.2d, #0 1: subs w3, w3, #2 ld1 {v20.16b}, [x0], x2 \macd v4.8h, v0.8b, v20.8b @@ -595,14 +595,18 @@ endfunc ld1 {v22.16b}, [x1], x2 \macs v4.8h, v1.8b, v22.8b \macs\()2 v6.8H, v1.16B, v22.16B - mov v24.16b, v16.16b + sqadd v4.8h, v4.8h, v16.8h + sqadd v6.8h, v6.8h, v16.8h + movi v24.2d, #0 ld1 {v28.16b}, [x0], x2 - mov v26.16b, v16.16b + movi v26.2d, #0 \macd v24.8h, v0.8b, v28.8b \macd\()2 v26.8H, v0.16B, v28.16B ld1 {v30.16b}, [x1], x2 \macs v24.8h, v1.8b, v30.8b \macs\()2 v26.8H, v1.16B, v30.16B + sqadd v24.8h, v24.8h, v16.8h + sqadd v26.8h, v26.8h, v16.8h sshl v4.8h, v4.8h, v18.8h sshl v6.8h, v6.8h, v18.8h sqxtun v4.8b, v4.8h @@ -611,9 +615,9 @@ endfunc sshl v26.8h, v26.8h, v18.8h sqxtun v24.8b, v24.8h sqxtun2 v24.16b, v26.8h - mov v6.16b, v16.16b + movi v6.2d, #0 st1 {v4.16b}, [x7], x2 - mov v4.16b, v16.16b + movi v4.2d, #0 st1 {v24.16b}, [x7], x2 b.ne 1b ret @@ -622,24 +626,26 @@ endfunc .macro biweight_8 macs, macd dup v0.8b, w5 dup v1.8b, w6 - mov v2.16b, v16.16b - mov v20.16b, v16.16b + movi v2.2d, #0 + movi v20.2d, #0 1: subs w3, w3, #2 ld1 {v4.8b}, [x0], x2 \macd v2.8h, v0.8b, v4.8b ld1 {v5.8b}, [x1], x2 \macs v2.8h, v1.8b, v5.8b + sqadd v2.8h, v2.8h, v16.8h ld1 {v6.8b}, [x0], x2 \macd v20.8h, v0.8b, v6.8b ld1 {v7.8b}, [x1], x2 \macs v20.8h, v1.8b, v7.8b + sqadd v20.8h, v20.8h, v16.8h sshl v2.8h, v2.8h, v18.8h sqxtun v2.8b, v2.8h sshl v20.8h, v20.8h, v18.8h sqxtun v4.8b, v20.8h - mov v20.16b, v16.16b + movi v20.2d, #0 st1 {v2.8b}, [x7], x2 - mov v2.16b, v16.16b + movi v2.2d, #0 st1 {v4.8b}, [x7], x2 b.ne 1b ret @@ -648,8 +654,8 @@ endfunc .macro biweight_4 macs, macd dup v0.8b, w5 dup v1.8b, w6 - mov v2.16b, v16.16b - mov v20.16b,v16.16b + movi v2.2d, #0 + movi v20.2d, #0 1: subs w3, w3, #4 ld1 {v4.s}[0], [x0], x2 ld1 {v4.s}[1], [x0], x2 @@ -657,6 +663,7 @@ endfunc ld1 {v5.s}[0], [x1], x2 ld1 {v5.s}[1], [x1], x2 \macs v2.8h, v1.8b, v5.8b + sqadd v2.8h, v2.8h, v16.8h b.lt 2f ld1 {v6.s}[0], [x0], x2 ld1 {v6.s}[1], [x0], x2 @@ -664,14 +671,15 @@ endfunc ld1 {v7.s}[0], [x1], x2 ld1 {v7.s}[1], [x1], x2 \macs v20.8h, v1.8b, v7.8b + sqadd v20.8h, v20.8h, v16.8h sshl v2.8h, v2.8h, v18.8h sqxtun v2.8b, v2.8h sshl v20.8h, v20.8h, v18.8h sqxtun v4.8b, v20.8h - mov v20.16b, v16.16b + movi v20.2d, #0 st1 {v2.s}[0], [x7], x2 st1 {v2.s}[1], [x7], x2 - mov v2.16b, v16.16b + movi v2.2d, #0 st1 {v4.s}[0], [x7], x2 st1 {v4.s}[1], [x7], x2 b.ne 1b diff --git a/libavcodec/arm/h264dsp_neon.S b/libavcodec/arm/h264dsp_neon.S index 975b61c8c0..ee40807784 100644 --- a/libavcodec/arm/h264dsp_neon.S +++ b/libavcodec/arm/h264dsp_neon.S @@ -295,8 +295,8 @@ endfunc .macro biweight_16 macs, macd vdup.8 d0, r4 vdup.8 d1, r5 - vmov q2, q8 - vmov q3, q8 + vmov.i64 q2, #0 + vmov.i64 q3, #0 1: subs r3, r3, #2 vld1.8 {d20-d21},[r0,:128], r2 \macd q2, d0, d20 @@ -306,9 +306,11 @@ endfunc \macs q2, d1, d22 pld [r1] \macs q3, d1, d23 - vmov q12, q8 + vqadd.s16 q2, q2, q8 + vqadd.s16 q3, q3, q8 + vmov.i64 q12, #0 vld1.8 {d28-d29},[r0,:128], r2 - vmov q13, q8 + vmov.i64 q13, #0 \macd q12, d0, d28 pld [r0] \macd q13, d0, d29 @@ -316,6 +318,8 @@ endfunc \macs q12, d1, d30 pld [r1] \macs q13, d1, d31 + vqadd.s16 q12, q12, q8 + vqadd.s16 q13, q13, q8 vshl.s16 q2, q2, q9 vshl.s16 q3, q3, q9 vqmovun.s16 d4, q2 @@ -324,9 +328,9 @@ endfunc vshl.s16 q13, q13, q9 vqmovun.s16 d24, q12 vqmovun.s16 d25, q13 - vmov q3, q8 + vmov.i64 q3, #0 vst1.8 {d4- d5}, [r6,:128], r2 - vmov q2, q8 + vmov.i64 q2, #0 vst1.8 {d24-d25},[r6,:128], r2 bne 1b pop {r4-r6, pc} @@ -335,8 +339,8 @@ endfunc .macro biweight_8 macs, macd vdup.8 d0, r4 vdup.8 d1, r5 - vmov q1, q8 - vmov q10, q8 + vmov.i64 q1, #0 + vmov.i64 q10, #0 1: subs r3, r3, #2 vld1.8 {d4},[r0,:64], r2 \macd q1, d0, d4 @@ -344,19 +348,21 @@ endfunc vld1.8 {d5},[r1,:64], r2 \macs q1, d1, d5 pld [r1] + vqadd.s16 q1, q1, q8 vld1.8 {d6},[r0,:64], r2 \macd q10, d0, d6 pld [r0] vld1.8 {d7},[r1,:64], r2 \macs q10, d1, d7 pld [r1] + vqadd.s16 q10, q10, q8 vshl.s16 q1, q1, q9 vqmovun.s16 d2, q1 vshl.s16 q10, q10, q9 vqmovun.s16 d4, q10 - vmov q10, q8 + vmov.i64 q10, #0 vst1.8 {d2},[r6,:64], r2 - vmov q1, q8 + vmov.i64 q1, #0 vst1.8 {d4},[r6,:64], r2 bne 1b pop {r4-r6, pc} @@ -365,8 +371,8 @@ endfunc .macro biweight_4 macs, macd vdup.8 d0, r4 vdup.8 d1, r5 - vmov q1, q8 - vmov q10, q8 + vmov.i64 q1, #0 + vmov.i64 q10, #0 1: subs r3, r3, #4 vld1.32 {d4[0]},[r0,:32], r2 vld1.32 {d4[1]},[r0,:32], r2 @@ -376,6 +382,7 @@ endfunc vld1.32 {d5[1]},[r1,:32], r2 \macs q1, d1, d5 pld [r1] + vqadd.s16 q1, q1, q8 blt 2f vld1.32 {d6[0]},[r0,:32], r2 vld1.32 {d6[1]},[r0,:32], r2 @@ -385,14 +392,15 @@ endfunc vld1.32 {d7[1]},[r1,:32], r2 \macs q10, d1, d7 pld [r1] + vqadd.s16 q10, q10, q8 vshl.s16 q1, q1, q9 vqmovun.s16 d2, q1 vshl.s16 q10, q10, q9 vqmovun.s16 d4, q10 - vmov q10, q8 + vmov.i64 q10, #0 vst1.32 {d2[0]},[r6,:32], r2 vst1.32 {d2[1]},[r6,:32], r2 - vmov q1, q8 + vmov.i64 q1, #0 vst1.32 {d4[0]},[r6,:32], r2 vst1.32 {d4[1]},[r6,:32], r2 bne 1b diff --git a/tests/checkasm/h264dsp.c b/tests/checkasm/h264dsp.c index 6cdc44b5ef..06d7cc325c 100644 --- a/tests/checkasm/h264dsp.c +++ b/tests/checkasm/h264dsp.c @@ -545,7 +545,7 @@ static void check_weight(void) } // only archs that can pass test -#define H264_CHECK_BIWEIGHT (ARCH_X86 || ARCH_PPC || ARCH_MIPS) +#define H264_CHECK_BIWEIGHT (ARCH_X86 || ARCH_PPC || ARCH_MIPS || ARCH_ARM || ARCH_AARCH64) #if H264_CHECK_BIWEIGHT static void check_biweight(void) -- To stop receiving notification emails like this one, please contact [email protected]. _______________________________________________ ffmpeg-cvslog mailing list -- [email protected] To unsubscribe send an email to [email protected]
