diff options
author | Hubert Mazur <hum@semihalf.com> | 2022-08-16 14:20:16 +0200 |
---|---|---|
committer | Martin Storsjö <martin@martin.st> | 2022-08-18 12:07:26 +0300 |
commit | 70efa4d01188b61efc0b82e7241a59a32c7e2e22 (patch) | |
tree | 7c34db235f57268d083ab65d76a3d1d46ae72e2d /libavcodec/aarch64/me_cmp_neon.S | |
parent | 74312e80d74eebf095d0092a6bb2f1f207626174 (diff) | |
download | ffmpeg-70efa4d01188b61efc0b82e7241a59a32c7e2e22.tar.gz |
lavc/aarch64: Add neon implementation for pix_abs8
Provide optimized implementation of pix_abs8 function for arm64.
Performance comparison tests are shown below.
- pix_abs_1_0_c: 101.2
- pix_abs_1_0_neon: 22.5
- sad_1_c: 101.2
- sad_1_neon: 22.5
Benchmarks and tests are run with checkasm tool on AWS Graviton 3.
Signed-off-by: Martin Storsjö <martin@martin.st>
Diffstat (limited to 'libavcodec/aarch64/me_cmp_neon.S')
-rw-r--r-- | libavcodec/aarch64/me_cmp_neon.S | 47 |
1 files changed, 47 insertions, 0 deletions
diff --git a/libavcodec/aarch64/me_cmp_neon.S b/libavcodec/aarch64/me_cmp_neon.S index c0647c49e9..b89c25438e 100644 --- a/libavcodec/aarch64/me_cmp_neon.S +++ b/libavcodec/aarch64/me_cmp_neon.S @@ -72,6 +72,53 @@ function ff_pix_abs16_neon, export=1 ret endfunc +function ff_pix_abs8_neon, export=1 + // x0 unused + // x1 uint8_t *pix1 + // x2 uint8_t *pix2 + // x3 ptrdiff_t stride + // w4 int h + + movi v30.8h, #0 + cmp w4, #4 + b.lt 2f + +// make 4 iterations at once +1: + ld1 {v0.8b}, [x1], x3 // Load pix1 for first iteration + ld1 {v1.8b}, [x2], x3 // Load pix2 for first iteration + ld1 {v2.8b}, [x1], x3 // Load pix1 for second iteration + uabal v30.8h, v0.8b, v1.8b // Absolute difference, first iteration + ld1 {v3.8b}, [x2], x3 // Load pix2 for second iteration + ld1 {v4.8b}, [x1], x3 // Load pix1 for third iteration + uabal v30.8h, v2.8b, v3.8b // Absolute difference, second iteration + ld1 {v5.8b}, [x2], x3 // Load pix2 for third iteration + sub w4, w4, #4 // h -= 4 + ld1 {v6.8b}, [x1], x3 // Load pix1 for foruth iteration + ld1 {v7.8b}, [x2], x3 // Load pix2 for fourth iteration + uabal v30.8h, v4.8b, v5.8b // Absolute difference, third iteration + cmp w4, #4 + uabal v30.8h, v6.8b, v7.8b // Absolute difference, foruth iteration + b.ge 1b + + cbz w4, 3f + +// iterate by one +2: + ld1 {v0.8b}, [x1], x3 // Load pix1 + ld1 {v1.8b}, [x2], x3 // Load pix2 + + subs w4, w4, #1 + uabal v30.8h, v0.8b, v1.8b + b.ne 2b + +3: + uaddlv s20, v30.8h // Add up vector + fmov w0, s20 + + ret +endfunc + function ff_pix_abs16_xy2_neon, export=1 // x0 unused // x1 uint8_t *pix1 |