From d7abb7d143fd1fbacb0084a8936bc4029afe5111 Mon Sep 17 00:00:00 2001 From: Hubert Mazur Date: Tue, 16 Aug 2022 14:20:13 +0200 Subject: [PATCH] lavc/aarch64: Add neon implementation for sse4 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Provide neon implementation for sse4 function. Performance comparison tests are shown below. - sse_2_c: 80.7 - sse_2_neon: 31.0 Benchmarks and tests are run with checkasm tool on AWS Graviton 3. Signed-off-by: Hubert Mazur Signed-off-by: Martin Storsjö --- libavcodec/aarch64/me_cmp_init_aarch64.c | 3 ++ libavcodec/aarch64/me_cmp_neon.S | 56 ++++++++++++++++++++++++ 2 files changed, 59 insertions(+) diff --git a/libavcodec/aarch64/me_cmp_init_aarch64.c b/libavcodec/aarch64/me_cmp_init_aarch64.c index ab2a1909ba..57722b6a9a 100644 --- a/libavcodec/aarch64/me_cmp_init_aarch64.c +++ b/libavcodec/aarch64/me_cmp_init_aarch64.c @@ -32,6 +32,8 @@ int ff_pix_abs16_x2_neon(MpegEncContext *v, const uint8_t *pix1, const uint8_t * int sse16_neon(MpegEncContext *v, const uint8_t *pix1, const uint8_t *pix2, ptrdiff_t stride, int h); +int sse4_neon(MpegEncContext *v, const uint8_t *pix1, const uint8_t *pix2, + ptrdiff_t stride, int h); av_cold void ff_me_cmp_init_aarch64(MECmpContext *c, AVCodecContext *avctx) { @@ -44,5 +46,6 @@ av_cold void ff_me_cmp_init_aarch64(MECmpContext *c, AVCodecContext *avctx) c->sad[0] = ff_pix_abs16_neon; c->sse[0] = sse16_neon; + c->sse[2] = sse4_neon; } } diff --git a/libavcodec/aarch64/me_cmp_neon.S b/libavcodec/aarch64/me_cmp_neon.S index b98b2b7e03..f3201739b8 100644 --- a/libavcodec/aarch64/me_cmp_neon.S +++ b/libavcodec/aarch64/me_cmp_neon.S @@ -344,3 +344,59 @@ function sse16_neon, export=1 ret endfunc + +function sse4_neon, export=1 + // x0 - unused + // x1 - pix1 + // x2 - pix2 + // x3 - stride + // w4 - h + + movi v16.4s, #0 // clear the result accumulator + cmp w4, #4 + b.le 2f + +// make 4 iterations at once +1: + + // res = abs(pix1[0] - pix2[0]) + // res * res + + ld1 {v0.s}[0], [x1], x3 // Load pix1, first iteration + ld1 {v1.s}[0], [x2], x3 // Load pix2, first iteration + ld1 {v2.s}[0], [x1], x3 // Load pix1, second iteration + ld1 {v3.s}[0], [x2], x3 // Load pix2, second iteration + uabdl v30.8h, v0.8b, v1.8b // Absolute difference, first iteration + ld1 {v4.s}[0], [x1], x3 // Load pix1, third iteration + ld1 {v5.s}[0], [x2], x3 // Load pix2, third iteration + uabdl v29.8h, v2.8b, v3.8b // Absolute difference, second iteration + umlal v16.4s, v30.4h, v30.4h // Multiply vectors, first iteration + ld1 {v6.s}[0], [x1], x3 // Load pix1, fourth iteration + ld1 {v7.s}[0], [x2], x3 // Load pix2, fourth iteration + uabdl v28.8h, v4.8b, v5.8b // Absolute difference, third iteration + umlal v16.4s, v29.4h, v29.4h // Multiply and accumulate, second iteration + sub w4, w4, #4 + uabdl v27.8h, v6.8b, v7.8b // Absolue difference, fourth iteration + umlal v16.4s, v28.4h, v28.4h // Multiply and accumulate, third iteration + cmp w4, #4 + umlal v16.4s, v27.4h, v27.4h // Multiply and accumulate, fourth iteration + b.ge 1b + + cbz w4, 3f + +// iterate by one +2: + ld1 {v0.s}[0], [x1], x3 // Load pix1 + ld1 {v1.s}[0], [x2], x3 // Load pix2 + uabdl v30.8h, v0.8b, v1.8b + subs w4, w4, #1 + umlal v16.4s, v30.4h, v30.4h + + b.ne 2b + +3: + uaddlv d17, v16.4s // Add vector + fmov w0, s17 + + ret +endfunc