diff options
Diffstat (limited to 'libavcodec/aarch64/me_cmp_neon.S')
-rw-r--r-- | libavcodec/aarch64/me_cmp_neon.S | 56 |
1 files changed, 56 insertions, 0 deletions
diff --git a/libavcodec/aarch64/me_cmp_neon.S b/libavcodec/aarch64/me_cmp_neon.S index b98b2b7e03..f3201739b8 100644 --- a/libavcodec/aarch64/me_cmp_neon.S +++ b/libavcodec/aarch64/me_cmp_neon.S @@ -344,3 +344,59 @@ function sse16_neon, export=1 ret endfunc + +function sse4_neon, export=1 + // x0 - unused + // x1 - pix1 + // x2 - pix2 + // x3 - stride + // w4 - h + + movi v16.4s, #0 // clear the result accumulator + cmp w4, #4 + b.le 2f + +// make 4 iterations at once +1: + + // res = abs(pix1[0] - pix2[0]) + // res * res + + ld1 {v0.s}[0], [x1], x3 // Load pix1, first iteration + ld1 {v1.s}[0], [x2], x3 // Load pix2, first iteration + ld1 {v2.s}[0], [x1], x3 // Load pix1, second iteration + ld1 {v3.s}[0], [x2], x3 // Load pix2, second iteration + uabdl v30.8h, v0.8b, v1.8b // Absolute difference, first iteration + ld1 {v4.s}[0], [x1], x3 // Load pix1, third iteration + ld1 {v5.s}[0], [x2], x3 // Load pix2, third iteration + uabdl v29.8h, v2.8b, v3.8b // Absolute difference, second iteration + umlal v16.4s, v30.4h, v30.4h // Multiply vectors, first iteration + ld1 {v6.s}[0], [x1], x3 // Load pix1, fourth iteration + ld1 {v7.s}[0], [x2], x3 // Load pix2, fourth iteration + uabdl v28.8h, v4.8b, v5.8b // Absolute difference, third iteration + umlal v16.4s, v29.4h, v29.4h // Multiply and accumulate, second iteration + sub w4, w4, #4 + uabdl v27.8h, v6.8b, v7.8b // Absolue difference, fourth iteration + umlal v16.4s, v28.4h, v28.4h // Multiply and accumulate, third iteration + cmp w4, #4 + umlal v16.4s, v27.4h, v27.4h // Multiply and accumulate, fourth iteration + b.ge 1b + + cbz w4, 3f + +// iterate by one +2: + ld1 {v0.s}[0], [x1], x3 // Load pix1 + ld1 {v1.s}[0], [x2], x3 // Load pix2 + uabdl v30.8h, v0.8b, v1.8b + subs w4, w4, #1 + umlal v16.4s, v30.4h, v30.4h + + b.ne 2b + +3: + uaddlv d17, v16.4s // Add vector + fmov w0, s17 + + ret +endfunc |