shithub: libvpx

--- a/vp9/common/vp9_rtcd_defs.pl

+++ b/vp9/common/vp9_rtcd_defs.pl

@@ -788,7 +788,7 @@

   specialize qw/vp9_block_error avx2 msa/, "$sse2_x86inc";

   add_proto qw/int64_t vp9_block_error_fp/, "const int16_t *coeff, const int16_t *dqcoeff, int block_size";

-  specialize qw/vp9_block_error_fp/, "$sse2_x86inc";

+  specialize qw/vp9_block_error_fp neon/, "$sse2_x86inc";

   add_proto qw/void vp9_quantize_fp/, "const tran_low_t *coeff_ptr, intptr_t n_coeffs, int skip_block, const int16_t *zbin_ptr, const int16_t *round_ptr, const int16_t *quant_ptr, const int16_t *quant_shift_ptr, tran_low_t *qcoeff_ptr, tran_low_t *dqcoeff_ptr, const int16_t *dequant_ptr, uint16_t *eob_ptr, const int16_t *scan, const int16_t *iscan";

   specialize qw/vp9_quantize_fp neon sse2/, "$ssse3_x86_64_x86inc";

--- /dev/null

+++ b/vp9/encoder/arm/neon/vp9_error_neon.c

@@ -1,0 +1,41 @@

+/*

+ *  Copyright (c) 2015 The WebM project authors. All Rights Reserved.

+ *

+ *  Use of this source code is governed by a BSD-style license

+ *  that can be found in the LICENSE file in the root of the source

+ *  tree. An additional intellectual property rights grant can be found

+ *  in the file PATENTS.  All contributing project authors may

+ *  be found in the AUTHORS file in the root of the source tree.

+ */

+#include <arm_neon.h>

+#include <assert.h>

+#include "./vp9_rtcd.h"

+int64_t vp9_block_error_fp_neon(const int16_t *coeff, const int16_t *dqcoeff,

+                                int block_size) {

+  int64x2_t error = vdupq_n_s64(0);

+  assert(block_size >= 8);

+  assert((block_size % 8) == 0);

+  do {

+    const int16x8_t c = vld1q_s16(coeff);

+    const int16x8_t d = vld1q_s16(dqcoeff);

+    const int16x8_t diff = vsubq_s16(c, d);

+    const int16x4_t diff_lo = vget_low_s16(diff);

+    const int16x4_t diff_hi = vget_high_s16(diff);

+    // diff is 15-bits, the squares 30, so we can store 2 in 31-bits before

+    // accumulating them in 64-bits.

+    const int32x4_t err0 = vmull_s16(diff_lo, diff_lo);

+    const int32x4_t err1 = vmlal_s16(err0, diff_hi, diff_hi);

+    const int64x2_t err2 = vaddl_s32(vget_low_s32(err1), vget_high_s32(err1));

+    error = vaddq_s64(error, err2);

+    coeff += 8;

+    dqcoeff += 8;

+    block_size -= 8;

+  } while (block_size != 0);

+  return vgetq_lane_s64(error, 0) + vgetq_lane_s64(error, 1);

+}

--- a/vp9/vp9cx.mk

+++ b/vp9/vp9cx.mk

@@ -130,6 +130,7 @@

 ifneq ($(CONFIG_VP9_HIGHBITDEPTH),yes)

 VP9_CX_SRCS-$(HAVE_NEON) += encoder/arm/neon/vp9_dct_neon.c

+VP9_CX_SRCS-$(HAVE_NEON) += encoder/arm/neon/vp9_error_neon.c

 endif

 VP9_CX_SRCS-$(HAVE_NEON) += encoder/arm/neon/vp9_avg_neon.c

 VP9_CX_SRCS-$(HAVE_NEON) += encoder/arm/neon/vp9_quantize_neon.c