Add Neon implementation of vpx_hadamard_32x32

author Andrew Salkeld <andrew.salkeld@arm.com>

Thu, 13 Oct 2022 15:28:41 +0000 (16:28 +0100)

committer Jonathan Wright <jonathan.wright@arm.com>

Fri, 4 Nov 2022 23:05:36 +0000 (23:05 +0000)
author Andrew Salkeld <andrew.salkeld@arm.com>
Thu, 13 Oct 2022 15:28:41 +0000 (16:28 +0100)
committer Jonathan Wright <jonathan.wright@arm.com>
Fri, 4 Nov 2022 23:05:36 +0000 (23:05 +0000)
diff --git a/test/hadamard_test.cc b/test/hadamard_test.cc

index 10b1e79c10eb06e91c04980bd101dc8843e76f35..f904e814ad90ffdf653fbe3bb3bef6d014cd2ed3 100644 (file)
--- a/test/hadamard_test.cc
+++ b/test/hadamard_test.cc
@@ -264,7 +264,8 @@ INSTANTIATE_TEST_SUITE_P(
  INSTANTIATE_TEST_SUITE_P(
      NEON, HadamardLowbdTest,
      ::testing::Values(HadamardFuncWithSize(&vpx_hadamard_8x8_neon, 8),
-                      HadamardFuncWithSize(&vpx_hadamard_16x16_neon, 16)));
+                      HadamardFuncWithSize(&vpx_hadamard_16x16_neon, 16),
+                      HadamardFuncWithSize(&vpx_hadamard_32x32_neon, 32)));
  #endif  // HAVE_NEON
  
  // TODO(jingning): Remove highbitdepth flag when the SIMD functions are
diff --git a/vpx_dsp/arm/hadamard_neon.c b/vpx_dsp/arm/hadamard_neon.c

index 523a63c6f73ecdd3dc96490f97700d31eff16036..f6b6d7e3ce9365072b9a70d3df25a81c7c13114e 100644 (file)
--- a/vpx_dsp/arm/hadamard_neon.c
+++ b/vpx_dsp/arm/hadamard_neon.c
@@ -114,3 +114,45 @@ void vpx_hadamard_16x16_neon(const int16_t *src_diff, ptrdiff_t src_stride,
      coeff += 8;
    }
  }
+
+void vpx_hadamard_32x32_neon(const int16_t *src_diff, ptrdiff_t src_stride,
+                             tran_low_t *coeff) {
+  int i;
+
+  /* Rearrange 32x32 to 16x64 and remove stride.
+   * Top left first. */
+  vpx_hadamard_16x16_neon(src_diff + 0 + 0 * src_stride, src_stride, coeff + 0);
+  /* Top right. */
+  vpx_hadamard_16x16_neon(src_diff + 16 + 0 * src_stride, src_stride,
+                          coeff + 256);
+  /* Bottom left. */
+  vpx_hadamard_16x16_neon(src_diff + 0 + 16 * src_stride, src_stride,
+                          coeff + 512);
+  /* Bottom right. */
+  vpx_hadamard_16x16_neon(src_diff + 16 + 16 * src_stride, src_stride,
+                          coeff + 768);
+
+  for (i = 0; i < 256; i += 8) {
+    const int16x8_t a0 = load_tran_low_to_s16q(coeff + 0);
+    const int16x8_t a1 = load_tran_low_to_s16q(coeff + 256);
+    const int16x8_t a2 = load_tran_low_to_s16q(coeff + 512);
+    const int16x8_t a3 = load_tran_low_to_s16q(coeff + 768);
+
+    const int16x8_t b0 = vhaddq_s16(a0, a1);
+    const int16x8_t b1 = vhsubq_s16(a0, a1);
+    const int16x8_t b2 = vhaddq_s16(a2, a3);
+    const int16x8_t b3 = vhsubq_s16(a2, a3);
+
+    const int16x8_t c0 = vhaddq_s16(b0, b2);
+    const int16x8_t c1 = vhaddq_s16(b1, b3);
+    const int16x8_t c2 = vhsubq_s16(b0, b2);
+    const int16x8_t c3 = vhsubq_s16(b1, b3);
+
+    store_s16q_to_tran_low(coeff + 0, c0);
+    store_s16q_to_tran_low(coeff + 256, c1);
+    store_s16q_to_tran_low(coeff + 512, c2);
+    store_s16q_to_tran_low(coeff + 768, c3);
+
+    coeff += 8;
+  }
+}
diff --git a/vpx_dsp/vpx_dsp_rtcd_defs.pl b/vpx_dsp/vpx_dsp_rtcd_defs.pl

index d55ab67ce1895c70cd72a0f6f15b958a26755696..51f5ebedd6971f80bb3e92f4af288b54019f90c0 100644 (file)
--- a/vpx_dsp/vpx_dsp_rtcd_defs.pl
+++ b/vpx_dsp/vpx_dsp_rtcd_defs.pl
@@ -799,7 +799,7 @@ if (vpx_config("CONFIG_VP9_ENCODER") eq "yes") {
      specialize qw/vpx_hadamard_16x16 avx2 sse2 neon vsx lsx/;
  
      add_proto qw/void vpx_hadamard_32x32/, "const int16_t *src_diff, ptrdiff_t src_stride, tran_low_t *coeff";
-    specialize qw/vpx_hadamard_32x32 sse2 avx2/;
+    specialize qw/vpx_hadamard_32x32 sse2 avx2 neon/;
  
      add_proto qw/void vpx_highbd_hadamard_8x8/, "const int16_t *src_diff, ptrdiff_t src_stride, tran_low_t *coeff";
      specialize qw/vpx_highbd_hadamard_8x8 avx2/;
@@ -823,7 +823,7 @@ if (vpx_config("CONFIG_VP9_ENCODER") eq "yes") {
      specialize qw/vpx_hadamard_16x16 avx2 sse2 neon msa vsx lsx/;
  
      add_proto qw/void vpx_hadamard_32x32/, "const int16_t *src_diff, ptrdiff_t src_stride, int16_t *coeff";
-    specialize qw/vpx_hadamard_32x32 sse2 avx2/;
+    specialize qw/vpx_hadamard_32x32 sse2 avx2 neon/;
  
      add_proto qw/int vpx_satd/, "const int16_t *coeff, int length";
      specialize qw/vpx_satd avx2 sse2 neon msa/;
author	Andrew Salkeld <andrew.salkeld@arm.com>
	Thu, 13 Oct 2022 15:28:41 +0000 (16:28 +0100)
committer	Jonathan Wright <jonathan.wright@arm.com>
	Fri, 4 Nov 2022 23:05:36 +0000 (23:05 +0000)
test/hadamard_test.cc		patch \| blob \| history
vpx_dsp/arm/hadamard_neon.c		patch \| blob \| history
vpx_dsp/vpx_dsp_rtcd_defs.pl		patch \| blob \| history