diff mbox series

[FFmpeg-devel,3/4] lavu/riscv: use Zbb CPOP/CPOPW at run-time

Message ID 20240608113717.1677043-3-remi@remlab.net
State New
Headers show
Series [FFmpeg-devel,1/4] riscv: probe for Zbb extension at load time | expand

Checks

Context Check Description
andriy/make_x86 success Make finished
andriy/make_fate_x86 success Make fate finished

Commit Message

Rémi Denis-Courmont June 8, 2024, 11:37 a.m. UTC
Zbb static    Zbb dynamic   I baseline
popcount  1.336129286   3.469067758   20.146362909
popcountl 1.336322291   3.340292968   20.224829821
(seconds for 1 billion iterations on a SiFive-U74 core)
---
 libavutil/riscv/intmath.h | 73 ++++++++++++++++++++++++++++++++++++---
 1 file changed, 69 insertions(+), 4 deletions(-)
diff mbox series

Patch

diff --git a/libavutil/riscv/intmath.h b/libavutil/riscv/intmath.h
index ae9ee7775b..1f0afbc81d 100644
--- a/libavutil/riscv/intmath.h
+++ b/libavutil/riscv/intmath.h
@@ -1,4 +1,6 @@ 
 /*
+ * Copyright © 2022-2024 Rémi Denis-Courmont.
+ *
  * This file is part of FFmpeg.
  *
  * FFmpeg is free software; you can redistribute it and/or
@@ -23,6 +25,7 @@ 
 
 #include "config.h"
 #include "libavutil/attributes.h"
+#include "libavutil/riscv/cpu.h"
 
 /*
  * The compiler is forced to sign-extend the result anyhow, so it is faster to
@@ -70,12 +73,74 @@  static av_always_inline av_const int av_clip_intp2_rvi(int a, int p)
 }
 
 #if defined (__GNUC__) || defined (__clang__)
-#define av_popcount   __builtin_popcount
-#if (__riscv_xlen >= 64)
-#define av_popcount64 __builtin_popcountl
+static inline av_const int av_popcount_rv(unsigned int x)
+{
+#if HAVE_RV && !defined(__riscv_zbb)
+    if (!__builtin_constant_p(x) &&
+        __builtin_expect(ff_rv_zbb_support(), true)) {
+        int y;
+
+        __asm__ (
+            ".option push\n"
+            ".option arch, +zbb\n"
+#if __riscv_xlen >= 64
+            "cpopw   %0, %1\n"
 #else
-#define av_popcount64 __builtin_popcountll
+            "cpop    %0, %1\n"
+#endif
+            ".option pop" : "=r" (y) : "r" (x));
+        if (y > 32)
+            __builtin_unreachable();
+        return y;
+    }
+#endif
+    return __builtin_popcount(x);
+}
+#define av_popcount av_popcount_rv
+
+static inline av_const int av_popcount64_rv(uint64_t x)
+{
+#if HAVE_RV && !defined(__riscv_zbb) && __riscv_xlen >= 64
+    if (!__builtin_constant_p(x) &&
+        __builtin_expect(ff_rv_zbb_support(), true)) {
+        int y;
+
+        __asm__ (
+            ".option push\n"
+            ".option arch, +zbb\n"
+            "cpop    %0, %1\n"
+            ".option pop" : "=r" (y) : "r" (x));
+        if (y > 64)
+            __builtin_unreachable();
+        return y;
+    }
 #endif
+    return __builtin_popcountl(x);
+}
+#define av_popcount64 av_popcount64_rv
+
+static inline av_const int av_parity_rv(unsigned int x)
+{
+#if HAVE_RV && !defined(__riscv_zbb)
+    if (!__builtin_constant_p(x) &&
+        __builtin_expect(ff_rv_zbb_support(), true)) {
+        int y;
+
+        __asm__ (
+            ".option push\n"
+            ".option arch, +zbb\n"
+#if __riscv_xlen >= 64
+            "cpopw   %0, %1\n"
+#else
+            "cpop    %0, %1\n"
+#endif
+            ".option pop" : "=r" (y) : "r" (x));
+        return y & 1;
+    }
+#endif
+    return __builtin_parity(x);
+}
+#define av_parity av_parity_rv
 #endif
 
 #endif /* AVUTIL_RISCV_INTMATH_H */