npm - whisper.rn - Versions diffs - 0.4.2 → 0.5.0-rc.0 - Mend

whisper.rn 0.4.2 → 0.5.0-rc.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (98) hide show

package/cpp/ggml-cpu/simd-mappings.h CHANGED Viewed

@@ -2,10 +2,167 @@
 #include "ggml-cpu-impl.h"
+#ifdef __ARM_FEATURE_SVE
+#include <arm_sve.h>
+#endif // __ARM_FEATURE_SVE
+#if defined(__ARM_NEON) && !defined(__CUDACC__) && !defined(__MUSACC__)
+// if YCM cannot find <arm_neon.h>, make a symbolic link to it, for example:
+//
+//   $ ln -sfn /Library/Developer/CommandLineTools/usr/lib/clang/13.1.6/include/arm_neon.h ./src/
+//
+#include <arm_neon.h>
+#endif
+#if defined(__F16C__)
+#include <immintrin.h>
+#endif
+#ifdef __cplusplus
+extern "C" {
+#endif
 //
 // simd mappings
 //
+// FP16 to FP32 conversion
+// 16-bit float
+// on Arm, we use __fp16
+// on x86, we use uint16_t
+//
+// for old CUDA compilers (<= 11), we use uint16_t: ref https://github.com/ggml-org/llama.cpp/pull/10616
+// for     MUSA compilers        , we use uint16_t: ref https://github.com/ggml-org/llama.cpp/pull/11843
+//
+#if defined(__ARM_NEON) && !(defined(__CUDACC__) && __CUDACC_VER_MAJOR__ <= 11) && !defined(__MUSACC__)
+    #define WSP_GGML_CPU_COMPUTE_FP16_TO_FP32(x) neon_compute_fp16_to_fp32(x)
+    #define WSP_GGML_CPU_COMPUTE_FP32_TO_FP16(x) neon_compute_fp32_to_fp16(x)
+    #define WSP_GGML_CPU_FP16_TO_FP32(x) WSP_GGML_CPU_COMPUTE_FP16_TO_FP32(x)
+    static inline float neon_compute_fp16_to_fp32(wsp_ggml_fp16_t h) {
+        __fp16 tmp;
+        memcpy(&tmp, &h, sizeof(wsp_ggml_fp16_t));
+        return (float)tmp;
+    }
+    static inline wsp_ggml_fp16_t neon_compute_fp32_to_fp16(float f) {
+        wsp_ggml_fp16_t res;
+        __fp16 tmp = f;
+        memcpy(&res, &tmp, sizeof(wsp_ggml_fp16_t));
+        return res;
+    }
+#elif defined(__F16C__)
+    #ifdef _MSC_VER
+        #define WSP_GGML_CPU_COMPUTE_FP16_TO_FP32(x) _mm_cvtss_f32(_mm_cvtph_ps(_mm_cvtsi32_si128(x)))
+        #define WSP_GGML_CPU_COMPUTE_FP32_TO_FP16(x) _mm_extract_epi16(_mm_cvtps_ph(_mm_set_ss(x), 0), 0)
+    #else
+        #define WSP_GGML_CPU_COMPUTE_FP16_TO_FP32(x) _cvtsh_ss(x)
+        #define WSP_GGML_CPU_COMPUTE_FP32_TO_FP16(x) _cvtss_sh(x, 0)
+    #endif
+#elif defined(__POWER9_VECTOR__)
+    #define WSP_GGML_CPU_COMPUTE_FP16_TO_FP32(x) power_compute_fp16_to_fp32(x)
+    #define WSP_GGML_CPU_COMPUTE_FP32_TO_FP16(x) power_compute_fp32_to_fp16(x)
+    /* the inline asm below is about 12% faster than the lookup method */
+    #define WSP_GGML_CPU_FP16_TO_FP32(x) WSP_GGML_CPU_COMPUTE_FP16_TO_FP32(x)
+    #define WSP_GGML_CPU_FP32_TO_FP16(x) WSP_GGML_CPU_COMPUTE_FP32_TO_FP16(x)
+    static inline float power_compute_fp16_to_fp32(wsp_ggml_fp16_t h) {
+        float f;
+        double d;
+        __asm__(
+            "mtfprd %0,%2\n"
+            "xscvhpdp %0,%0\n"
+            "frsp %1,%0\n" :
+            /* temp */ "=d"(d),
+            /* out */  "=f"(f):
+            /* in */   "r"(h));
+        return f;
+    }
+    static inline wsp_ggml_fp16_t power_compute_fp32_to_fp16(float f) {
+        double d;
+        wsp_ggml_fp16_t r;
+        __asm__( /* xscvdphp can work on double or single precision */
+            "xscvdphp %0,%2\n"
+            "mffprd %1,%0\n" :
+            /* temp */ "=d"(d),
+            /* out */  "=r"(r):
+            /* in */   "f"(f));
+        return r;
+    }
+#elif defined(__riscv) && defined(__riscv_zfhmin)
+    static inline float riscv_compute_fp16_to_fp32(wsp_ggml_fp16_t h) {
+        float f;
+        __asm__(
+            "fmv.h.x %[f], %[h]\n\t"
+            "fcvt.s.h %[f], %[f]"
+            : [f] "=&f" (f)
+            : [h] "r" (h)
+        );
+        return f;
+    }
+    static inline wsp_ggml_fp16_t riscv_compute_fp32_to_fp16(float f) {
+        wsp_ggml_fp16_t res;
+        __asm__(
+            "fcvt.h.s %[f], %[f]\n\t"
+            "fmv.x.h %[h], %[f]"
+            : [h] "=&r" (res)
+            : [f] "f" (f)
+        );
+        return res;
+    }
+    #define WSP_GGML_CPU_COMPUTE_FP16_TO_FP32(x) riscv_compute_fp16_to_fp32(x)
+    #define WSP_GGML_CPU_COMPUTE_FP32_TO_FP16(x) riscv_compute_fp32_to_fp16(x)
+    #define WSP_GGML_CPU_FP16_TO_FP32(x) WSP_GGML_CPU_COMPUTE_FP16_TO_FP32(x)
+    #define WSP_GGML_CPU_FP32_TO_FP16(x) WSP_GGML_CPU_COMPUTE_FP32_TO_FP16(x)
+#elif defined(__NNPA__)
+    #define WSP_GGML_CPU_COMPUTE_FP16_TO_FP32(x) nnpa_compute_fp16_to_fp32(x)
+    #define WSP_GGML_CPU_COMPUTE_FP32_TO_FP16(x) nnpa_compute_fp32_to_fp16(x)
+    #define WSP_GGML_CPU_FP16_TO_FP32(x) WSP_GGML_CPU_COMPUTE_FP16_TO_FP32(x)
+    #define WSP_GGML_CPU_FP32_TO_FP16(x) WSP_GGML_CPU_COMPUTE_FP32_TO_FP16(x)
+    static inline float nnpa_compute_fp16_to_fp32(wsp_ggml_fp16_t h) {
+        uint16x8_t v_h = vec_splats(h);
+        uint16x8_t v_hd = vec_convert_from_fp16(v_h, 0);
+        return vec_extend_to_fp32_hi(v_hd, 0)[0];
+    }
+    static inline wsp_ggml_fp16_t nnpa_compute_fp32_to_fp16(float f) {
+        float32x4_t v_f = vec_splats(f);
+        float32x4_t v_zero = vec_splats(0.0f);
+        uint16x8_t v_hd = vec_round_from_fp32(v_f, v_zero, 0);
+        uint16x8_t v_h = vec_convert_to_fp16(v_hd, 0);
+        return vec_extract(v_h, 0);
+    }
+#endif
+// precomputed f32 table for f16 (256 KB)
+// defined in ggml-cpu.c, initialized in wsp_ggml_cpu_init()
+extern float wsp_ggml_table_f32_f16[1 << 16];
+// On ARM NEON, it's quicker to directly convert x -> x instead of calling into wsp_ggml_lookup_fp16_to_fp32,
+// so we define WSP_GGML_CPU_FP16_TO_FP32 and WSP_GGML_CPU_FP32_TO_FP16 elsewhere for NEON.
+// This is also true for POWER9.
+#if !defined(WSP_GGML_CPU_FP16_TO_FP32)
+inline static float wsp_ggml_lookup_fp16_to_fp32(wsp_ggml_fp16_t f) {
+    uint16_t s;
+    memcpy(&s, &f, sizeof(uint16_t));
+    return wsp_ggml_table_f32_f16[s];
+}
+#define WSP_GGML_CPU_FP16_TO_FP32(x) wsp_ggml_lookup_fp16_to_fp32(x)
+#endif
+#if !defined(WSP_GGML_CPU_FP32_TO_FP16)
+#define WSP_GGML_CPU_FP32_TO_FP16(x) WSP_GGML_COMPUTE_FP32_TO_FP16(x)
+#endif
 // we define a common set of C macros which map to specific intrinsics based on the current architecture
 // we then implement the fundamental computation operations below using only these macros
 // adding support for new architectures requires to define the corresponding SIMD macros
@@ -415,7 +572,7 @@ static inline __m256 __avx_f32cx8_load(const wsp_ggml_fp16_t * x) {
     float tmp[8];
     for (int i = 0; i < 8; i++) {
-        tmp[i] = WSP_GGML_FP16_TO_FP32(x[i]);
+        tmp[i] = WSP_GGML_CPU_FP16_TO_FP32(x[i]);
     }
     return _mm256_loadu_ps(tmp);
@@ -426,7 +583,7 @@ static inline void __avx_f32cx8_store(wsp_ggml_fp16_t *x, __m256 y) {
     _mm256_storeu_ps(arr, y);
     for (int i = 0; i < 8; i++)
-        x[i] = WSP_GGML_FP32_TO_FP16(arr[i]);
+        x[i] = WSP_GGML_CPU_FP32_TO_FP16(arr[i]);
 }
 #define WSP_GGML_F32Cx8_LOAD(x)     __avx_f32cx8_load(x)
 #define WSP_GGML_F32Cx8_STORE(x, y) __avx_f32cx8_store(x, y)
@@ -574,10 +731,10 @@ static inline unsigned char wsp_ggml_endian_byte(int i) {
 inline static v128_t __wasm_f16x4_load(const wsp_ggml_fp16_t * p) {
     float tmp[4];
-    tmp[0] = WSP_GGML_FP16_TO_FP32(p[0]);
-    tmp[1] = WSP_GGML_FP16_TO_FP32(p[1]);
-    tmp[2] = WSP_GGML_FP16_TO_FP32(p[2]);
-    tmp[3] = WSP_GGML_FP16_TO_FP32(p[3]);
+    tmp[0] = WSP_GGML_CPU_FP16_TO_FP32(p[0]);
+    tmp[1] = WSP_GGML_CPU_FP16_TO_FP32(p[1]);
+    tmp[2] = WSP_GGML_CPU_FP16_TO_FP32(p[2]);
+    tmp[3] = WSP_GGML_CPU_FP16_TO_FP32(p[3]);
     return wasm_v128_load(tmp);
 }
@@ -587,10 +744,10 @@ inline static void __wasm_f16x4_store(wsp_ggml_fp16_t * p, v128_t x) {
     wasm_v128_store(tmp, x);
-    p[0] = WSP_GGML_FP32_TO_FP16(tmp[0]);
-    p[1] = WSP_GGML_FP32_TO_FP16(tmp[1]);
-    p[2] = WSP_GGML_FP32_TO_FP16(tmp[2]);
-    p[3] = WSP_GGML_FP32_TO_FP16(tmp[3]);
+    p[0] = WSP_GGML_CPU_FP32_TO_FP16(tmp[0]);
+    p[1] = WSP_GGML_CPU_FP32_TO_FP16(tmp[1]);
+    p[2] = WSP_GGML_CPU_FP32_TO_FP16(tmp[2]);
+    p[3] = WSP_GGML_CPU_FP32_TO_FP16(tmp[3]);
 }
 #define WSP_GGML_F16x4             v128_t
@@ -690,10 +847,10 @@ inline static void __wasm_f16x4_store(wsp_ggml_fp16_t * p, v128_t x) {
 static inline __m128 __sse_f16x4_load(const wsp_ggml_fp16_t * x) {
     float tmp[4];
-    tmp[0] = WSP_GGML_FP16_TO_FP32(x[0]);
-    tmp[1] = WSP_GGML_FP16_TO_FP32(x[1]);
-    tmp[2] = WSP_GGML_FP16_TO_FP32(x[2]);
-    tmp[3] = WSP_GGML_FP16_TO_FP32(x[3]);
+    tmp[0] = WSP_GGML_CPU_FP16_TO_FP32(x[0]);
+    tmp[1] = WSP_GGML_CPU_FP16_TO_FP32(x[1]);
+    tmp[2] = WSP_GGML_CPU_FP16_TO_FP32(x[2]);
+    tmp[3] = WSP_GGML_CPU_FP16_TO_FP32(x[3]);
     return _mm_loadu_ps(tmp);
 }
@@ -703,10 +860,10 @@ static inline void __sse_f16x4_store(wsp_ggml_fp16_t * x, __m128 y) {
     _mm_storeu_ps(arr, y);
-    x[0] = WSP_GGML_FP32_TO_FP16(arr[0]);
-    x[1] = WSP_GGML_FP32_TO_FP16(arr[1]);
-    x[2] = WSP_GGML_FP32_TO_FP16(arr[2]);
-    x[3] = WSP_GGML_FP32_TO_FP16(arr[3]);
+    x[0] = WSP_GGML_CPU_FP32_TO_FP16(arr[0]);
+    x[1] = WSP_GGML_CPU_FP32_TO_FP16(arr[1]);
+    x[2] = WSP_GGML_CPU_FP32_TO_FP16(arr[2]);
+    x[3] = WSP_GGML_CPU_FP32_TO_FP16(arr[3]);
 }
 #define WSP_GGML_F32Cx4             __m128
@@ -828,7 +985,7 @@ static inline void __lasx_f32cx8_store(wsp_ggml_fp16_t * x, __m256 y) {
 #define WSP_GGML_F32x4_ZERO    __lsx_vldi(0)
 #define WSP_GGML_F32x4_SET1(x) __lsx_vinsgr2vr_w(__lsx_vldi(0),(x), 0)
 #define WSP_GGML_F32x4_LOAD(x) __lsx_vld((x), 0)
-#define WSP_GGML_F32x4_STORE((x),(y))   __lsx_vst((y), (x), 0)
+#define WSP_GGML_F32x4_STORE(x, y)   __lsx_vst(y, x, 0)
 #define WSP_GGML_F32x4_FMA(a, b, c) __lsx_vfmadd_s(b, c, a)
 #define WSP_GGML_F32x4_ADD     __lsx_vfadd_s
 #define WSP_GGML_F32x4_MUL     __lsx_vfmul_s
@@ -874,10 +1031,10 @@ static inline void __lasx_f32cx8_store(wsp_ggml_fp16_t * x, __m256 y) {
 static inline __m128 __lsx_f16x4_load(const wsp_ggml_fp16_t * x) {
     float tmp[4];
-    tmp[0] = WSP_GGML_FP16_TO_FP32(x[0]);
-    tmp[1] = WSP_GGML_FP16_TO_FP32(x[1]);
-    tmp[2] = WSP_GGML_FP16_TO_FP32(x[2]);
-    tmp[3] = WSP_GGML_FP16_TO_FP32(x[3]);
+    tmp[0] = WSP_GGML_CPU_FP16_TO_FP32(x[0]);
+    tmp[1] = WSP_GGML_CPU_FP16_TO_FP32(x[1]);
+    tmp[2] = WSP_GGML_CPU_FP16_TO_FP32(x[2]);
+    tmp[3] = WSP_GGML_CPU_FP16_TO_FP32(x[3]);
     return __lsx_vld(tmp, 0);
 }
@@ -887,10 +1044,10 @@ static inline void __lsx_f16x4_store(wsp_ggml_fp16_t * x, __m128 y) {
     __lsx_vst(y, arr, 0);
-    x[0] = WSP_GGML_FP32_TO_FP16(arr[0]);
-    x[1] = WSP_GGML_FP32_TO_FP16(arr[1]);
-    x[2] = WSP_GGML_FP32_TO_FP16(arr[2]);
-    x[3] = WSP_GGML_FP32_TO_FP16(arr[3]);
+    x[0] = WSP_GGML_CPU_FP32_TO_FP16(arr[0]);
+    x[1] = WSP_GGML_CPU_FP32_TO_FP16(arr[1]);
+    x[2] = WSP_GGML_CPU_FP32_TO_FP16(arr[2]);
+    x[3] = WSP_GGML_CPU_FP32_TO_FP16(arr[3]);
 }
 #define WSP_GGML_F32Cx4             __m128
@@ -922,7 +1079,7 @@ static inline void __lsx_f16x4_store(wsp_ggml_fp16_t * x, __m128 y) {
 #define WSP_GGML_F32_STEP 32
 #define WSP_GGML_F32_EPR  4
-#define WSP_GGML_F32x4              __vector float
+#define WSP_GGML_F32x4              float32x4_t
 #define WSP_GGML_F32x4_ZERO         vec_splats(0.0f)
 #define WSP_GGML_F32x4_SET1         vec_splats
 #define WSP_GGML_F32x4_LOAD(p)      vec_xl(0, p)
@@ -962,28 +1119,45 @@ static inline void __lsx_f16x4_store(wsp_ggml_fp16_t * x, __m128 y) {
 #define WSP_GGML_F16_STEP WSP_GGML_F32_STEP
 #define WSP_GGML_F16_EPR  WSP_GGML_F32_EPR
-static inline __vector float __lzs_f16cx4_load(const wsp_ggml_fp16_t * x) {
+static inline float32x4_t __lzs_f16cx4_load(const wsp_ggml_fp16_t * x) {
+#if defined(__NNPA__)
+    uint16x8_t v_x = vec_xl(0, (const wsp_ggml_fp16_t *)x);
+    uint16x8_t v_xd = vec_convert_from_fp16(v_x, 0);
+    return vec_extend_to_fp32_hi(v_xd, 0);
+#else
     float tmp[4];
     for (int i = 0; i < 4; i++) {
-        tmp[i] = WSP_GGML_FP16_TO_FP32(x[i]);
+        tmp[i] = WSP_GGML_CPU_FP16_TO_FP32(x[i]);
     }
     // note: keep type-cast here to prevent compiler bugs
     // see: https://github.com/ggml-org/llama.cpp/issues/12846
     return vec_xl(0, (const float *)(tmp));
+#endif
 }
-static inline void __lzs_f16cx4_store(wsp_ggml_fp16_t * x, __vector float y) {
+static inline void __lzs_f16cx4_store(wsp_ggml_fp16_t * x, float32x4_t v_y) {
+#if defined(__NNPA__)
+    float32x4_t v_zero = vec_splats(0.0f);
+    uint16x8_t v_xd = vec_round_from_fp32(v_y, v_zero, 0);
+    uint16x8_t v_x = vec_convert_to_fp16(v_xd, 0);
+    x[0] = vec_extract(v_x, 0);
+    x[1] = vec_extract(v_x, 1);
+    x[2] = vec_extract(v_x, 2);
+    x[3] = vec_extract(v_x, 3);
+#else
     float arr[4];
     // note: keep type-cast here to prevent compiler bugs
     // see: https://github.com/ggml-org/llama.cpp/issues/12846
-    vec_xst(y, 0, (float *)(arr));
+    vec_xst(v_y, 0, (float *)(arr));
     for (int i = 0; i < 4; i++) {
-        x[i] = WSP_GGML_FP32_TO_FP16(arr[i]);
+        x[i] = WSP_GGML_CPU_FP32_TO_FP16(arr[i]);
     }
+#endif
 }
 #define WSP_GGML_F16_VEC                WSP_GGML_F32x4
@@ -1004,3 +1178,7 @@ static inline void __lzs_f16cx4_store(wsp_ggml_fp16_t * x, __vector float y) {
 #define WSP_GGML_F32_ARR (WSP_GGML_F32_STEP/WSP_GGML_F32_EPR)
 #define WSP_GGML_F16_ARR (WSP_GGML_F16_STEP/WSP_GGML_F16_EPR)
 #endif
+#ifdef __cplusplus
+}
+#endif

package/cpp/ggml-cpu/vec.cpp CHANGED Viewed

@@ -219,11 +219,11 @@ void wsp_ggml_vec_dot_f16(int n, float * WSP_GGML_RESTRICT s, size_t bs, wsp_ggm
     // leftovers
     for (int i = np; i < n; ++i) {
-        sumf += (wsp_ggml_float)(WSP_GGML_FP16_TO_FP32(x[i])*WSP_GGML_FP16_TO_FP32(y[i]));
+        sumf += (wsp_ggml_float)(WSP_GGML_CPU_FP16_TO_FP32(x[i])*WSP_GGML_CPU_FP16_TO_FP32(y[i]));
     }
 #else
     for (int i = 0; i < n; ++i) {
-        sumf += (wsp_ggml_float)(WSP_GGML_FP16_TO_FP32(x[i])*WSP_GGML_FP16_TO_FP32(y[i]));
+        sumf += (wsp_ggml_float)(WSP_GGML_CPU_FP16_TO_FP32(x[i])*WSP_GGML_CPU_FP16_TO_FP32(y[i]));
     }
 #endif
@@ -254,6 +254,30 @@ void wsp_ggml_vec_silu_f32(const int n, float * y, const float * x) {
     }
 }
+void wsp_ggml_vec_swiglu_f32(const int n, float * y, const float * x, const float * g) {
+    int i = 0;
+#if defined(__AVX512F__) && defined(__AVX512DQ__)
+    for (; i + 15 < n; i += 16) {
+        _mm512_storeu_ps(y + i, _mm512_mul_ps(wsp_ggml_v_silu(_mm512_loadu_ps(x + i)), _mm512_loadu_ps(g + i)));
+    }
+#elif defined(__AVX2__) && defined(__FMA__)
+    for (; i + 7 < n; i += 8) {
+        _mm256_storeu_ps(y + i, _mm256_mul_ps(wsp_ggml_v_silu(_mm256_loadu_ps(x + i)), _mm256_loadu_ps(g + i)));
+    }
+#elif defined(__SSE2__)
+    for (; i + 3 < n; i += 4) {
+        _mm_storeu_ps(y + i, _mm_mul_ps(wsp_ggml_v_silu(_mm_loadu_ps(x + i)), _mm_loadu_ps(g + i)));
+    }
+#elif defined(__ARM_NEON) && defined(__aarch64__)
+    for (; i + 3 < n; i += 4) {
+        vst1q_f32(y + i, vmulq_f32(wsp_ggml_v_silu(vld1q_f32(x + i)), vld1q_f32(g + i)));
+    }
+#endif
+    for (; i < n; ++i) {
+        y[i] = wsp_ggml_silu_f32(x[i]) * g[i];
+    }
+}
 wsp_ggml_float wsp_ggml_vec_soft_max_f32(const int n, float * y, const float * x, float max) {
     int i = 0;
     wsp_ggml_float sum = 0;