npm - whisper.rn - Versions diffs - 0.4.0-rc.9 → 0.4.0 - Mend

whisper.rn 0.4.0-rc.9 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (183) hide show

package/cpp/ggml-cpu.h ADDED Viewed

@@ -0,0 +1,143 @@
+#pragma once
+#include "ggml.h"
+#include "ggml-backend.h"
+#ifdef  __cplusplus
+extern "C" {
+#endif
+    // the compute plan that needs to be prepared for wsp_ggml_graph_compute()
+    // since https://github.com/ggml-org/ggml/issues/287
+    struct wsp_ggml_cplan {
+        size_t    work_size; // size of work buffer, calculated by `wsp_ggml_graph_plan()`
+        uint8_t * work_data; // work buffer, to be allocated by caller before calling to `wsp_ggml_graph_compute()`
+        int n_threads;
+        struct wsp_ggml_threadpool * threadpool;
+        // abort wsp_ggml_graph_compute when true
+        wsp_ggml_abort_callback abort_callback;
+        void *              abort_callback_data;
+    };
+    // numa strategies
+    enum wsp_ggml_numa_strategy {
+        WSP_GGML_NUMA_STRATEGY_DISABLED   = 0,
+        WSP_GGML_NUMA_STRATEGY_DISTRIBUTE = 1,
+        WSP_GGML_NUMA_STRATEGY_ISOLATE    = 2,
+        WSP_GGML_NUMA_STRATEGY_NUMACTL    = 3,
+        WSP_GGML_NUMA_STRATEGY_MIRROR     = 4,
+        WSP_GGML_NUMA_STRATEGY_COUNT
+    };
+    WSP_GGML_BACKEND_API void    wsp_ggml_numa_init(enum wsp_ggml_numa_strategy numa); // call once for better performance on NUMA systems
+    WSP_GGML_BACKEND_API bool    wsp_ggml_is_numa(void); // true if init detected that system has >1 NUMA node
+    WSP_GGML_BACKEND_API struct wsp_ggml_tensor * wsp_ggml_new_i32(struct wsp_ggml_context * ctx, int32_t value);
+    WSP_GGML_BACKEND_API struct wsp_ggml_tensor * wsp_ggml_new_f32(struct wsp_ggml_context * ctx, float value);
+    WSP_GGML_BACKEND_API struct wsp_ggml_tensor * wsp_ggml_set_i32 (struct wsp_ggml_tensor * tensor, int32_t value);
+    WSP_GGML_BACKEND_API struct wsp_ggml_tensor * wsp_ggml_set_f32 (struct wsp_ggml_tensor * tensor, float value);
+    WSP_GGML_BACKEND_API int32_t wsp_ggml_get_i32_1d(const struct wsp_ggml_tensor * tensor, int i);
+    WSP_GGML_BACKEND_API void    wsp_ggml_set_i32_1d(const struct wsp_ggml_tensor * tensor, int i, int32_t value);
+    WSP_GGML_BACKEND_API int32_t wsp_ggml_get_i32_nd(const struct wsp_ggml_tensor * tensor, int i0, int i1, int i2, int i3);
+    WSP_GGML_BACKEND_API void    wsp_ggml_set_i32_nd(const struct wsp_ggml_tensor * tensor, int i0, int i1, int i2, int i3, int32_t value);
+    WSP_GGML_BACKEND_API float   wsp_ggml_get_f32_1d(const struct wsp_ggml_tensor * tensor, int i);
+    WSP_GGML_BACKEND_API void    wsp_ggml_set_f32_1d(const struct wsp_ggml_tensor * tensor, int i, float value);
+    WSP_GGML_BACKEND_API float   wsp_ggml_get_f32_nd(const struct wsp_ggml_tensor * tensor, int i0, int i1, int i2, int i3);
+    WSP_GGML_BACKEND_API void    wsp_ggml_set_f32_nd(const struct wsp_ggml_tensor * tensor, int i0, int i1, int i2, int i3, float value);
+    WSP_GGML_BACKEND_API struct wsp_ggml_threadpool *      wsp_ggml_threadpool_new           (struct wsp_ggml_threadpool_params  * params);
+    WSP_GGML_BACKEND_API void                          wsp_ggml_threadpool_free          (struct wsp_ggml_threadpool * threadpool);
+    WSP_GGML_BACKEND_API int                           wsp_ggml_threadpool_get_n_threads (struct wsp_ggml_threadpool * threadpool);
+    WSP_GGML_BACKEND_API void                          wsp_ggml_threadpool_pause         (struct wsp_ggml_threadpool * threadpool);
+    WSP_GGML_BACKEND_API void                          wsp_ggml_threadpool_resume        (struct wsp_ggml_threadpool * threadpool);
+    // wsp_ggml_graph_plan() has to be called before wsp_ggml_graph_compute()
+    // when plan.work_size > 0, caller must allocate memory for plan.work_data
+    WSP_GGML_BACKEND_API struct wsp_ggml_cplan wsp_ggml_graph_plan(
+                  const struct wsp_ggml_cgraph * cgraph,
+                                       int   n_threads, /* = WSP_GGML_DEFAULT_N_THREADS */
+                    struct wsp_ggml_threadpool * threadpool /* = NULL */ );
+    WSP_GGML_BACKEND_API enum wsp_ggml_status  wsp_ggml_graph_compute(struct wsp_ggml_cgraph * cgraph, struct wsp_ggml_cplan * cplan);
+    // same as wsp_ggml_graph_compute() but the work data is allocated as a part of the context
+    // note: the drawback of this API is that you must have ensured that the context has enough memory for the work data
+    WSP_GGML_BACKEND_API enum wsp_ggml_status  wsp_ggml_graph_compute_with_ctx(struct wsp_ggml_context * ctx, struct wsp_ggml_cgraph * cgraph, int n_threads);
+    //
+    // system info
+    //
+    // x86
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_sse3       (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_ssse3      (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_avx        (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_avx_vnni   (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_avx2       (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_bmi2       (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_f16c       (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_fma        (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_avx512     (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_avx512_vbmi(void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_avx512_vnni(void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_avx512_bf16(void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_amx_int8   (void);
+    // ARM
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_neon       (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_arm_fma    (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_fp16_va    (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_dotprod    (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_matmul_int8(void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_sve        (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_get_sve_cnt    (void);  // sve vector length in bytes
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_sme        (void);
+    // other
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_riscv_v    (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_vsx        (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_vxe        (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_wasm_simd  (void);
+    WSP_GGML_BACKEND_API int wsp_ggml_cpu_has_llamafile  (void);
+    // Internal types and functions exposed for tests and benchmarks
+    typedef void (*wsp_ggml_vec_dot_t)  (int n, float * WSP_GGML_RESTRICT s, size_t bs, const void * WSP_GGML_RESTRICT x, size_t bx,
+                                       const void * WSP_GGML_RESTRICT y, size_t by, int nrc);
+    struct wsp_ggml_type_traits_cpu {
+        wsp_ggml_from_float_t        from_float;
+        wsp_ggml_vec_dot_t           vec_dot;
+        enum wsp_ggml_type           vec_dot_type;
+        int64_t                  nrows; // number of rows to process simultaneously
+    };
+    WSP_GGML_BACKEND_API const struct wsp_ggml_type_traits_cpu * wsp_ggml_get_type_traits_cpu(enum wsp_ggml_type type);
+    WSP_GGML_BACKEND_API void wsp_ggml_cpu_init(void);
+    //
+    // CPU backend
+    //
+    WSP_GGML_BACKEND_API wsp_ggml_backend_t wsp_ggml_backend_cpu_init(void);
+    WSP_GGML_BACKEND_API bool wsp_ggml_backend_is_cpu                (wsp_ggml_backend_t backend);
+    WSP_GGML_BACKEND_API void wsp_ggml_backend_cpu_set_n_threads     (wsp_ggml_backend_t backend_cpu, int n_threads);
+    WSP_GGML_BACKEND_API void wsp_ggml_backend_cpu_set_threadpool    (wsp_ggml_backend_t backend_cpu, wsp_ggml_threadpool_t threadpool);
+    WSP_GGML_BACKEND_API void wsp_ggml_backend_cpu_set_abort_callback(wsp_ggml_backend_t backend_cpu, wsp_ggml_abort_callback abort_callback, void * abort_callback_data);
+    WSP_GGML_BACKEND_API wsp_ggml_backend_reg_t wsp_ggml_backend_cpu_reg(void);
+    WSP_GGML_BACKEND_API void wsp_ggml_cpu_fp32_to_fp16(const float *, wsp_ggml_fp16_t *, int64_t);
+    WSP_GGML_BACKEND_API void wsp_ggml_cpu_fp16_to_fp32(const wsp_ggml_fp16_t *, float *, int64_t);
+    WSP_GGML_BACKEND_API void wsp_ggml_cpu_fp32_to_bf16(const float *, wsp_ggml_bf16_t *, int64_t);
+    WSP_GGML_BACKEND_API void wsp_ggml_cpu_bf16_to_fp32(const wsp_ggml_bf16_t *, float *, int64_t);
+#ifdef __cplusplus
+}
+#endif

package/cpp/ggml-impl.h CHANGED Viewed

@@ -3,21 +3,44 @@
 // GGML internal header
 #include "ggml.h"
+#include "gguf.h"
 #include <assert.h>
+#include <math.h>
 #include <stdlib.h> // load `stdlib.h` before other headers to work around MinGW bug: https://sourceforge.net/p/mingw-w64/bugs/192/
 #include <stdbool.h>
 #include <stdint.h>
+#include <string.h>
+#ifdef __ARM_FEATURE_SVE
+#include <arm_sve.h>
+#endif // __ARM_FEATURE_SVE
+#if defined(__ARM_NEON) && !defined(__CUDACC__) && !defined(__MUSACC__)
+// if YCM cannot find <arm_neon.h>, make a symbolic link to it, for example:
+//
+//   $ ln -sfn /Library/Developer/CommandLineTools/usr/lib/clang/13.1.6/include/arm_neon.h ./src/
+//
+#include <arm_neon.h>
+#endif
+#if defined(__F16C__)
+#include <immintrin.h>
+#endif
 #ifdef __cplusplus
 extern "C" {
 #endif
-#undef MIN
-#undef MAX
+void wsp_ggml_print_backtrace(void);
+#ifndef MIN
+#    define MIN(a, b) ((a) < (b) ? (a) : (b))
+#endif
-#define MIN(a, b) ((a) < (b) ? (a) : (b))
-#define MAX(a, b) ((a) > (b) ? (a) : (b))
+#ifndef MAX
+#    define MAX(a, b) ((a) > (b) ? (a) : (b))
+#endif
 // required for mmap as gguf only guarantees 32-byte alignment
 #define TENSOR_ALIGNMENT 32
@@ -27,22 +50,36 @@ extern "C" {
 // if C99 - static_assert is noop
 // ref: https://stackoverflow.com/a/53923785/4039976
 #ifndef __cplusplus
-#ifndef static_assert
-#if defined(__STDC_VERSION__) && (__STDC_VERSION__ >= 201100L)
-#define static_assert(cond, msg) _Static_assert(cond, msg)
-#else
-#define static_assert(cond, msg) struct global_scope_noop_trick
-#endif
-#endif
+    #ifndef static_assert
+        #if defined(__STDC_VERSION__) && (__STDC_VERSION__ >= 201100L)
+            #define static_assert(cond, msg) _Static_assert(cond, msg)
+        #else
+            #define static_assert(cond, msg) struct global_scope_noop_trick
+        #endif
+    #endif
 #endif
+static inline int wsp_ggml_up32(int n) {
+    return (n + 31) & ~31;
+}
+//static inline int wsp_ggml_up64(int n) {
+//    return (n + 63) & ~63;
+//}
+static inline int wsp_ggml_up(int n, int m) {
+    // assert m is a power of 2
+    WSP_GGML_ASSERT((m & (m - 1)) == 0);
+    return (n + m - 1) & ~(m - 1);
+}
 //
 // logging
 //
 WSP_GGML_ATTRIBUTE_FORMAT(2, 3)
-void wsp_ggml_log_internal        (enum wsp_ggml_log_level level, const char * format, ...);
-void wsp_ggml_log_callback_default(enum wsp_ggml_log_level level, const char * text, void * user_data);
+WSP_GGML_API void wsp_ggml_log_internal        (enum wsp_ggml_log_level level, const char * format, ...);
+WSP_GGML_API void wsp_ggml_log_callback_default(enum wsp_ggml_log_level level, const char * text, void * user_data);
 #define WSP_GGML_LOG(...)       wsp_ggml_log_internal(WSP_GGML_LOG_LEVEL_NONE , __VA_ARGS__)
 #define WSP_GGML_LOG_INFO(...)  wsp_ggml_log_internal(WSP_GGML_LOG_LEVEL_INFO , __VA_ARGS__)
@@ -51,6 +88,78 @@ void wsp_ggml_log_callback_default(enum wsp_ggml_log_level level, const char * t
 #define WSP_GGML_LOG_DEBUG(...) wsp_ggml_log_internal(WSP_GGML_LOG_LEVEL_DEBUG, __VA_ARGS__)
 #define WSP_GGML_LOG_CONT(...)  wsp_ggml_log_internal(WSP_GGML_LOG_LEVEL_CONT , __VA_ARGS__)
+#define WSP_GGML_DEBUG 0
+#if (WSP_GGML_DEBUG >= 1)
+#define WSP_GGML_PRINT_DEBUG(...) WSP_GGML_LOG_DEBUG(__VA_ARGS__)
+#else
+#define WSP_GGML_PRINT_DEBUG(...)
+#endif
+#if (WSP_GGML_DEBUG >= 5)
+#define WSP_GGML_PRINT_DEBUG_5(...) WSP_GGML_LOG_DEBUG(__VA_ARGS__)
+#else
+#define WSP_GGML_PRINT_DEBUG_5(...)
+#endif
+#if (WSP_GGML_DEBUG >= 10)
+#define WSP_GGML_PRINT_DEBUG_10(...) WSP_GGML_LOG_DEBUG(__VA_ARGS__)
+#else
+#define WSP_GGML_PRINT_DEBUG_10(...)
+#endif
+// tensor params
+static void wsp_ggml_set_op_params(struct wsp_ggml_tensor * tensor, const void * params, size_t params_size) {
+    WSP_GGML_ASSERT(tensor != NULL); // silence -Warray-bounds warnings
+    assert(params_size <= WSP_GGML_MAX_OP_PARAMS);
+    memcpy(tensor->op_params, params, params_size);
+}
+static int32_t wsp_ggml_get_op_params_i32(const struct wsp_ggml_tensor * tensor, uint32_t i) {
+    assert(i < WSP_GGML_MAX_OP_PARAMS / sizeof(int32_t));
+    return ((const int32_t *)(tensor->op_params))[i];
+}
+static float wsp_ggml_get_op_params_f32(const struct wsp_ggml_tensor * tensor, uint32_t i) {
+    assert(i < WSP_GGML_MAX_OP_PARAMS / sizeof(float));
+    return ((const float *)(tensor->op_params))[i];
+}
+static void wsp_ggml_set_op_params_i32(struct wsp_ggml_tensor * tensor, uint32_t i, int32_t value) {
+    assert(i < WSP_GGML_MAX_OP_PARAMS / sizeof(int32_t));
+    ((int32_t *)(tensor->op_params))[i] = value;
+}
+static void wsp_ggml_set_op_params_f32(struct wsp_ggml_tensor * tensor, uint32_t i, float value) {
+    assert(i < WSP_GGML_MAX_OP_PARAMS / sizeof(float));
+    ((float *)(tensor->op_params))[i] = value;
+}
+struct wsp_ggml_map_custom1_op_params {
+    wsp_ggml_custom1_op_t  fun;
+    int                n_tasks;
+    void             * userdata;
+};
+struct wsp_ggml_map_custom2_op_params {
+    wsp_ggml_custom2_op_t   fun;
+    int                 n_tasks;
+    void              * userdata;
+};
+struct wsp_ggml_map_custom3_op_params {
+    wsp_ggml_custom3_op_t fun;
+    int               n_tasks;
+    void            * userdata;
+};
+struct wsp_ggml_custom_op_params {
+    wsp_ggml_custom_op_t fun;
+    int              n_tasks;
+    void           * userdata;
+};
 // bitset
 typedef uint32_t wsp_ggml_bitset_t;
@@ -99,7 +208,7 @@ void wsp_ggml_hash_set_reset(struct wsp_ggml_hash_set * hash_set);
 static bool wsp_ggml_hash_contains(const struct wsp_ggml_hash_set * hash_set, struct wsp_ggml_tensor * key);
 // returns WSP_GGML_HASHSET_FULL if table is full, otherwise the current index of the key or where it should be inserted
-static size_t wsp_ggml_hash_find(const struct wsp_ggml_hash_set * hash_set, struct wsp_ggml_tensor * key);
+static size_t wsp_ggml_hash_find(const struct wsp_ggml_hash_set * hash_set, const struct wsp_ggml_tensor * key);
 // returns WSP_GGML_HASHSET_ALREADY_EXISTS if key already exists, index otherwise, asserts if table is full
 static size_t wsp_ggml_hash_insert(struct wsp_ggml_hash_set * hash_set, struct wsp_ggml_tensor * key);
@@ -113,7 +222,7 @@ static inline size_t wsp_ggml_hash(const struct wsp_ggml_tensor * p) {
     return (size_t)(uintptr_t)p >> 4;
 }
-static size_t wsp_ggml_hash_find(const struct wsp_ggml_hash_set * hash_set, struct wsp_ggml_tensor * key) {
+static size_t wsp_ggml_hash_find(const struct wsp_ggml_hash_set * hash_set, const struct wsp_ggml_tensor * key) {
     size_t h = wsp_ggml_hash(key) % hash_set->size;
     // linear probing
@@ -184,26 +293,311 @@ enum wsp_ggml_cgraph_eval_order {
 };
 struct wsp_ggml_cgraph {
-    int size;
-    int n_nodes;
-    int n_leafs;
+    int size;    // maximum number of nodes/leafs/grads/grad_accs
+    int n_nodes; // number of nodes currently in use
+    int n_leafs; // number of leafs currently in use
-    struct wsp_ggml_tensor ** nodes;
-    struct wsp_ggml_tensor ** grads;
-    struct wsp_ggml_tensor ** leafs;
+    struct wsp_ggml_tensor ** nodes;     // tensors with data that can change if the graph is evaluated
+    struct wsp_ggml_tensor ** grads;     // the outputs of these tensors are the gradients of the nodes
+    struct wsp_ggml_tensor ** grad_accs; // accumulators for node gradients
+    struct wsp_ggml_tensor ** leafs;     // tensors with constant data
     struct wsp_ggml_hash_set visited_hash_set;
     enum wsp_ggml_cgraph_eval_order order;
 };
+// returns a slice of cgraph with nodes [i0, i1)
+// the slice does not have leafs or gradients
+// if you need the gradients, get them from the original graph
 struct wsp_ggml_cgraph wsp_ggml_graph_view(struct wsp_ggml_cgraph * cgraph, int i0, int i1);
 // Memory allocation
-void * wsp_ggml_aligned_malloc(size_t size);
-void wsp_ggml_aligned_free(void * ptr, size_t size);
+WSP_GGML_API void * wsp_ggml_aligned_malloc(size_t size);
+WSP_GGML_API void wsp_ggml_aligned_free(void * ptr, size_t size);
+// FP16 to FP32 conversion
+// 16-bit float
+// on Arm, we use __fp16
+// on x86, we use uint16_t
+//
+// for old CUDA compilers (<= 11), we use uint16_t: ref https://github.com/ggml-org/llama.cpp/pull/10616
+// for     MUSA compilers        , we use uint16_t: ref https://github.com/ggml-org/llama.cpp/pull/11843
+//
+#if defined(__ARM_NEON) && !(defined(__CUDACC__) && __CUDACC_VER_MAJOR__ <= 11) && !defined(__MUSACC__)
+    #define WSP_GGML_COMPUTE_FP16_TO_FP32(x) wsp_ggml_compute_fp16_to_fp32(x)
+    #define WSP_GGML_COMPUTE_FP32_TO_FP16(x) wsp_ggml_compute_fp32_to_fp16(x)
+    #define WSP_GGML_FP16_TO_FP32(x) wsp_ggml_compute_fp16_to_fp32(x)
+    static inline float wsp_ggml_compute_fp16_to_fp32(wsp_ggml_fp16_t h) {
+        __fp16 tmp;
+        memcpy(&tmp, &h, sizeof(wsp_ggml_fp16_t));
+        return (float)tmp;
+    }
+    static inline wsp_ggml_fp16_t wsp_ggml_compute_fp32_to_fp16(float f) {
+        wsp_ggml_fp16_t res;
+        __fp16 tmp = f;
+        memcpy(&res, &tmp, sizeof(wsp_ggml_fp16_t));
+        return res;
+    }
+#elif defined(__F16C__)
+    #ifdef _MSC_VER
+        #define WSP_GGML_COMPUTE_FP16_TO_FP32(x) _mm_cvtss_f32(_mm_cvtph_ps(_mm_cvtsi32_si128(x)))
+        #define WSP_GGML_COMPUTE_FP32_TO_FP16(x) _mm_extract_epi16(_mm_cvtps_ph(_mm_set_ss(x), 0), 0)
+    #else
+        #define WSP_GGML_COMPUTE_FP16_TO_FP32(x) _cvtsh_ss(x)
+        #define WSP_GGML_COMPUTE_FP32_TO_FP16(x) _cvtss_sh(x, 0)
+    #endif
+#elif defined(__POWER9_VECTOR__)
+    #define WSP_GGML_COMPUTE_FP16_TO_FP32(x) wsp_ggml_compute_fp16_to_fp32(x)
+    #define WSP_GGML_COMPUTE_FP32_TO_FP16(x) wsp_ggml_compute_fp32_to_fp16(x)
+    /* the inline asm below is about 12% faster than the lookup method */
+    #define WSP_GGML_FP16_TO_FP32(x) WSP_GGML_COMPUTE_FP16_TO_FP32(x)
+    #define WSP_GGML_FP32_TO_FP16(x) WSP_GGML_COMPUTE_FP32_TO_FP16(x)
+    static inline float wsp_ggml_compute_fp16_to_fp32(wsp_ggml_fp16_t h) {
+        float f;
+        double d;
+        __asm__(
+            "mtfprd %0,%2\n"
+            "xscvhpdp %0,%0\n"
+            "frsp %1,%0\n" :
+            /* temp */ "=d"(d),
+            /* out */  "=f"(f):
+            /* in */   "r"(h));
+        return f;
+    }
+    static inline wsp_ggml_fp16_t wsp_ggml_compute_fp32_to_fp16(float f) {
+        double d;
+        wsp_ggml_fp16_t r;
+        __asm__( /* xscvdphp can work on double or single precision */
+            "xscvdphp %0,%2\n"
+            "mffprd %1,%0\n" :
+            /* temp */ "=d"(d),
+            /* out */  "=r"(r):
+            /* in */   "f"(f));
+        return r;
+    }
+#elif defined(__riscv) && defined(__riscv_zfhmin)
+    static inline float wsp_ggml_compute_fp16_to_fp32(wsp_ggml_fp16_t h) {
+        float f;
+        __asm__(
+            "fmv.h.x %[f], %[h]\n\t"
+            "fcvt.s.h %[f], %[f]"
+            : [f] "=&f" (f)
+            : [h] "r" (h)
+        );
+        return f;
+    }
+    static inline wsp_ggml_fp16_t wsp_ggml_compute_fp32_to_fp16(float f) {
+        wsp_ggml_fp16_t res;
+        __asm__(
+            "fcvt.h.s %[f], %[f]\n\t"
+            "fmv.x.h %[h], %[f]"
+            : [h] "=&r" (res)
+            : [f] "f" (f)
+        );
+        return res;
+    }
+    #define WSP_GGML_COMPUTE_FP16_TO_FP32(x) wsp_ggml_compute_fp16_to_fp32(x)
+    #define WSP_GGML_COMPUTE_FP32_TO_FP16(x) wsp_ggml_compute_fp32_to_fp16(x)
+    #define WSP_GGML_FP16_TO_FP32(x) WSP_GGML_COMPUTE_FP16_TO_FP32(x)
+    #define WSP_GGML_FP32_TO_FP16(x) WSP_GGML_COMPUTE_FP32_TO_FP16(x)
+#else
+    // FP16 <-> FP32
+    // ref: https://github.com/Maratyszcza/FP16
+    static inline float fp32_from_bits(uint32_t w) {
+        union {
+            uint32_t as_bits;
+            float as_value;
+        } fp32;
+        fp32.as_bits = w;
+        return fp32.as_value;
+    }
+    static inline uint32_t fp32_to_bits(float f) {
+        union {
+            float as_value;
+            uint32_t as_bits;
+        } fp32;
+        fp32.as_value = f;
+        return fp32.as_bits;
+    }
+    static inline float wsp_ggml_compute_fp16_to_fp32(wsp_ggml_fp16_t h) {
+        const uint32_t w = (uint32_t) h << 16;
+        const uint32_t sign = w & UINT32_C(0x80000000);
+        const uint32_t two_w = w + w;
+        const uint32_t exp_offset = UINT32_C(0xE0) << 23;
+    #if (defined(__STDC_VERSION__) && (__STDC_VERSION__ >= 199901L) || defined(__GNUC__) && !defined(__STRICT_ANSI__)) && (!defined(__cplusplus) || __cplusplus >= 201703L)
+        const float exp_scale = 0x1.0p-112f;
+    #else
+        const float exp_scale = fp32_from_bits(UINT32_C(0x7800000));
+    #endif
+        const float normalized_value = fp32_from_bits((two_w >> 4) + exp_offset) * exp_scale;
+        const uint32_t magic_mask = UINT32_C(126) << 23;
+        const float magic_bias = 0.5f;
+        const float denormalized_value = fp32_from_bits((two_w >> 17) | magic_mask) - magic_bias;
+        const uint32_t denormalized_cutoff = UINT32_C(1) << 27;
+        const uint32_t result = sign |
+            (two_w < denormalized_cutoff ? fp32_to_bits(denormalized_value) : fp32_to_bits(normalized_value));
+        return fp32_from_bits(result);
+    }
+    static inline wsp_ggml_fp16_t wsp_ggml_compute_fp32_to_fp16(float f) {
+    #if (defined(__STDC_VERSION__) && (__STDC_VERSION__ >= 199901L) || defined(__GNUC__) && !defined(__STRICT_ANSI__)) && (!defined(__cplusplus) || __cplusplus >= 201703L)
+        const float scale_to_inf = 0x1.0p+112f;
+        const float scale_to_zero = 0x1.0p-110f;
+    #else
+        const float scale_to_inf = fp32_from_bits(UINT32_C(0x77800000));
+        const float scale_to_zero = fp32_from_bits(UINT32_C(0x08800000));
+    #endif
+        float base = (fabsf(f) * scale_to_inf) * scale_to_zero;
+        const uint32_t w = fp32_to_bits(f);
+        const uint32_t shl1_w = w + w;
+        const uint32_t sign = w & UINT32_C(0x80000000);
+        uint32_t bias = shl1_w & UINT32_C(0xFF000000);
+        if (bias < UINT32_C(0x71000000)) {
+            bias = UINT32_C(0x71000000);
+        }
+        base = fp32_from_bits((bias >> 1) + UINT32_C(0x07800000)) + base;
+        const uint32_t bits = fp32_to_bits(base);
+        const uint32_t exp_bits = (bits >> 13) & UINT32_C(0x00007C00);
+        const uint32_t mantissa_bits = bits & UINT32_C(0x00000FFF);
+        const uint32_t nonsign = exp_bits + mantissa_bits;
+        return (sign >> 16) | (shl1_w > UINT32_C(0xFF000000) ? UINT16_C(0x7E00) : nonsign);
+    }
+    #define WSP_GGML_COMPUTE_FP16_TO_FP32(x) wsp_ggml_compute_fp16_to_fp32(x)
+    #define WSP_GGML_COMPUTE_FP32_TO_FP16(x) wsp_ggml_compute_fp32_to_fp16(x)
+#endif // defined(__ARM_NEON) && !(defined(__CUDACC__) && __CUDACC_VER_MAJOR__ <= 11) && !defined(__MUSACC__)
+// precomputed f32 table for f16 (256 KB)
+// defined in ggml.c, initialized in wsp_ggml_init()
+WSP_GGML_API float wsp_ggml_table_f32_f16[1 << 16];
+// On ARM NEON, it's quicker to directly convert x -> x instead of calling into wsp_ggml_lookup_fp16_to_fp32,
+// so we define WSP_GGML_FP16_TO_FP32 and WSP_GGML_FP32_TO_FP16 elsewhere for NEON.
+// This is also true for POWER9.
+#if !defined(WSP_GGML_FP16_TO_FP32)
+inline static float wsp_ggml_lookup_fp16_to_fp32(wsp_ggml_fp16_t f) {
+    uint16_t s;
+    memcpy(&s, &f, sizeof(uint16_t));
+    return wsp_ggml_table_f32_f16[s];
+}
+#define WSP_GGML_FP16_TO_FP32(x) wsp_ggml_lookup_fp16_to_fp32(x)
+#endif
+#if !defined(WSP_GGML_FP32_TO_FP16)
+#define WSP_GGML_FP32_TO_FP16(x) WSP_GGML_COMPUTE_FP32_TO_FP16(x)
+#endif
+/**
+ * Converts brain16 to float32.
+ *
+ * The bfloat16 floating point format has the following structure:
+ *
+ *       ┌sign
+ *       │
+ *       │   ┌exponent
+ *       │   │
+ *       │   │      ┌mantissa
+ *       │   │      │
+ *       │┌──┴───┐┌─┴───┐
+ *     0b0000000000000000 brain16
+ *
+ * Since bf16 has the same number of exponent bits as a 32bit float,
+ * encoding and decoding numbers becomes relatively straightforward.
+ *
+ *       ┌sign
+ *       │
+ *       │   ┌exponent
+ *       │   │
+ *       │   │      ┌mantissa
+ *       │   │      │
+ *       │┌──┴───┐┌─┴───────────────────┐
+ *     0b00000000000000000000000000000000 IEEE binary32
+ *
+ * For comparison, the standard fp16 format has fewer exponent bits.
+ *
+ *       ┌sign
+ *       │
+ *       │  ┌exponent
+ *       │  │
+ *       │  │    ┌mantissa
+ *       │  │    │
+ *       │┌─┴─┐┌─┴──────┐
+ *     0b0000000000000000 IEEE binary16
+ *
+ * @see IEEE 754-2008
+ */
+static inline float wsp_ggml_compute_bf16_to_fp32(wsp_ggml_bf16_t h) {
+    union {
+        float f;
+        uint32_t i;
+    } u;
+    u.i = (uint32_t)h.bits << 16;
+    return u.f;
+}
+/**
+ * Converts float32 to brain16.
+ *
+ * This is binary identical with Google Brain float conversion.
+ * Floats shall round to nearest even, and NANs shall be quiet.
+ * Subnormals aren't flushed to zero, except perhaps when used.
+ * This code should vectorize nicely if using modern compilers.
+ */
+static inline wsp_ggml_bf16_t wsp_ggml_compute_fp32_to_bf16(float s) {
+    wsp_ggml_bf16_t h;
+    union {
+        float f;
+        uint32_t i;
+    } u;
+    u.f = s;
+    if ((u.i & 0x7fffffff) > 0x7f800000) { /* nan */
+        h.bits = (u.i >> 16) | 64; /* force to quiet */
+        return h;
+    }
+    h.bits = (u.i + (0x7fff + ((u.i >> 16) & 1))) >> 16;
+    return h;
+}
+#define WSP_GGML_FP32_TO_BF16(x) wsp_ggml_compute_fp32_to_bf16(x)
+#define WSP_GGML_BF16_TO_FP32(x) wsp_ggml_compute_bf16_to_fp32(x)
 #ifdef __cplusplus
 }
 #endif
+#ifdef __cplusplus
+#include <vector>
+// expose GGUF internals for test code
+WSP_GGML_API size_t wsp_gguf_type_size(enum wsp_gguf_type type);
+WSP_GGML_API struct wsp_gguf_context * wsp_gguf_init_from_file_impl(FILE * file, struct wsp_gguf_init_params params);
+WSP_GGML_API void wsp_gguf_write_to_buf(const struct wsp_gguf_context * ctx, std::vector<int8_t> & buf, bool only_meta);
+#endif // __cplusplus