RubyGems - whispercpp - Versions diffs - 1.2.0.2 → 1.3.1 - Mend

whispercpp 1.2.0.2 → 1.3.1

Files changed (135) hide show

checksums.yaml +4 -4
data/.gitignore +5 -0
data/LICENSE +1 -1
data/README.md +165 -434
data/Rakefile +46 -86
data/ext/.gitignore +13 -0
data/ext/cpu.mk +9 -0
data/ext/{dr_wav.h → examples/dr_wav.h} +3560 -1179
data/ext/extconf.rb +185 -7
data/ext/ggml/include/ggml-alloc.h +76 -0
data/ext/ggml/include/ggml-backend.h +352 -0
data/ext/ggml/include/ggml-blas.h +25 -0
data/ext/ggml/include/ggml-cann.h +123 -0
data/ext/ggml/include/ggml-cpp.h +38 -0
data/ext/ggml/include/ggml-cpu.h +135 -0
data/ext/ggml/include/ggml-cuda.h +47 -0
data/ext/ggml/include/ggml-kompute.h +50 -0
data/ext/ggml/include/ggml-metal.h +66 -0
data/ext/ggml/include/ggml-opencl.h +26 -0
data/ext/ggml/include/ggml-opt.h +216 -0
data/ext/ggml/include/ggml-rpc.h +28 -0
data/ext/ggml/include/ggml-sycl.h +49 -0
data/ext/ggml/include/ggml-vulkan.h +31 -0
data/ext/ggml/include/ggml.h +2285 -0
data/ext/ggml/src/ggml-alloc.c +1037 -0
data/ext/ggml/src/ggml-amx/common.h +94 -0
data/ext/ggml/src/ggml-amx/ggml-amx.cpp +446 -0
data/ext/ggml/src/ggml-amx/mmq.cpp +2510 -0
data/ext/ggml/src/ggml-amx/mmq.h +17 -0
data/ext/ggml/src/ggml-backend-impl.h +256 -0
data/ext/ggml/src/ggml-backend-reg.cpp +552 -0
data/ext/ggml/src/ggml-backend.cpp +1999 -0
data/ext/ggml/src/ggml-blas/ggml-blas.cpp +517 -0
data/ext/ggml/src/ggml-cann/acl_tensor.cpp +175 -0
data/ext/ggml/src/ggml-cann/acl_tensor.h +258 -0
data/ext/ggml/src/ggml-cann/aclnn_ops.cpp +3427 -0
data/ext/ggml/src/ggml-cann/aclnn_ops.h +592 -0
data/ext/ggml/src/ggml-cann/common.h +286 -0
data/ext/ggml/src/ggml-cann/ggml-cann.cpp +2188 -0
data/ext/ggml/src/ggml-cann/kernels/ascendc_kernels.h +19 -0
data/ext/ggml/src/ggml-cann/kernels/dup.cpp +236 -0
data/ext/ggml/src/ggml-cann/kernels/get_row_f16.cpp +197 -0
data/ext/ggml/src/ggml-cann/kernels/get_row_f32.cpp +190 -0
data/ext/ggml/src/ggml-cann/kernels/get_row_q4_0.cpp +204 -0
data/ext/ggml/src/ggml-cann/kernels/get_row_q8_0.cpp +191 -0
data/ext/ggml/src/ggml-cann/kernels/quantize_f16_q8_0.cpp +218 -0
data/ext/ggml/src/ggml-cann/kernels/quantize_f32_q8_0.cpp +216 -0
data/ext/ggml/src/ggml-cann/kernels/quantize_float_to_q4_0.cpp +295 -0
data/ext/ggml/src/ggml-common.h +1853 -0
data/ext/ggml/src/ggml-cpu/amx/amx.cpp +220 -0
data/ext/ggml/src/ggml-cpu/amx/amx.h +8 -0
data/ext/ggml/src/ggml-cpu/amx/common.h +91 -0
data/ext/ggml/src/ggml-cpu/amx/mmq.cpp +2511 -0
data/ext/ggml/src/ggml-cpu/amx/mmq.h +10 -0
data/ext/ggml/src/ggml-cpu/cpu-feats-x86.cpp +323 -0
data/ext/ggml/src/ggml-cpu/ggml-cpu-aarch64.cpp +4262 -0
data/ext/ggml/src/ggml-cpu/ggml-cpu-aarch64.h +8 -0
data/ext/ggml/src/ggml-cpu/ggml-cpu-hbm.cpp +55 -0
data/ext/ggml/src/ggml-cpu/ggml-cpu-hbm.h +8 -0
data/ext/ggml/src/ggml-cpu/ggml-cpu-impl.h +386 -0
data/ext/ggml/src/ggml-cpu/ggml-cpu-quants.c +10835 -0
data/ext/ggml/src/ggml-cpu/ggml-cpu-quants.h +63 -0
data/ext/ggml/src/ggml-cpu/ggml-cpu-traits.cpp +36 -0
data/ext/ggml/src/ggml-cpu/ggml-cpu-traits.h +38 -0
data/ext/ggml/src/ggml-cpu/ggml-cpu.c +14123 -0
data/ext/ggml/src/ggml-cpu/ggml-cpu.cpp +622 -0
data/ext/ggml/src/ggml-cpu/llamafile/sgemm.cpp +1884 -0
data/ext/ggml/src/ggml-cpu/llamafile/sgemm.h +14 -0
data/ext/ggml/src/ggml-cuda/vendors/cuda.h +14 -0
data/ext/ggml/src/ggml-cuda/vendors/hip.h +186 -0
data/ext/ggml/src/ggml-cuda/vendors/musa.h +134 -0
data/ext/ggml/src/ggml-impl.h +556 -0
data/ext/ggml/src/ggml-kompute/ggml-kompute.cpp +2251 -0
data/ext/ggml/src/ggml-metal/ggml-metal-impl.h +288 -0
data/ext/ggml/src/ggml-metal/ggml-metal.m +4884 -0
data/ext/ggml/src/ggml-metal/ggml-metal.metal +6732 -0
data/ext/ggml/src/ggml-opt.cpp +854 -0
data/ext/ggml/src/ggml-quants.c +5238 -0
data/ext/ggml/src/ggml-quants.h +100 -0
data/ext/ggml/src/ggml-rpc/ggml-rpc.cpp +1406 -0
data/ext/ggml/src/ggml-sycl/common.cpp +95 -0
data/ext/ggml/src/ggml-sycl/concat.cpp +196 -0
data/ext/ggml/src/ggml-sycl/conv.cpp +99 -0
data/ext/ggml/src/ggml-sycl/convert.cpp +547 -0
data/ext/ggml/src/ggml-sycl/dmmv.cpp +1023 -0
data/ext/ggml/src/ggml-sycl/element_wise.cpp +1030 -0
data/ext/ggml/src/ggml-sycl/ggml-sycl.cpp +4729 -0
data/ext/ggml/src/ggml-sycl/im2col.cpp +126 -0
data/ext/ggml/src/ggml-sycl/mmq.cpp +3031 -0
data/ext/ggml/src/ggml-sycl/mmvq.cpp +1015 -0
data/ext/ggml/src/ggml-sycl/norm.cpp +378 -0
data/ext/ggml/src/ggml-sycl/outprod.cpp +56 -0
data/ext/ggml/src/ggml-sycl/rope.cpp +276 -0
data/ext/ggml/src/ggml-sycl/softmax.cpp +251 -0
data/ext/ggml/src/ggml-sycl/tsembd.cpp +72 -0
data/ext/ggml/src/ggml-sycl/wkv6.cpp +141 -0
data/ext/ggml/src/ggml-threading.cpp +12 -0
data/ext/ggml/src/ggml-threading.h +14 -0
data/ext/ggml/src/ggml-vulkan/ggml-vulkan.cpp +8657 -0
data/ext/ggml/src/ggml-vulkan/vulkan-shaders/vulkan-shaders-gen.cpp +593 -0
data/ext/ggml/src/ggml.c +7694 -0
data/ext/include/whisper.h +672 -0
data/ext/metal-embed.mk +17 -0
data/ext/metal.mk +6 -0
data/ext/ruby_whisper.cpp +1608 -159
data/ext/ruby_whisper.h +10 -0
data/ext/scripts/get-flags.mk +38 -0
data/ext/src/coreml/whisper-decoder-impl.h +146 -0
data/ext/src/coreml/whisper-decoder-impl.m +201 -0
data/ext/src/coreml/whisper-encoder-impl.h +142 -0
data/ext/src/coreml/whisper-encoder-impl.m +197 -0
data/ext/src/coreml/whisper-encoder.h +26 -0
data/ext/src/openvino/whisper-openvino-encoder.cpp +108 -0
data/ext/src/openvino/whisper-openvino-encoder.h +31 -0
data/ext/src/whisper.cpp +7393 -0
data/extsources.rb +6 -0
data/lib/whisper/model/uri.rb +157 -0
data/lib/whisper.rb +2 -0
data/tests/helper.rb +7 -0
data/tests/jfk_reader/.gitignore +5 -0
data/tests/jfk_reader/extconf.rb +3 -0
data/tests/jfk_reader/jfk_reader.c +68 -0
data/tests/test_callback.rb +160 -0
data/tests/test_error.rb +20 -0
data/tests/test_model.rb +71 -0
data/tests/test_package.rb +31 -0
data/tests/test_params.rb +160 -0
data/tests/test_segment.rb +83 -0
data/tests/test_whisper.rb +211 -123
data/whispercpp.gemspec +36 -0
metadata +137 -11
data/ext/ggml.c +0 -8616
data/ext/ggml.h +0 -748
data/ext/whisper.cpp +0 -4829
data/ext/whisper.h +0 -402

data/ext/ggml/include/ggml-cann.h ADDED Viewed

@@ -0,0 +1,123 @@
+/*
+ * Copyright (c) 2023-2024 The ggml authors
+ *
+ * Permission is hereby granted, free of charge, to any person obtaining a copy
+ * of this software and associated documentation files (the "Software"), to
+ * deal in the Software without restriction, including without limitation the
+ * rights to use, copy, modify, merge, publish, distribute, sublicense, and/or
+ * sell copies of the Software, and to permit persons to whom the Software is
+ * furnished to do so, subject to the following conditions:
+ *
+ * The above copyright notice and this permission notice shall be included in
+ * all copies or substantial portions of the Software.
+ *
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+ * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+ * FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+ * AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+ * LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING
+ * FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS
+ * IN THE SOFTWARE.
+ */
+#pragma once
+#include "ggml-backend.h"
+#include "ggml.h"
+#ifdef __cplusplus
+extern "C" {
+#endif
+/**
+ * @brief Maximum number of CANN devices supported.
+ */
+#define GGML_CANN_MAX_DEVICES 16
+GGML_BACKEND_API ggml_backend_reg_t ggml_backend_cann_reg(void);
+/**
+ * @brief Initializes the CANN backend for a specified device.
+ *
+ * This function initializes the CANN backend for the given device.
+ * It verifies the device index, allocates a context, and creates a backend
+ * instance.
+ *
+ * @param device The index of the device to initialize.
+ * @return A pointer to the initialized backend instance, or nullptr on failure.
+ */
+GGML_BACKEND_API ggml_backend_t ggml_backend_cann_init(int32_t device);
+/**
+ * @brief Checks if a given backend is a CANN backend.
+ *
+ * This function verifies if the provided backend is a CANN backend by comparing
+ * its GUID with the CANN backend's GUID.
+ *
+ * @param backend The backend instance to check.
+ * @return True if the backend is a CANN backend, false otherwise.
+ */
+GGML_BACKEND_API bool ggml_backend_is_cann(ggml_backend_t backend);
+/**
+ * @brief Retrieves the CANN buffer type for a specified device.
+ *
+ * This function initializes and returns the buffer type interface associated
+ * with the given device. It ensures thread-safe access using a mutex.
+ *
+ * @param device The device index for which to retrieve the buffer type.
+ * @return A pointer to the buffer type interface for the specified device, or
+ * nullptr if the device index is out of range.
+ */
+GGML_BACKEND_API ggml_backend_buffer_type_t
+ggml_backend_cann_buffer_type(int32_t device);
+/**
+ * @brief Retrieves the number of CANN devices available.
+ *
+ * This function returns the number of CANN devices available based on
+ * information obtained from `ggml_cann_info()`.
+ *
+ * @return The number of CANN devices available.
+ */
+GGML_BACKEND_API int32_t ggml_backend_cann_get_device_count(void);
+/**
+ * @brief pinned host buffer for use with the CPU backend for faster copies between CPU and NPU.
+ *
+ * @return A pointer to the host buffer type interface.
+ */
+GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_cann_host_buffer_type(void);
+/**
+ * @brief Retrieves the description of a specific CANN device.
+ *
+ * This function sets the specified device, retrieves the SoC name,
+ * and writes it into the provided description buffer.
+ *
+ * @param device The device index to retrieve the description for.
+ * @param description Pointer to a buffer where the description will be written.
+ * @param description_size Size of the description buffer.
+ */
+GGML_BACKEND_API void ggml_backend_cann_get_device_description(
+    int32_t device, char* description, size_t description_size);
+/**
+ * @brief Retrieves the memory information of a specific CANN device.
+ *
+ * This function sets the specified device, retrieves the free and total
+ * memory information of the specified type (ACL_HBM_MEM), and stores them
+ * in the provided pointers.
+ *
+ * @param device The device index to retrieve memory information for.
+ * @param free Pointer to a variable where the free memory size will be stored.
+ * @param total Pointer to a variable where the total memory size will be
+ * stored.
+ */
+GGML_BACKEND_API void ggml_backend_cann_get_device_memory(int32_t device,
+                                                  size_t* free,
+                                                  size_t* total);
+#ifdef __cplusplus
+}
+#endif

data/ext/ggml/include/ggml-cpp.h ADDED Viewed

@@ -0,0 +1,38 @@
+#pragma once
+#ifndef __cplusplus
+#error "This header is for C++ only"
+#endif
+#include "ggml.h"
+#include "ggml-alloc.h"
+#include "ggml-backend.h"
+#include <memory>
+// Smart pointers for ggml types
+// ggml
+struct ggml_context_deleter { void operator()(ggml_context * ctx) { ggml_free(ctx); } };
+struct gguf_context_deleter { void operator()(gguf_context * ctx) { gguf_free(ctx); } };
+typedef std::unique_ptr<ggml_context, ggml_context_deleter> ggml_context_ptr;
+typedef std::unique_ptr<gguf_context, gguf_context_deleter> gguf_context_ptr;
+// ggml-alloc
+struct ggml_gallocr_deleter { void operator()(ggml_gallocr_t galloc) { ggml_gallocr_free(galloc); } };
+typedef std::unique_ptr<ggml_gallocr_t, ggml_gallocr_deleter> ggml_gallocr_ptr;
+// ggml-backend
+struct ggml_backend_deleter        { void operator()(ggml_backend_t backend)       { ggml_backend_free(backend); } };
+struct ggml_backend_buffer_deleter { void operator()(ggml_backend_buffer_t buffer) { ggml_backend_buffer_free(buffer); } };
+struct ggml_backend_event_deleter  { void operator()(ggml_backend_event_t event)   { ggml_backend_event_free(event); } };
+struct ggml_backend_sched_deleter  { void operator()(ggml_backend_sched_t sched)   { ggml_backend_sched_free(sched); } };
+typedef std::unique_ptr<ggml_backend,        ggml_backend_deleter>        ggml_backend_ptr;
+typedef std::unique_ptr<ggml_backend_buffer, ggml_backend_buffer_deleter> ggml_backend_buffer_ptr;
+typedef std::unique_ptr<ggml_backend_event,  ggml_backend_event_deleter>  ggml_backend_event_ptr;
+typedef std::unique_ptr<ggml_backend_sched,  ggml_backend_sched_deleter>  ggml_backend_sched_ptr;

data/ext/ggml/include/ggml-cpu.h ADDED Viewed

@@ -0,0 +1,135 @@
+#pragma once
+#include "ggml.h"
+#include "ggml-backend.h"
+#ifdef  __cplusplus
+extern "C" {
+#endif
+    // the compute plan that needs to be prepared for ggml_graph_compute()
+    // since https://github.com/ggerganov/ggml/issues/287
+    struct ggml_cplan {
+        size_t    work_size; // size of work buffer, calculated by `ggml_graph_plan()`
+        uint8_t * work_data; // work buffer, to be allocated by caller before calling to `ggml_graph_compute()`
+        int n_threads;
+        struct ggml_threadpool * threadpool;
+        // abort ggml_graph_compute when true
+        ggml_abort_callback abort_callback;
+        void *              abort_callback_data;
+    };
+    // numa strategies
+    enum ggml_numa_strategy {
+        GGML_NUMA_STRATEGY_DISABLED   = 0,
+        GGML_NUMA_STRATEGY_DISTRIBUTE = 1,
+        GGML_NUMA_STRATEGY_ISOLATE    = 2,
+        GGML_NUMA_STRATEGY_NUMACTL    = 3,
+        GGML_NUMA_STRATEGY_MIRROR     = 4,
+        GGML_NUMA_STRATEGY_COUNT
+    };
+    GGML_BACKEND_API void    ggml_numa_init(enum ggml_numa_strategy numa); // call once for better performance on NUMA systems
+    GGML_BACKEND_API bool    ggml_is_numa(void); // true if init detected that system has >1 NUMA node
+    GGML_BACKEND_API struct ggml_tensor * ggml_new_i32(struct ggml_context * ctx, int32_t value);
+    GGML_BACKEND_API struct ggml_tensor * ggml_new_f32(struct ggml_context * ctx, float value);
+    GGML_BACKEND_API struct ggml_tensor * ggml_set_i32 (struct ggml_tensor * tensor, int32_t value);
+    GGML_BACKEND_API struct ggml_tensor * ggml_set_f32 (struct ggml_tensor * tensor, float value);
+    GGML_BACKEND_API int32_t ggml_get_i32_1d(const struct ggml_tensor * tensor, int i);
+    GGML_BACKEND_API void    ggml_set_i32_1d(const struct ggml_tensor * tensor, int i, int32_t value);
+    GGML_BACKEND_API int32_t ggml_get_i32_nd(const struct ggml_tensor * tensor, int i0, int i1, int i2, int i3);
+    GGML_BACKEND_API void    ggml_set_i32_nd(const struct ggml_tensor * tensor, int i0, int i1, int i2, int i3, int32_t value);
+    GGML_BACKEND_API float   ggml_get_f32_1d(const struct ggml_tensor * tensor, int i);
+    GGML_BACKEND_API void    ggml_set_f32_1d(const struct ggml_tensor * tensor, int i, float value);
+    GGML_BACKEND_API float   ggml_get_f32_nd(const struct ggml_tensor * tensor, int i0, int i1, int i2, int i3);
+    GGML_BACKEND_API void    ggml_set_f32_nd(const struct ggml_tensor * tensor, int i0, int i1, int i2, int i3, float value);
+    GGML_BACKEND_API struct ggml_threadpool *      ggml_threadpool_new           (struct ggml_threadpool_params  * params);
+    GGML_BACKEND_API void                          ggml_threadpool_free          (struct ggml_threadpool * threadpool);
+    GGML_BACKEND_API int                           ggml_threadpool_get_n_threads (struct ggml_threadpool * threadpool);
+    GGML_BACKEND_API void                          ggml_threadpool_pause         (struct ggml_threadpool * threadpool);
+    GGML_BACKEND_API void                          ggml_threadpool_resume        (struct ggml_threadpool * threadpool);
+    // ggml_graph_plan() has to be called before ggml_graph_compute()
+    // when plan.work_size > 0, caller must allocate memory for plan.work_data
+    GGML_BACKEND_API struct ggml_cplan ggml_graph_plan(
+                  const struct ggml_cgraph * cgraph,
+                                       int   n_threads, /* = GGML_DEFAULT_N_THREADS */
+                    struct ggml_threadpool * threadpool /* = NULL */ );
+    GGML_BACKEND_API enum ggml_status  ggml_graph_compute(struct ggml_cgraph * cgraph, struct ggml_cplan * cplan);
+    // same as ggml_graph_compute() but the work data is allocated as a part of the context
+    // note: the drawback of this API is that you must have ensured that the context has enough memory for the work data
+    GGML_BACKEND_API enum ggml_status  ggml_graph_compute_with_ctx(struct ggml_context * ctx, struct ggml_cgraph * cgraph, int n_threads);
+    //
+    // system info
+    //
+    // x86
+    GGML_BACKEND_API int ggml_cpu_has_sse3       (void);
+    GGML_BACKEND_API int ggml_cpu_has_ssse3      (void);
+    GGML_BACKEND_API int ggml_cpu_has_avx        (void);
+    GGML_BACKEND_API int ggml_cpu_has_avx_vnni   (void);
+    GGML_BACKEND_API int ggml_cpu_has_avx2       (void);
+    GGML_BACKEND_API int ggml_cpu_has_f16c       (void);
+    GGML_BACKEND_API int ggml_cpu_has_fma        (void);
+    GGML_BACKEND_API int ggml_cpu_has_avx512     (void);
+    GGML_BACKEND_API int ggml_cpu_has_avx512_vbmi(void);
+    GGML_BACKEND_API int ggml_cpu_has_avx512_vnni(void);
+    GGML_BACKEND_API int ggml_cpu_has_avx512_bf16(void);
+    GGML_BACKEND_API int ggml_cpu_has_amx_int8   (void);
+    // ARM
+    GGML_BACKEND_API int ggml_cpu_has_neon       (void);
+    GGML_BACKEND_API int ggml_cpu_has_arm_fma    (void);
+    GGML_BACKEND_API int ggml_cpu_has_fp16_va    (void);
+    GGML_BACKEND_API int ggml_cpu_has_dotprod    (void);
+    GGML_BACKEND_API int ggml_cpu_has_matmul_int8(void);
+    GGML_BACKEND_API int ggml_cpu_has_sve        (void);
+    GGML_BACKEND_API int ggml_cpu_get_sve_cnt    (void);  // sve vector length in bytes
+    // other
+    GGML_BACKEND_API int ggml_cpu_has_riscv_v    (void);
+    GGML_BACKEND_API int ggml_cpu_has_vsx        (void);
+    GGML_BACKEND_API int ggml_cpu_has_wasm_simd  (void);
+    GGML_BACKEND_API int ggml_cpu_has_llamafile  (void);
+    // Internal types and functions exposed for tests and benchmarks
+    typedef void (*ggml_vec_dot_t)  (int n, float * GGML_RESTRICT s, size_t bs, const void * GGML_RESTRICT x, size_t bx,
+                                       const void * GGML_RESTRICT y, size_t by, int nrc);
+    struct ggml_type_traits_cpu {
+        ggml_from_float_t        from_float;
+        ggml_vec_dot_t           vec_dot;
+        enum ggml_type           vec_dot_type;
+        int64_t                  nrows; // number of rows to process simultaneously
+    };
+    GGML_BACKEND_API const struct ggml_type_traits_cpu * ggml_get_type_traits_cpu(enum ggml_type type);
+    GGML_BACKEND_API void ggml_cpu_init(void);
+    //
+    // CPU backend
+    //
+    GGML_BACKEND_API ggml_backend_t ggml_backend_cpu_init(void);
+    GGML_BACKEND_API bool ggml_backend_is_cpu                (ggml_backend_t backend);
+    GGML_BACKEND_API void ggml_backend_cpu_set_n_threads     (ggml_backend_t backend_cpu, int n_threads);
+    GGML_BACKEND_API void ggml_backend_cpu_set_threadpool    (ggml_backend_t backend_cpu, ggml_threadpool_t threadpool);
+    GGML_BACKEND_API void ggml_backend_cpu_set_abort_callback(ggml_backend_t backend_cpu, ggml_abort_callback abort_callback, void * abort_callback_data);
+    GGML_BACKEND_API ggml_backend_reg_t ggml_backend_cpu_reg(void);
+#ifdef __cplusplus
+}
+#endif

data/ext/ggml/include/ggml-cuda.h ADDED Viewed

@@ -0,0 +1,47 @@
+#pragma once
+#include "ggml.h"
+#include "ggml-backend.h"
+#ifdef  __cplusplus
+extern "C" {
+#endif
+#ifdef GGML_USE_HIP
+#define GGML_CUDA_NAME "ROCm"
+#define GGML_CUBLAS_NAME "hipBLAS"
+#elif defined(GGML_USE_MUSA)
+#define GGML_CUDA_NAME "MUSA"
+#define GGML_CUBLAS_NAME "muBLAS"
+#else
+#define GGML_CUDA_NAME "CUDA"
+#define GGML_CUBLAS_NAME "cuBLAS"
+#endif
+#define GGML_CUDA_MAX_DEVICES       16
+// backend API
+GGML_BACKEND_API ggml_backend_t ggml_backend_cuda_init(int device);
+GGML_BACKEND_API bool ggml_backend_is_cuda(ggml_backend_t backend);
+// device buffer
+GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_cuda_buffer_type(int device);
+// split tensor buffer that splits matrices by rows across multiple devices
+GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_cuda_split_buffer_type(int main_device, const float * tensor_split);
+// pinned host buffer for use with the CPU backend for faster copies between CPU and GPU
+GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_cuda_host_buffer_type(void);
+GGML_BACKEND_API int  ggml_backend_cuda_get_device_count(void);
+GGML_BACKEND_API void ggml_backend_cuda_get_device_description(int device, char * description, size_t description_size);
+GGML_BACKEND_API void ggml_backend_cuda_get_device_memory(int device, size_t * free, size_t * total);
+GGML_BACKEND_API bool ggml_backend_cuda_register_host_buffer(void * buffer, size_t size);
+GGML_BACKEND_API void ggml_backend_cuda_unregister_host_buffer(void * buffer);
+GGML_BACKEND_API ggml_backend_reg_t ggml_backend_cuda_reg(void);
+#ifdef  __cplusplus
+}
+#endif

data/ext/ggml/include/ggml-kompute.h ADDED Viewed

@@ -0,0 +1,50 @@
+#pragma once
+#include "ggml.h"
+#include "ggml-backend.h"
+#include <stdbool.h>
+#include <stddef.h>
+#include <stdint.h>
+#ifdef __cplusplus
+extern "C" {
+#endif
+#define GGML_KOMPUTE_MAX_DEVICES 16
+struct ggml_vk_device {
+    int index;
+    int type; // same as VkPhysicalDeviceType
+    size_t heapSize;
+    const char * name;
+    const char * vendor;
+    int subgroupSize;
+    uint64_t bufferAlignment;
+    uint64_t maxAlloc;
+};
+struct ggml_vk_device * ggml_vk_available_devices(size_t memoryRequired, size_t * count);
+bool ggml_vk_get_device(struct ggml_vk_device * device, size_t memoryRequired, const char * name);
+bool ggml_vk_has_vulkan(void);
+bool ggml_vk_has_device(void);
+struct ggml_vk_device ggml_vk_current_device(void);
+//
+// backend API
+//
+// forward declaration
+typedef struct ggml_backend * ggml_backend_t;
+GGML_BACKEND_API ggml_backend_t ggml_backend_kompute_init(int device);
+GGML_BACKEND_API bool ggml_backend_is_kompute(ggml_backend_t backend);
+GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_kompute_buffer_type(int device);
+GGML_BACKEND_API ggml_backend_reg_t ggml_backend_kompute_reg(void);
+#ifdef __cplusplus
+}
+#endif

data/ext/ggml/include/ggml-metal.h ADDED Viewed

@@ -0,0 +1,66 @@
+// Note: this description is outdated
+//
+// An interface allowing to compute ggml_cgraph with Metal
+//
+// This is a fully functional interface that extends ggml with GPU support for Apple devices.
+// A similar interface can be created for other GPU backends (e.g. Vulkan, CUDA, etc.)
+//
+// How it works?
+//
+// As long as your program can create and evaluate a ggml_cgraph on the CPU, you can use this
+// interface to evaluate the same graph on the GPU. Instead of using ggml_graph_compute(), you
+// use ggml_metal_graph_compute() (or ggml_vulkan_graph_compute(), etc.)
+//
+// You only need to make sure that all memory buffers that you used during the graph creation
+// are mapped to the device memory with the ggml_metal_add_buffer() function. This mapping is
+// used during the graph evaluation to determine the arguments of the compute kernels.
+//
+// Synchronization between device and host memory (for example for input and output tensors)
+// is done with the ggml_metal_set_tensor() and ggml_metal_get_tensor() functions.
+//
+#pragma once
+#include "ggml.h"
+#include "ggml-backend.h"
+#include <stddef.h>
+#include <stdbool.h>
+struct ggml_tensor;
+struct ggml_cgraph;
+#ifdef __cplusplus
+extern "C" {
+#endif
+//
+// backend API
+// user-code should use only these functions
+//
+GGML_BACKEND_API ggml_backend_t ggml_backend_metal_init(void);
+GGML_BACKEND_API bool ggml_backend_is_metal(ggml_backend_t backend);
+GGML_DEPRECATED(
+        GGML_BACKEND_API ggml_backend_buffer_t ggml_backend_metal_buffer_from_ptr(void * data, size_t size, size_t max_size),
+        "obsoleted by the new device interface - https://github.com/ggerganov/llama.cpp/pull/9713");
+GGML_BACKEND_API void ggml_backend_metal_set_abort_callback(ggml_backend_t backend, ggml_abort_callback abort_callback, void * user_data);
+GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_metal_buffer_type(void);
+// helper to check if the device supports a specific family
+// ideally, the user code should be doing these checks
+// ref: https://developer.apple.com/metal/Metal-Feature-Set-Tables.pdf
+GGML_BACKEND_API bool ggml_backend_metal_supports_family(ggml_backend_t backend, int family);
+// capture all command buffers committed the next time `ggml_backend_graph_compute` is called
+GGML_BACKEND_API void ggml_backend_metal_capture_next_compute(ggml_backend_t backend);
+GGML_BACKEND_API ggml_backend_reg_t ggml_backend_metal_reg(void);
+#ifdef __cplusplus
+}
+#endif

data/ext/ggml/include/ggml-opencl.h ADDED Viewed

@@ -0,0 +1,26 @@
+#ifndef GGML_OPENCL_H
+#define GGML_OPENCL_H
+#include "ggml.h"
+#include "ggml-backend.h"
+#ifdef  __cplusplus
+extern "C" {
+#endif
+//
+// backend API
+//
+GGML_BACKEND_API ggml_backend_t ggml_backend_opencl_init(void);
+GGML_BACKEND_API bool ggml_backend_is_opencl(ggml_backend_t backend);
+GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_opencl_buffer_type(void);
+GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_opencl_host_buffer_type(void);
+GGML_BACKEND_API ggml_backend_reg_t ggml_backend_opencl_reg(void);
+#ifdef  __cplusplus
+}
+#endif
+#endif // GGML_OPENCL_H