RubyGems - llama_cpp - Versions diffs - 0.12.6 → 0.12.7 - Mend

llama_cpp 0.12.6 → 0.12.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (22) hide show

checksums.yaml +4 -4
data/CHANGELOG.md +10 -0
data/ext/llama_cpp/llama_cpp.cpp +21 -10
data/lib/llama_cpp/version.rb +2 -2
data/sig/llama_cpp.rbs +8 -1
data/vendor/tmp/llama.cpp/Makefile +43 -12
data/vendor/tmp/llama.cpp/ggml-alloc.c +73 -43
data/vendor/tmp/llama.cpp/ggml-backend.c +18 -9
data/vendor/tmp/llama.cpp/ggml-cuda.cu +560 -346
data/vendor/tmp/llama.cpp/ggml-impl.h +20 -7
data/vendor/tmp/llama.cpp/ggml-metal.m +99 -11
data/vendor/tmp/llama.cpp/ggml-metal.metal +608 -9
data/vendor/tmp/llama.cpp/ggml-quants.c +908 -54
data/vendor/tmp/llama.cpp/ggml-quants.h +25 -2
data/vendor/tmp/llama.cpp/ggml-sycl.cpp +81 -203
data/vendor/tmp/llama.cpp/ggml-vulkan.cpp +124 -52
data/vendor/tmp/llama.cpp/ggml.c +948 -504
data/vendor/tmp/llama.cpp/ggml.h +24 -11
data/vendor/tmp/llama.cpp/llama.cpp +688 -163
data/vendor/tmp/llama.cpp/llama.h +37 -1
data/vendor/tmp/llama.cpp/scripts/get-flags.mk +1 -1
metadata +2 -2

data/vendor/tmp/llama.cpp/ggml.h CHANGED Viewed

@@ -315,13 +315,7 @@
 extern "C" {
 #endif
-#if defined(__ARM_NEON) && defined(__CUDACC__)
-    typedef half ggml_fp16_t;
-#elif defined(__ARM_NEON) && !defined(_MSC_VER)
-    typedef __fp16 ggml_fp16_t;
-#else
     typedef uint16_t ggml_fp16_t;
-#endif
     // convert FP16 <-> FP32
     GGML_API float       ggml_fp16_to_fp32(ggml_fp16_t x);
@@ -354,6 +348,8 @@ extern "C" {
         GGML_TYPE_IQ2_XXS = 16,
         GGML_TYPE_IQ2_XS  = 17,
         GGML_TYPE_IQ3_XXS = 18,
+        GGML_TYPE_IQ1_S   = 19,
+        GGML_TYPE_IQ4_NL  = 20,
         GGML_TYPE_I8,
         GGML_TYPE_I16,
         GGML_TYPE_I32,
@@ -391,6 +387,8 @@ extern "C" {
         GGML_FTYPE_MOSTLY_IQ2_XXS = 15, // except 1d tensors
         GGML_FTYPE_MOSTLY_IQ2_XS  = 16, // except 1d tensors
         GGML_FTYPE_MOSTLY_IQ3_XXS = 17, // except 1d tensors
+        GGML_FTYPE_MOSTLY_IQ1_S   = 18, // except 1d tensors
+        GGML_FTYPE_MOSTLY_IQ4_NL  = 19, // except 1d tensors
     };
     // available tensor operations:
@@ -658,6 +656,16 @@ extern "C" {
         void * wdata;
     };
+    // numa strategies
+    enum ggml_numa_strategy {
+        GGML_NUMA_STRATEGY_DISABLED   = 0,
+        GGML_NUMA_STRATEGY_DISTRIBUTE = 1,
+        GGML_NUMA_STRATEGY_ISOLATE    = 2,
+        GGML_NUMA_STRATEGY_NUMACTL    = 3,
+        GGML_NUMA_STRATEGY_MIRROR     = 4,
+        GGML_NUMA_STRATEGY_COUNT
+    };
     // misc
     GGML_API void    ggml_time_init(void); // call this once at the beginning of the program
@@ -668,7 +676,7 @@ extern "C" {
     GGML_API void    ggml_print_backtrace(void);
-    GGML_API void    ggml_numa_init(void); // call once for better performance on NUMA systems
+    GGML_API void    ggml_numa_init(enum ggml_numa_strategy numa); // call once for better performance on NUMA systems
     GGML_API bool    ggml_is_numa(void); // true if init detected that system has >1 NUMA node
     GGML_API void    ggml_print_object (const struct ggml_object * obj);
@@ -1373,13 +1381,17 @@ extern "C" {
             struct ggml_context * ctx,
             struct ggml_tensor  * a);
-    // fused soft_max(a*scale + mask)
+    // fused soft_max(a*scale + mask + pos[i]*(ALiBi slope))
     // mask is optional
+    // pos is required when max_bias > 0.0f
+    // max_bias = 0.0f for no ALiBi
     GGML_API struct ggml_tensor * ggml_soft_max_ext(
             struct ggml_context * ctx,
             struct ggml_tensor  * a,
             struct ggml_tensor  * mask,
-            float                 scale);
+            struct ggml_tensor  * pos,
+            float                 scale,
+            float                 max_bias);
     GGML_API struct ggml_tensor * ggml_soft_max_back(
             struct ggml_context * ctx,
@@ -1481,12 +1493,13 @@ extern "C" {
     // alibi position embedding
     // in-place, returns view(a)
-    GGML_API struct ggml_tensor * ggml_alibi(
+    GGML_DEPRECATED(GGML_API struct ggml_tensor * ggml_alibi(
             struct ggml_context * ctx,
             struct ggml_tensor  * a,
             int                   n_past,
             int                   n_head,
-            float                 bias_max);
+            float                 bias_max),
+        "use ggml_soft_max_ext instead (will be removed in Mar 2024)");
     // clamp
     // in-place, returns view(a)