npm - cui-llama.rn - Versions diffs - 1.2.4 → 1.3.0 - Mend

cui-llama.rn 1.2.4 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (70) hide show

package/README.md +3 -4
package/android/src/main/CMakeLists.txt +21 -5
package/android/src/main/java/com/rnllama/LlamaContext.java +115 -30
package/android/src/main/java/com/rnllama/RNLlama.java +40 -7
package/android/src/main/jni.cpp +222 -36
package/android/src/newarch/java/com/rnllama/RNLlamaModule.java +9 -4
package/android/src/oldarch/java/com/rnllama/RNLlamaModule.java +9 -4
package/cpp/common.cpp +1682 -2122
package/cpp/common.h +600 -594
package/cpp/ggml-aarch64.c +129 -3209
package/cpp/ggml-aarch64.h +19 -39
package/cpp/ggml-alloc.c +1040 -1040
package/cpp/ggml-alloc.h +76 -76
package/cpp/ggml-backend-impl.h +216 -227
package/cpp/ggml-backend-reg.cpp +195 -0
package/cpp/ggml-backend.cpp +1997 -2625
package/cpp/ggml-backend.h +328 -326
package/cpp/ggml-common.h +1853 -1853
package/cpp/ggml-cpp.h +38 -0
package/cpp/ggml-cpu-aarch64.c +3560 -0
package/cpp/ggml-cpu-aarch64.h +30 -0
package/cpp/ggml-cpu-impl.h +371 -614
package/cpp/ggml-cpu-quants.c +10822 -0
package/cpp/ggml-cpu-quants.h +63 -0
package/cpp/ggml-cpu.c +13975 -0
package/cpp/ggml-cpu.cpp +663 -0
package/cpp/ggml-cpu.h +177 -0
package/cpp/ggml-impl.h +550 -209
package/cpp/ggml-metal.h +66 -66
package/cpp/ggml-metal.m +4294 -3819
package/cpp/ggml-quants.c +5247 -15752
package/cpp/ggml-quants.h +100 -147
package/cpp/ggml-threading.cpp +12 -0
package/cpp/ggml-threading.h +12 -0
package/cpp/ggml.c +8180 -23464
package/cpp/ggml.h +2411 -2562
package/cpp/llama-grammar.cpp +1138 -1138
package/cpp/llama-grammar.h +144 -144
package/cpp/llama-impl.h +181 -181
package/cpp/llama-sampling.cpp +2348 -2194
package/cpp/llama-sampling.h +48 -30
package/cpp/llama-vocab.cpp +1984 -1968
package/cpp/llama-vocab.h +170 -165
package/cpp/llama.cpp +22132 -21969
package/cpp/llama.h +1253 -1253
package/cpp/log.cpp +401 -401
package/cpp/log.h +121 -121
package/cpp/rn-llama.hpp +83 -19
package/cpp/sampling.cpp +466 -458
package/cpp/sgemm.cpp +1884 -1219
package/ios/RNLlama.mm +43 -20
package/ios/RNLlamaContext.h +9 -3
package/ios/RNLlamaContext.mm +133 -33
package/jest/mock.js +0 -1
package/lib/commonjs/NativeRNLlama.js.map +1 -1
package/lib/commonjs/index.js +52 -15
package/lib/commonjs/index.js.map +1 -1
package/lib/module/NativeRNLlama.js.map +1 -1
package/lib/module/index.js +51 -15
package/lib/module/index.js.map +1 -1
package/lib/typescript/NativeRNLlama.d.ts +29 -6
package/lib/typescript/NativeRNLlama.d.ts.map +1 -1
package/lib/typescript/index.d.ts +12 -5
package/lib/typescript/index.d.ts.map +1 -1
package/package.json +1 -1
package/src/NativeRNLlama.ts +41 -7
package/src/index.ts +82 -27
package/cpp/json-schema-to-grammar.cpp +0 -1045
package/cpp/json-schema-to-grammar.h +0 -8
package/cpp/json.hpp +0 -24766

package/cpp/log.h CHANGED Viewed

@@ -1,121 +1,121 @@
-#pragma once
-#include "ggml.h" // for lm_ggml_log_level
-#ifndef __GNUC__
-#    define LOG_ATTRIBUTE_FORMAT(...)
-#elif defined(__MINGW32__)
-#    define LOG_ATTRIBUTE_FORMAT(...) __attribute__((format(gnu_printf, __VA_ARGS__)))
-#else
-#    define LOG_ATTRIBUTE_FORMAT(...) __attribute__((format(printf, __VA_ARGS__)))
-#endif
-#define LOG_DEFAULT_DEBUG 1
-#define LOG_DEFAULT_LLAMA 0
-// needed by the LOG_TMPL macro to avoid computing log arguments if the verbosity lower
-// set via common_log_set_verbosity()
-extern int common_log_verbosity_thold;
-void common_log_set_verbosity_thold(int verbosity); // not thread-safe
-// the common_log uses an internal worker thread to print/write log messages
-// when the worker thread is paused, incoming log messages are discarded
-struct common_log;
-struct common_log * common_log_init();
-struct common_log * common_log_main(); // singleton, automatically destroys itself on exit
-void                common_log_pause (struct common_log * log); // pause  the worker thread, not thread-safe
-void                common_log_resume(struct common_log * log); // resume the worker thread, not thread-safe
-void                common_log_free  (struct common_log * log);
-LOG_ATTRIBUTE_FORMAT(3, 4)
-void common_log_add(struct common_log * log, enum lm_ggml_log_level level, const char * fmt, ...);
-// defaults: file = NULL, colors = false, prefix = false, timestamps = false
-//
-// regular log output:
-//
-//   lm_ggml_backend_metal_log_allocated_size: allocated buffer, size =  6695.84 MiB, ( 6695.91 / 21845.34)
-//   llm_load_tensors: ggml ctx size =    0.27 MiB
-//   llm_load_tensors: offloading 32 repeating layers to GPU
-//   llm_load_tensors: offloading non-repeating layers to GPU
-//
-// with prefix = true, timestamps = true, the log output will look like this:
-//
-//   0.00.035.060 D lm_ggml_backend_metal_log_allocated_size: allocated buffer, size =  6695.84 MiB, ( 6695.91 / 21845.34)
-//   0.00.035.064 I llm_load_tensors: ggml ctx size =    0.27 MiB
-//   0.00.090.578 I llm_load_tensors: offloading 32 repeating layers to GPU
-//   0.00.090.579 I llm_load_tensors: offloading non-repeating layers to GPU
-//
-// I - info    (stdout, V = 0)
-// W - warning (stderr, V = 0)
-// E - error   (stderr, V = 0)
-// D - debug   (stderr, V = LOG_DEFAULT_DEBUG)
-//
-void common_log_set_file      (struct common_log * log, const char * file);       // not thread-safe
-void common_log_set_colors    (struct common_log * log,       bool   colors);     // not thread-safe
-void common_log_set_prefix    (struct common_log * log,       bool   prefix);     // whether to output prefix to each log
-void common_log_set_timestamps(struct common_log * log,       bool   timestamps); // whether to output timestamps in the prefix
-// helper macros for logging
-// use these to avoid computing log arguments if the verbosity of the log is higher than the threshold
-//
-// for example:
-//
-//   LOG_DBG("this is a debug message: %d\n", expensive_function());
-//
-// this will avoid calling expensive_function() if LOG_DEFAULT_DEBUG > common_log_verbosity_thold
-//
-#if defined(__ANDROID__)
-#include <android/log.h>
-#define LLAMA_ANDROID_LOG_TAG "RNLLAMA_LOG_ANDROID"
-#if defined(RNLLAMA_ANDROID_ENABLE_LOGGING)
-#define RNLLAMA_LOG_LEVEL 1
-#else
-#define RNLLAMA_LOG_LEVEL 0
-#endif
-#define LOG_TMPL(level, verbosity, ...) \
-    do { \
-        if ((verbosity) <= RNLLAMA_LOG_LEVEL) { \
-            int android_log_level = ANDROID_LOG_DEFAULT; \
-            switch (level) { \
-                case LM_GGML_LOG_LEVEL_INFO:  android_log_level = ANDROID_LOG_INFO; break; \
-                case LM_GGML_LOG_LEVEL_WARN:  android_log_level = ANDROID_LOG_WARN; break; \
-                case LM_GGML_LOG_LEVEL_ERROR: android_log_level = ANDROID_LOG_ERROR; break; \
-                default:             android_log_level = ANDROID_LOG_DEFAULT; \
-            } \
-            __android_log_print(android_log_level, LLAMA_ANDROID_LOG_TAG, __VA_ARGS__); \
-        } \
-    } while(0)
-#else
-#define LOG_TMPL(level, verbosity, ...) \
-    do { \
-        if ((verbosity) <= common_log_verbosity_thold) { \
-            common_log_add(common_log_main(), (level), __VA_ARGS__); \
-        } \
-    } while (0)
-#endif
-#define LOG(...)             LOG_TMPL(LM_GGML_LOG_LEVEL_NONE, 0,         __VA_ARGS__)
-#define LOGV(verbosity, ...) LOG_TMPL(LM_GGML_LOG_LEVEL_NONE, verbosity, __VA_ARGS__)
-#define LOG_INF(...) LOG_TMPL(LM_GGML_LOG_LEVEL_INFO,  0,                 __VA_ARGS__)
-#define LOG_WRN(...) LOG_TMPL(LM_GGML_LOG_LEVEL_WARN,  0,                 __VA_ARGS__)
-#define LOG_ERR(...) LOG_TMPL(LM_GGML_LOG_LEVEL_ERROR, 0,                 __VA_ARGS__)
-#define LOG_DBG(...) LOG_TMPL(LM_GGML_LOG_LEVEL_DEBUG, LOG_DEFAULT_DEBUG, __VA_ARGS__)
-#define LOG_CNT(...) LOG_TMPL(LM_GGML_LOG_LEVEL_CONT,  0,                 __VA_ARGS__)
-#define LOG_INFV(verbosity, ...) LOG_TMPL(LM_GGML_LOG_LEVEL_INFO,  verbosity, __VA_ARGS__)
-#define LOG_WRNV(verbosity, ...) LOG_TMPL(LM_GGML_LOG_LEVEL_WARN,  verbosity, __VA_ARGS__)
-#define LOG_ERRV(verbosity, ...) LOG_TMPL(LM_GGML_LOG_LEVEL_ERROR, verbosity, __VA_ARGS__)
-#define LOG_DBGV(verbosity, ...) LOG_TMPL(LM_GGML_LOG_LEVEL_DEBUG, verbosity, __VA_ARGS__)
-#define LOG_CNTV(verbosity, ...) LOG_TMPL(LM_GGML_LOG_LEVEL_CONT,  verbosity, __VA_ARGS__)
+#pragma once
+#include "ggml.h" // for lm_ggml_log_level
+#ifndef __GNUC__
+#    define LOG_ATTRIBUTE_FORMAT(...)
+#elif defined(__MINGW32__)
+#    define LOG_ATTRIBUTE_FORMAT(...) __attribute__((format(gnu_printf, __VA_ARGS__)))
+#else
+#    define LOG_ATTRIBUTE_FORMAT(...) __attribute__((format(printf, __VA_ARGS__)))
+#endif
+#define LOG_DEFAULT_DEBUG 1
+#define LOG_DEFAULT_LLAMA 0
+// needed by the LOG_TMPL macro to avoid computing log arguments if the verbosity lower
+// set via common_log_set_verbosity()
+extern int common_log_verbosity_thold;
+void common_log_set_verbosity_thold(int verbosity); // not thread-safe
+// the common_log uses an internal worker thread to print/write log messages
+// when the worker thread is paused, incoming log messages are discarded
+struct common_log;
+struct common_log * common_log_init();
+struct common_log * common_log_main(); // singleton, automatically destroys itself on exit
+void                common_log_pause (struct common_log * log); // pause  the worker thread, not thread-safe
+void                common_log_resume(struct common_log * log); // resume the worker thread, not thread-safe
+void                common_log_free  (struct common_log * log);
+LOG_ATTRIBUTE_FORMAT(3, 4)
+void common_log_add(struct common_log * log, enum lm_ggml_log_level level, const char * fmt, ...);
+// defaults: file = NULL, colors = false, prefix = false, timestamps = false
+//
+// regular log output:
+//
+//   lm_ggml_backend_metal_log_allocated_size: allocated buffer, size =  6695.84 MiB, ( 6695.91 / 21845.34)
+//   llm_load_tensors: ggml ctx size =    0.27 MiB
+//   llm_load_tensors: offloading 32 repeating layers to GPU
+//   llm_load_tensors: offloading non-repeating layers to GPU
+//
+// with prefix = true, timestamps = true, the log output will look like this:
+//
+//   0.00.035.060 D lm_ggml_backend_metal_log_allocated_size: allocated buffer, size =  6695.84 MiB, ( 6695.91 / 21845.34)
+//   0.00.035.064 I llm_load_tensors: ggml ctx size =    0.27 MiB
+//   0.00.090.578 I llm_load_tensors: offloading 32 repeating layers to GPU
+//   0.00.090.579 I llm_load_tensors: offloading non-repeating layers to GPU
+//
+// I - info    (stdout, V = 0)
+// W - warning (stderr, V = 0)
+// E - error   (stderr, V = 0)
+// D - debug   (stderr, V = LOG_DEFAULT_DEBUG)
+//
+void common_log_set_file      (struct common_log * log, const char * file);       // not thread-safe
+void common_log_set_colors    (struct common_log * log,       bool   colors);     // not thread-safe
+void common_log_set_prefix    (struct common_log * log,       bool   prefix);     // whether to output prefix to each log
+void common_log_set_timestamps(struct common_log * log,       bool   timestamps); // whether to output timestamps in the prefix
+// helper macros for logging
+// use these to avoid computing log arguments if the verbosity of the log is higher than the threshold
+//
+// for example:
+//
+//   LOG_DBG("this is a debug message: %d\n", expensive_function());
+//
+// this will avoid calling expensive_function() if LOG_DEFAULT_DEBUG > common_log_verbosity_thold
+//
+#if defined(__ANDROID__)
+#include <android/log.h>
+#define LLAMA_ANDROID_LOG_TAG "RNLLAMA_LOG_ANDROID"
+#if defined(RNLLAMA_ANDROID_ENABLE_LOGGING)
+#define RNLLAMA_LOG_LEVEL 1
+#else
+#define RNLLAMA_LOG_LEVEL 0
+#endif
+#define LOG_TMPL(level, verbosity, ...) \
+    do { \
+        if ((verbosity) <= RNLLAMA_LOG_LEVEL) { \
+            int android_log_level = ANDROID_LOG_DEFAULT; \
+            switch (level) { \
+                case LM_GGML_LOG_LEVEL_INFO:  android_log_level = ANDROID_LOG_INFO; break; \
+                case LM_GGML_LOG_LEVEL_WARN:  android_log_level = ANDROID_LOG_WARN; break; \
+                case LM_GGML_LOG_LEVEL_ERROR: android_log_level = ANDROID_LOG_ERROR; break; \
+                default:             android_log_level = ANDROID_LOG_DEFAULT; \
+            } \
+            __android_log_print(android_log_level, LLAMA_ANDROID_LOG_TAG, __VA_ARGS__); \
+        } \
+    } while(0)
+#else
+#define LOG_TMPL(level, verbosity, ...) \
+    do { \
+        if ((verbosity) <= common_log_verbosity_thold) { \
+            common_log_add(common_log_main(), (level), __VA_ARGS__); \
+        } \
+    } while (0)
+#endif
+#define LOG(...)             LOG_TMPL(LM_GGML_LOG_LEVEL_NONE, 0,         __VA_ARGS__)
+#define LOGV(verbosity, ...) LOG_TMPL(LM_GGML_LOG_LEVEL_NONE, verbosity, __VA_ARGS__)
+#define LOG_INF(...) LOG_TMPL(LM_GGML_LOG_LEVEL_INFO,  0,                 __VA_ARGS__)
+#define LOG_WRN(...) LOG_TMPL(LM_GGML_LOG_LEVEL_WARN,  0,                 __VA_ARGS__)
+#define LOG_ERR(...) LOG_TMPL(LM_GGML_LOG_LEVEL_ERROR, 0,                 __VA_ARGS__)
+#define LOG_DBG(...) LOG_TMPL(LM_GGML_LOG_LEVEL_DEBUG, LOG_DEFAULT_DEBUG, __VA_ARGS__)
+#define LOG_CNT(...) LOG_TMPL(LM_GGML_LOG_LEVEL_CONT,  0,                 __VA_ARGS__)
+#define LOG_INFV(verbosity, ...) LOG_TMPL(LM_GGML_LOG_LEVEL_INFO,  verbosity, __VA_ARGS__)
+#define LOG_WRNV(verbosity, ...) LOG_TMPL(LM_GGML_LOG_LEVEL_WARN,  verbosity, __VA_ARGS__)
+#define LOG_ERRV(verbosity, ...) LOG_TMPL(LM_GGML_LOG_LEVEL_ERROR, verbosity, __VA_ARGS__)
+#define LOG_DBGV(verbosity, ...) LOG_TMPL(LM_GGML_LOG_LEVEL_DEBUG, verbosity, __VA_ARGS__)
+#define LOG_CNTV(verbosity, ...) LOG_TMPL(LM_GGML_LOG_LEVEL_CONT,  verbosity, __VA_ARGS__)

package/cpp/rn-llama.hpp CHANGED Viewed

@@ -4,11 +4,67 @@
 #include <sstream>
 #include <iostream>
 #include "common.h"
+#include "ggml.h"
 #include "llama.h"
+#include "llama-impl.h"
 #include "sampling.h"
 namespace rnllama {
+static std::string lm_gguf_data_to_str(enum lm_gguf_type type, const void * data, int i) {
+    switch (type) {
+        case LM_GGUF_TYPE_UINT8:   return std::to_string(((const uint8_t  *)data)[i]);
+        case LM_GGUF_TYPE_INT8:    return std::to_string(((const int8_t   *)data)[i]);
+        case LM_GGUF_TYPE_UINT16:  return std::to_string(((const uint16_t *)data)[i]);
+        case LM_GGUF_TYPE_INT16:   return std::to_string(((const int16_t  *)data)[i]);
+        case LM_GGUF_TYPE_UINT32:  return std::to_string(((const uint32_t *)data)[i]);
+        case LM_GGUF_TYPE_INT32:   return std::to_string(((const int32_t  *)data)[i]);
+        case LM_GGUF_TYPE_UINT64:  return std::to_string(((const uint64_t *)data)[i]);
+        case LM_GGUF_TYPE_INT64:   return std::to_string(((const int64_t  *)data)[i]);
+        case LM_GGUF_TYPE_FLOAT32: return std::to_string(((const float    *)data)[i]);
+        case LM_GGUF_TYPE_FLOAT64: return std::to_string(((const double   *)data)[i]);
+        case LM_GGUF_TYPE_BOOL:    return ((const bool *)data)[i] ? "true" : "false";
+        default:                   return "unknown type: " + std::to_string(type);
+    }
+}
+static std::string lm_gguf_kv_to_str(const struct lm_gguf_context * ctx_gguf, int i) {
+    const enum lm_gguf_type type = lm_gguf_get_kv_type(ctx_gguf, i);
+    switch (type) {
+        case LM_GGUF_TYPE_STRING:
+            return lm_gguf_get_val_str(ctx_gguf, i);
+        case LM_GGUF_TYPE_ARRAY:
+            {
+                const enum lm_gguf_type arr_type = lm_gguf_get_arr_type(ctx_gguf, i);
+                int arr_n = lm_gguf_get_arr_n(ctx_gguf, i);
+                const void * data = lm_gguf_get_arr_data(ctx_gguf, i);
+                std::stringstream ss;
+                ss << "[";
+                for (int j = 0; j < arr_n; j++) {
+                    if (arr_type == LM_GGUF_TYPE_STRING) {
+                        std::string val = lm_gguf_get_arr_str(ctx_gguf, i, j);
+                        // escape quotes
+                        replace_all(val, "\\", "\\\\");
+                        replace_all(val, "\"", "\\\"");
+                        ss << '"' << val << '"';
+                    } else if (arr_type == LM_GGUF_TYPE_ARRAY) {
+                        ss << "???";
+                    } else {
+                        ss << lm_gguf_data_to_str(arr_type, data, j);
+                    }
+                    if (j < arr_n - 1) {
+                        ss << ", ";
+                    }
+                }
+                ss << "]";
+                return ss.str();
+            }
+        default:
+            return lm_gguf_data_to_str(type, lm_gguf_get_val_data(ctx_gguf, i), 0);
+    }
+}
 static void llama_batch_clear(llama_batch *batch) {
     batch->n_tokens = 0;
 }
@@ -160,9 +216,12 @@ struct llama_rn_context
     common_params params;
     llama_model *model = nullptr;
+    float loading_progress = 0;
+    bool is_load_interrupted = false;
     llama_context *ctx = nullptr;
     common_sampler *ctx_sampling = nullptr;
     int n_ctx;
     bool truncated = false;
@@ -235,13 +294,16 @@ struct llama_rn_context
     }
     bool validateModelChatTemplate() const {
-        llama_chat_message chat[] = {{"user", "test"}};
         std::vector<char> model_template(2048, 0); // longest known template is about 1200 bytes
         std::string template_key = "tokenizer.chat_template";
         int32_t res = llama_model_meta_val_str(model, template_key.c_str(), model_template.data(), model_template.size());
-        return res >= 0;
+        if (res >= 0) {
+            llama_chat_message chat[] = {{"user", "test"}};
+            std::string tmpl = std::string(model_template.data(), model_template.size());
+            int32_t chat_res = llama_chat_apply_template(model, tmpl.c_str(), chat, 1, true, nullptr, 0);
+            return chat_res > 0;
+        }
+        return res > 0;
     }
     void truncatePrompt(std::vector<llama_token> &prompt_tokens) {
@@ -376,7 +438,7 @@ struct llama_rn_context
                 n_eval = params.n_batch;
             }
             if (llama_decode(ctx, llama_batch_get_one(&embd[n_past], n_eval)))
-            {
+            {
                 LOG_ERROR("failed to eval, n_eval: %d, n_past: %d, n_threads: %d, embd: %s",
                     n_eval,
                     n_past,
@@ -387,7 +449,7 @@ struct llama_rn_context
                 return result;
             }
             n_past += n_eval;
             if(is_interrupted) {
                 LOG_INFO("Decoding Interrupted");
                 embd.resize(n_past);
@@ -409,11 +471,11 @@ struct llama_rn_context
             candidates.reserve(llama_n_vocab(model));
             result.tok = common_sampler_sample(ctx_sampling, ctx, -1);
             llama_token_data_array cur_p = *common_sampler_get_candidates(ctx_sampling);
             const int32_t n_probs = params.sparams.n_probs;
             // deprecated
             /*if (params.sparams.temp <= 0 && n_probs > 0)
             {
@@ -421,7 +483,7 @@ struct llama_rn_context
                 llama_sampler_init_softmax();
             }*/
             for (size_t i = 0; i < std::min(cur_p.size, (size_t)n_probs); ++i)
             {
@@ -542,26 +604,28 @@ struct llama_rn_context
         return token_with_probs;
     }
-    std::vector<float> getEmbedding()
+    std::vector<float> getEmbedding(common_params &embd_params)
     {
         static const int n_embd = llama_n_embd(llama_get_model(ctx));
-        if (!params.embedding)
+        if (!embd_params.embedding)
         {
-            LOG_WARNING("embedding disabled, embedding: %s", params.embedding);
+            LOG_WARNING("embedding disabled, embedding: %s", embd_params.embedding);
             return std::vector<float>(n_embd, 0.0f);
         }
         float *data;
-        if(params.pooling_type == 0){
+        const enum llama_pooling_type pooling_type = llama_pooling_type(ctx);
+        printf("pooling_type: %d\n", pooling_type);
+        if (pooling_type == LLAMA_POOLING_TYPE_NONE) {
             data = llama_get_embeddings(ctx);
-        }
-        else {
+        } else {
             data = llama_get_embeddings_seq(ctx, 0);
         }
-        if(!data) {
+        if (!data) {
             return std::vector<float>(n_embd, 0.0f);
         }
         std::vector<float> embedding(data, data + n_embd), out(data, data + n_embd);
         common_embd_normalize(embedding.data(), out.data(), n_embd, params.embd_normalize);
         return out;