npm - cui-llama.rn - Versions diffs - 1.6.0 → 1.6.1 - Mend

cui-llama.rn 1.6.0 → 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (195) hide show

package/README.md CHANGED Viewed

@@ -123,22 +123,50 @@ console.log('Result:', textResult.text)
 console.log('Timings:', textResult.timings)
 ```
-The binding’s deisgn inspired by [server.cpp](https://github.com/ggerganov/llama.cpp/tree/master/examples/server) example in llama.cpp, so you can map its API to LlamaContext:
+The binding’s deisgn inspired by [server.cpp](https://github.com/ggerganov/llama.cpp/tree/master/examples/server) example in llama.cpp:
 - `/completion` and `/chat/completions`: `context.completion(params, partialCompletionCallback)`
 - `/tokenize`: `context.tokenize(content)`
 - `/detokenize`: `context.detokenize(tokens)`
 - `/embedding`: `context.embedding(content)`
-- Other methods
-  - `context.loadSession(path)`
-  - `context.saveSession(path)`
-  - `context.stopCompletion()`
-  - `context.release()`
+- ... Other methods
 Please visit the [Documentation](docs/API) for more details.
 You can also visit the [example](example) to see how to use it.
+## Session (State)
+The session file is a binary file that contains the state of the context, it can saves time of prompt processing.
+```js
+const context = await initLlama({ ...params })
+// After prompt processing or completion ...
+// Save the session
+await context.saveSession('<path to save session>')
+// Load the session
+await context.loadSession('<path to load session>')
+```
+## Embedding
+The embedding API is used to get the embedding of a text.
+```js
+const context = await initLlama({
+  ...params,
+  embedding: true,
+})
+const { embedding } = await context.embedding('Hello, world!')
+```
+- You can use model like [nomic-ai/nomic-embed-text-v1.5-GGUF](https://huggingface.co/nomic-ai/nomic-embed-text-v1.5-GGUF) for better embedding quality.
+- You can use DB like [op-sqlite](https://github.com/OP-Engineering/op-sqlite) with sqlite-vec support to store and search embeddings.
 ## Tool Calling
 `llama.rn` has universal tool call support by using [minja](https://github.com/google/minja) (as Jinja template parser) and [chat.cpp](https://github.com/ggerganov/llama.cpp/blob/master/common/chat.cpp) in llama.cpp.
@@ -273,7 +301,7 @@ jest.mock('llama.rn', () => require('llama.rn/jest/mock'))
 iOS:
-- The [Extended Virtual Addressing](https://developer.apple.com/documentation/bundleresources/entitlements/com_apple_developer_kernel_extended-virtual-addressing) capability is recommended to enable on iOS project.
+- The [Extended Virtual Addressing](https://developer.apple.com/documentation/bundleresources/entitlements/com_apple_developer_kernel_extended-virtual-addressing) and [Increased Memory Limit](https://developer.apple.com/documentation/bundleresources/entitlements/com.apple.developer.kernel.increased-memory-limit?language=objc) capabilities are recommended to enable on iOS project.
 - Metal:
   - We have tested to know some devices is not able to use Metal (GPU) due to llama.cpp used SIMD-scoped operation, you can check if your device is supported in [Metal feature set tables](https://developer.apple.com/metal/Metal-Feature-Set-Tables.pdf), Apple7 GPU will be the minimum requirement.
   - It's also not supported in iOS simulator due to [this limitation](https://developer.apple.com/documentation/metal/developing_metal_apps_that_run_in_simulator#3241609), we used constant buffers more than 14.

package/android/src/main/CMakeLists.txt CHANGED Viewed

@@ -11,7 +11,10 @@ endif(CCACHE_FOUND)
 set(CMAKE_CXX_STANDARD 17)
 set(RNLLAMA_LIB_DIR ${CMAKE_SOURCE_DIR}/../../../cpp)
-include_directories(${RNLLAMA_LIB_DIR})
+include_directories(
+    ${RNLLAMA_LIB_DIR}
+    ${RNLLAMA_LIB_DIR}/ggml-cpu
+)
 set(
     SOURCE_FILES
@@ -19,15 +22,18 @@ set(
     ${RNLLAMA_LIB_DIR}/ggml-alloc.c
     ${RNLLAMA_LIB_DIR}/ggml-backend.cpp
     ${RNLLAMA_LIB_DIR}/ggml-backend-reg.cpp
-    ${RNLLAMA_LIB_DIR}/ops.cpp
-    ${RNLLAMA_LIB_DIR}/unary-ops.cpp
-    ${RNLLAMA_LIB_DIR}/binary-ops.cpp
-    ${RNLLAMA_LIB_DIR}/vec.cpp
-    ${RNLLAMA_LIB_DIR}/ggml-cpu.c
-    ${RNLLAMA_LIB_DIR}/ggml-cpu.cpp
-    ${RNLLAMA_LIB_DIR}/ggml-cpu-aarch64.cpp
-    ${RNLLAMA_LIB_DIR}/ggml-cpu-quants.c
-    ${RNLLAMA_LIB_DIR}/ggml-cpu-traits.cpp
+    ${RNLLAMA_LIB_DIR}/ggml-cpu/amx/amx.cpp
+    ${RNLLAMA_LIB_DIR}/ggml-cpu/amx/mmq.cpp
+    ${RNLLAMA_LIB_DIR}/ggml-cpu/ggml-cpu.c
+    ${RNLLAMA_LIB_DIR}/ggml-cpu/ggml-cpu.cpp
+    ${RNLLAMA_LIB_DIR}/ggml-cpu/ggml-cpu-aarch64.cpp
+    ${RNLLAMA_LIB_DIR}/ggml-cpu/ggml-cpu-quants.c
+    ${RNLLAMA_LIB_DIR}/ggml-cpu/ggml-cpu-traits.cpp
+    ${RNLLAMA_LIB_DIR}/ggml-cpu/unary-ops.cpp
+    ${RNLLAMA_LIB_DIR}/ggml-cpu/binary-ops.cpp
+    ${RNLLAMA_LIB_DIR}/ggml-cpu/sgemm.cpp
+    ${RNLLAMA_LIB_DIR}/ggml-cpu/vec.cpp
+    ${RNLLAMA_LIB_DIR}/ggml-cpu/ops.cpp
     ${RNLLAMA_LIB_DIR}/ggml-opt.cpp
     ${RNLLAMA_LIB_DIR}/ggml-threading.cpp
     ${RNLLAMA_LIB_DIR}/ggml-quants.c
@@ -56,7 +62,6 @@ set(
     ${RNLLAMA_LIB_DIR}/sampling.cpp
     ${RNLLAMA_LIB_DIR}/unicode-data.cpp
     ${RNLLAMA_LIB_DIR}/unicode.cpp
-    ${RNLLAMA_LIB_DIR}/sgemm.cpp
     ${RNLLAMA_LIB_DIR}/common.cpp
     ${RNLLAMA_LIB_DIR}/chat.cpp
     ${RNLLAMA_LIB_DIR}/json-schema-to-grammar.cpp

package/android/src/main/java/com/rnllama/LlamaContext.java CHANGED Viewed

@@ -170,6 +170,8 @@ public class LlamaContext {
       params.hasKey("rope_freq_scale") ? (float) params.getDouble("rope_freq_scale") : 0.0f,
       // int pooling_type,
       params.hasKey("pooling_type") ? params.getInt("pooling_type") : -1,
+      // boolean ctx_shift,
+      params.hasKey("ctx_shift") ? params.getBoolean("ctx_shift") : true,
       // LoadProgressCallback load_progress_callback
       params.hasKey("use_progress_callback") ? new LoadProgressCallback(this) : null
     );
@@ -536,7 +538,7 @@ public class LlamaContext {
     String[] skip
   );
   protected static native long initContext(
-    String model,
+    String model_path,
     String chat_template,
     String reasoning_format,
     boolean embedding,
@@ -558,6 +560,7 @@ public class LlamaContext {
     float rope_freq_base,
     float rope_freq_scale,
     int pooling_type,
+    boolean ctx_shift,
     LoadProgressCallback load_progress_callback
   );
   protected static native void interruptLoad(long contextPtr);

package/android/src/main/jni.cpp CHANGED Viewed

@@ -9,6 +9,7 @@
 #include <string>
 #include <thread>
 #include <unordered_map>
+#include "json.hpp"
 #include "json-schema-to-grammar.h"
 #include "llama.h"
 #include "chat.h"
@@ -252,6 +253,7 @@ Java_com_rnllama_LlamaContext_initContext(
     jfloat rope_freq_base,
     jfloat rope_freq_scale,
     jint pooling_type,
+    jboolean ctx_shift,
     jobject load_progress_callback
 ) {
     UNUSED(thiz);
@@ -264,7 +266,7 @@ Java_com_rnllama_LlamaContext_initContext(
     }
     const char *model_path_chars = env->GetStringUTFChars(model_path_str, nullptr);
-    defaultParams.model = { model_path_chars };
+    defaultParams.model.path = model_path_chars;
     const char *chat_template_chars = env->GetStringUTFChars(chat_template, nullptr);
     defaultParams.chat_template = chat_template_chars;
@@ -279,6 +281,7 @@ Java_com_rnllama_LlamaContext_initContext(
     defaultParams.n_ctx = n_ctx;
     defaultParams.n_batch = n_batch;
     defaultParams.n_ubatch = n_ubatch;
+    defaultParams.ctx_shift = ctx_shift;
     if (pooling_type != -1) {
         defaultParams.pooling_type = static_cast<enum llama_pooling_type>(pooling_type);
@@ -298,7 +301,7 @@ Java_com_rnllama_LlamaContext_initContext(
     int default_n_threads = max_threads == 4 ? 2 : min(4, max_threads);
     defaultParams.cpuparams.n_threads = n_threads > 0 ? n_threads : default_n_threads;
-    // defaultParams.n_gpu_layers = n_gpu_layers;
+    defaultParams.n_gpu_layers = n_gpu_layers;
     defaultParams.flash_attn = flash_attn;
     const char *cache_type_k_chars = env->GetStringUTFChars(cache_type_k, nullptr);
@@ -534,9 +537,15 @@ Java_com_rnllama_LlamaContext_getFormattedChatWithJinja(
             pushString(env, additional_stops, stop.c_str());
         }
         putArray(env, result, "additional_stops", additional_stops);
+    } catch (const nlohmann::json_abi_v3_11_3::detail::parse_error& e) {
+        std::string errorMessage = "JSON parse error in getFormattedChat: " + std::string(e.what());
+        putString(env, result, "_error", errorMessage.c_str());
+        LOGI("[RNLlama] %s", errorMessage.c_str());
     } catch (const std::runtime_error &e) {
-        LOGI("[RNLlama] Error: %s", e.what());
         putString(env, result, "_error", e.what());
+        LOGI("[RNLlama] Error: %s", e.what());
+    } catch (...) {
+        putString(env, result, "_error", "Unknown error in getFormattedChat");
     }
     env->ReleaseStringUTFChars(tools, tools_chars);
     env->ReleaseStringUTFChars(messages, messages_chars);
@@ -855,6 +864,12 @@ Java_com_rnllama_LlamaContext_doCompletion(
     llama->beginCompletion();
     llama->loadPrompt();
+    if (llama->context_full) {
+        auto result = createWriteableMap(env);
+        putString(env, result, "error", "Context is full");
+        return reinterpret_cast<jobject>(result);
+    }
     size_t sent_count = 0;
     size_t sent_token_probs_index = 0;
@@ -945,7 +960,7 @@ Java_com_rnllama_LlamaContext_doCompletion(
                 toolCallsSize++;
             }
         } catch (const std::exception &e) {
-            // LOGI("Error parsing tool calls: %s", e.what());
+        } catch (...) {
         }
     }
@@ -964,6 +979,7 @@ Java_com_rnllama_LlamaContext_doCompletion(
     putInt(env, result, "tokens_predicted", llama->num_tokens_predicted);
     putInt(env, result, "tokens_evaluated", llama->num_prompt_tokens);
     putInt(env, result, "truncated", llama->truncated);
+    putBoolean(env, result, "context_full", llama->context_full);
     putInt(env, result, "stopped_eos", llama->stopped_eos);
     putInt(env, result, "stopped_word", llama->stopped_word);
     putInt(env, result, "stopped_limit", llama->stopped_limit);

package/android/src/main/jniLibs/arm64-v8a/librnllama.so CHANGED Viewed

Binary file

package/android/src/main/jniLibs/arm64-v8a/librnllama_v8.so CHANGED Viewed

Binary file

package/android/src/main/jniLibs/arm64-v8a/librnllama_v8_2.so CHANGED Viewed

Binary file

package/android/src/main/jniLibs/arm64-v8a/librnllama_v8_2_dotprod.so CHANGED Viewed

Binary file

package/android/src/main/jniLibs/arm64-v8a/librnllama_v8_2_dotprod_i8mm.so CHANGED Viewed

Binary file

package/android/src/main/jniLibs/arm64-v8a/librnllama_v8_2_i8mm.so CHANGED Viewed

Binary file

package/android/src/main/jniLibs/x86_64/librnllama.so CHANGED Viewed

Binary file

package/android/src/main/jniLibs/x86_64/librnllama_x86_64.so CHANGED Viewed

Binary file

package/cpp/LICENSE ADDED Viewed

@@ -0,0 +1,21 @@
+MIT License
+Copyright (c) 2023-2024 The ggml authors
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
+The above copyright notice and this permission notice shall be included in all
+copies or substantial portions of the Software.
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+SOFTWARE.

package/cpp/chat.cpp CHANGED Viewed

@@ -1612,7 +1612,7 @@ static common_chat_params common_chat_templates_apply_jinja(
     }
     // Hermes 2/3 Pro, Qwen 2.5 Instruct (w/ tools)
-    if (src.find("<tool_call>") != std::string::npos && params.json_schema.is_null()) {
+    if (src.find("<tool_call>") != std::string::npos && params.json_schema.is_null() && params.tools.is_array() && params.json_schema.is_null()) {
         return common_chat_params_init_hermes_2_pro(tmpl, params);
     }

package/cpp/common.cpp CHANGED Viewed

@@ -837,7 +837,7 @@ std::string fs_get_cache_directory() {
     if (getenv("LLAMA_CACHE")) {
         cache_directory = std::getenv("LLAMA_CACHE");
     } else {
-#ifdef __linux__
+#if defined(__linux__) || defined(__FreeBSD__) || defined(_AIX)
         if (std::getenv("XDG_CACHE_HOME")) {
             cache_directory = std::getenv("XDG_CACHE_HOME");
         } else {
@@ -847,7 +847,9 @@ std::string fs_get_cache_directory() {
         cache_directory = std::getenv("HOME") + std::string("/Library/Caches/");
 #elif defined(_WIN32)
         cache_directory = std::getenv("LOCALAPPDATA");
-#endif // __linux__
+#else
+#  error Unknown architecture
+#endif
         cache_directory = ensure_trailing_slash(cache_directory);
         cache_directory += "llama.cpp";
     }
@@ -1034,6 +1036,19 @@ struct common_init_result common_init_from_params(common_params & params) {
     return iparams;
 }
+std::string get_model_endpoint() {
+    const char * model_endpoint_env = getenv("MODEL_ENDPOINT");
+    // We still respect the use of environment-variable "HF_ENDPOINT" for backward-compatibility.
+    const char * hf_endpoint_env = getenv("HF_ENDPOINT");
+    const char * endpoint_env = model_endpoint_env ? model_endpoint_env : hf_endpoint_env;
+    std::string model_endpoint = "https://huggingface.co/";
+    if (endpoint_env) {
+        model_endpoint = endpoint_env;
+        if (model_endpoint.back() != '/') model_endpoint += '/';
+    }
+    return model_endpoint;
+}
 void common_set_adapter_lora(struct llama_context * ctx, std::vector<common_adapter_lora_info> & lora) {
     llama_clear_adapter_lora(ctx);
     for (auto & la : lora) {

package/cpp/common.h CHANGED Viewed

@@ -355,8 +355,10 @@ struct common_params {
     common_conversation_mode conversation_mode = COMMON_CONVERSATION_MODE_AUTO;
-    // multimodal models (see examples/llava)
+    // multimodal models (see tools/llava)
     struct common_params_model mmproj;
+    bool mmproj_use_gpu = true;     // use GPU for multimodal model
+    bool no_mmproj = false;         // explicitly disable multimodal model
     std::vector<std::string> image; // path to image file(s)
     // embedding
@@ -427,8 +429,8 @@ struct common_params {
     int n_pca_batch = 100;
     int n_pca_iterations = 1000;
     dimre_method cvector_dimre_method = DIMRE_METHOD_PCA;
-    std::string cvector_positive_file = "examples/cvector-generator/positive.txt";
-    std::string cvector_negative_file = "examples/cvector-generator/negative.txt";
+    std::string cvector_positive_file = "tools/cvector-generator/positive.txt";
+    std::string cvector_negative_file = "tools/cvector-generator/negative.txt";
     bool spm_infill = false; // suffix/prefix/middle pattern for infill
@@ -558,6 +560,8 @@ struct lm_ggml_threadpool_params lm_ggml_threadpool_params_from_cpu_params(const
 // clear LoRA adapters from context, then apply new list of adapters
 void common_set_adapter_lora(struct llama_context * ctx, std::vector<common_adapter_lora_info> & lora);
+std::string                   get_model_endpoint();
 //
 // Batch utils
 //

package/cpp/ggml-alloc.c CHANGED Viewed

@@ -816,7 +816,10 @@ static void lm_ggml_gallocr_init_tensor(lm_ggml_gallocr_t galloc, struct lm_ggml
 static bool lm_ggml_gallocr_node_needs_realloc(lm_ggml_gallocr_t galloc, struct lm_ggml_tensor * node, struct tensor_alloc * talloc) {
     size_t node_size = 0;
     if (!node->data && !node->view_src) {
-        LM_GGML_ASSERT(talloc->buffer_id >= 0); // prevent segfault when misusing the API
+        // If we previously had data but don't now then reallocate
+        if (talloc->buffer_id < 0) {
+            return false;
+        }
         node_size = lm_ggml_backend_buft_get_alloc_size(galloc->bufts[talloc->buffer_id], node);
     }
     return talloc->size_max >= node_size;

package/cpp/ggml-cpp.h CHANGED Viewed

@@ -24,7 +24,7 @@ typedef std::unique_ptr<lm_gguf_context, lm_gguf_context_deleter> lm_gguf_contex
 struct lm_ggml_gallocr_deleter { void operator()(lm_ggml_gallocr_t galloc) { lm_ggml_gallocr_free(galloc); } };
-typedef std::unique_ptr<lm_ggml_gallocr_t, lm_ggml_gallocr_deleter> lm_ggml_gallocr_ptr;
+typedef std::unique_ptr<lm_ggml_gallocr, lm_ggml_gallocr_deleter> lm_ggml_gallocr_ptr;
 // ggml-backend

package/cpp/ggml-cpu/amx/amx.cpp ADDED Viewed

@@ -0,0 +1,221 @@
+#include "amx.h"
+#include "common.h"
+#include "mmq.h"
+#include "ggml-backend-impl.h"
+#include "ggml-backend.h"
+#include "ggml-impl.h"
+#include "ggml-cpu.h"
+#include "ggml-cpu-traits.h"
+#if defined(__gnu_linux__)
+#include <sys/syscall.h>
+#include <unistd.h>
+#endif
+#include <cstdlib>
+#include <cstring>
+#include <memory>
+#if defined(__AMX_INT8__) && defined(__AVX512VNNI__)
+// AMX type_trais
+namespace ggml::cpu::amx {
+class tensor_traits : public ggml::cpu::tensor_traits {
+    bool work_size(int /* n_threads */, const struct lm_ggml_tensor * op, size_t & size) override {
+        size = lm_ggml_backend_amx_desired_wsize(op);
+        return true;
+    }
+    bool compute_forward(struct lm_ggml_compute_params * params, struct lm_ggml_tensor * op) override {
+        if (op->op == LM_GGML_OP_MUL_MAT) {
+            lm_ggml_backend_amx_mul_mat(params, op);
+            return true;
+        }
+        return false;
+    }
+};
+static ggml::cpu::tensor_traits * get_tensor_traits(lm_ggml_backend_buffer_t, struct lm_ggml_tensor *) {
+    static tensor_traits traits;
+    return &traits;
+}
+}  // namespace ggml::cpu::amx
+// AMX buffer interface
+static void lm_ggml_backend_amx_buffer_free_buffer(lm_ggml_backend_buffer_t buffer) {
+    free(buffer->context);
+}
+static void * lm_ggml_backend_amx_buffer_get_base(lm_ggml_backend_buffer_t buffer) {
+    return (void *) (buffer->context);
+}
+static enum lm_ggml_status lm_ggml_backend_amx_buffer_init_tensor(lm_ggml_backend_buffer_t buffer, struct lm_ggml_tensor * tensor) {
+    tensor->extra = (void *) ggml::cpu::amx::get_tensor_traits(buffer, tensor);
+    LM_GGML_UNUSED(buffer);
+    return LM_GGML_STATUS_SUCCESS;
+}
+static void lm_ggml_backend_amx_buffer_memset_tensor(lm_ggml_backend_buffer_t buffer, struct lm_ggml_tensor * tensor,
+                                                  uint8_t value, size_t offset, size_t size) {
+    memset((char *) tensor->data + offset, value, size);
+    LM_GGML_UNUSED(buffer);
+}
+static void lm_ggml_backend_amx_buffer_set_tensor(lm_ggml_backend_buffer_t buffer, struct lm_ggml_tensor * tensor,
+                                               const void * data, size_t offset, size_t size) {
+    if (qtype_has_amx_kernels(tensor->type)) {
+        LM_GGML_LOG_DEBUG("%s: amx repack tensor %s of type %s\n", __func__, tensor->name, lm_ggml_type_name(tensor->type));
+        lm_ggml_backend_amx_convert_weight(tensor, data, offset, size);
+    } else {
+        memcpy((char *) tensor->data + offset, data, size);
+    }
+    LM_GGML_UNUSED(buffer);
+}
+/*
+// need to figure what we need to do with buffer->extra.
+static void lm_ggml_backend_amx_buffer_get_tensor(lm_ggml_backend_buffer_t buffer, const struct lm_ggml_tensor * tensor, void * data, size_t offset, size_t size) {
+    LM_GGML_ASSERT(!qtype_has_amx_kernels(tensor->type));
+    memcpy(data, (const char *)tensor->data + offset, size);
+    LM_GGML_UNUSED(buffer);
+}
+static bool lm_ggml_backend_amx_buffer_cpy_tensor(lm_ggml_backend_buffer_t buffer, const struct lm_ggml_tensor * src, struct lm_ggml_tensor * dst) {
+    if (lm_ggml_backend_buffer_is_host(src->buffer)) {
+        if (qtype_has_amx_kernels(src->type)) {
+            lm_ggml_backend_amx_convert_weight(dst, src->data, 0, lm_ggml_nbytes(dst));
+        } else {
+            memcpy(dst->data, src->data, lm_ggml_nbytes(src));
+        }
+        return true;
+    }
+    return false;
+    LM_GGML_UNUSED(buffer);
+}
+*/
+static void lm_ggml_backend_amx_buffer_clear(lm_ggml_backend_buffer_t buffer, uint8_t value) {
+    memset(buffer->context, value, buffer->size);
+}
+static lm_ggml_backend_buffer_i lm_ggml_backend_amx_buffer_interface = {
+    /* .free_buffer     = */ lm_ggml_backend_amx_buffer_free_buffer,
+    /* .get_base        = */ lm_ggml_backend_amx_buffer_get_base,
+    /* .init_tensor     = */ lm_ggml_backend_amx_buffer_init_tensor,
+    /* .memset_tensor   = */ lm_ggml_backend_amx_buffer_memset_tensor,
+    /* .set_tensor      = */ lm_ggml_backend_amx_buffer_set_tensor,
+    /* .get_tensor      = */ nullptr,
+    /* .cpy_tensor      = */ nullptr,
+    /* .clear           = */ lm_ggml_backend_amx_buffer_clear,
+    /* .reset           = */ nullptr,
+};
+static const char * lm_ggml_backend_amx_buffer_type_get_name(lm_ggml_backend_buffer_type_t buft) {
+    return "AMX";
+    LM_GGML_UNUSED(buft);
+}
+static lm_ggml_backend_buffer_t lm_ggml_backend_amx_buffer_type_alloc_buffer(lm_ggml_backend_buffer_type_t buft, size_t size) {
+    void * data = lm_ggml_aligned_malloc(size);
+    if (data == NULL) {
+        fprintf(stderr, "%s: failed to allocate buffer of size %zu\n", __func__, size);
+        return NULL;
+    }
+    return lm_ggml_backend_buffer_init(buft, lm_ggml_backend_amx_buffer_interface, data, size);
+}
+static size_t lm_ggml_backend_amx_buffer_type_get_alignment(lm_ggml_backend_buffer_type_t buft) {
+    return TENSOR_ALIGNMENT;
+    LM_GGML_UNUSED(buft);
+}
+namespace ggml::cpu::amx {
+class extra_buffer_type : ggml::cpu::extra_buffer_type {
+    bool supports_op(lm_ggml_backend_dev_t, const struct lm_ggml_tensor * op) override {
+        // handle only 2d gemm for now
+        auto is_contiguous_2d = [](const struct lm_ggml_tensor * t) {
+            return lm_ggml_is_contiguous(t) && t->ne[3] == 1 && t->ne[2] == 1;
+        };
+        if (op->op == LM_GGML_OP_MUL_MAT && is_contiguous_2d(op->src[0]) &&  // src0 must be contiguous
+            is_contiguous_2d(op->src[1]) &&                               // src1 must be contiguous
+            op->src[0]->buffer && op->src[0]->buffer->buft == lm_ggml_backend_amx_buffer_type() &&
+            op->ne[0] % (TILE_N * 2) == 0 &&                              // out_features is 32x
+            (qtype_has_amx_kernels(op->src[0]->type) || (op->src[0]->type == LM_GGML_TYPE_F16))) {
+            // src1 must be host buffer
+            if (op->src[1]->buffer && !lm_ggml_backend_buft_is_host(op->src[1]->buffer->buft)) {
+                return false;
+            }
+            // src1 must be float32
+            if (op->src[1]->type == LM_GGML_TYPE_F32) {
+                return true;
+            }
+        }
+        return false;
+    }
+    ggml::cpu::tensor_traits * get_tensor_traits(const struct lm_ggml_tensor * op) override {
+        if (op->op == LM_GGML_OP_MUL_MAT && op->src[0]->buffer &&
+            op->src[0]->buffer->buft == lm_ggml_backend_amx_buffer_type()) {
+            return (ggml::cpu::tensor_traits *) op->src[0]->extra;
+        }
+        return nullptr;
+    }
+};
+}  // namespace ggml::cpu::amx
+static size_t lm_ggml_backend_amx_buffer_type_get_alloc_size(lm_ggml_backend_buffer_type_t buft, const lm_ggml_tensor * tensor) {
+    return lm_ggml_backend_amx_get_alloc_size(tensor);
+    LM_GGML_UNUSED(buft);
+}
+#define ARCH_GET_XCOMP_PERM     0x1022
+#define ARCH_REQ_XCOMP_PERM     0x1023
+#define XFEATURE_XTILECFG       17
+#define XFEATURE_XTILEDATA      18
+static bool lm_ggml_amx_init() {
+#if defined(__gnu_linux__)
+    if (syscall(SYS_arch_prctl, ARCH_REQ_XCOMP_PERM, XFEATURE_XTILEDATA)) {
+        fprintf(stderr, "AMX is not ready to be used!\n");
+        return false;
+    }
+    return true;
+#elif defined(_WIN32)
+    return true;
+#endif
+}
+lm_ggml_backend_buffer_type_t lm_ggml_backend_amx_buffer_type() {
+    static struct lm_ggml_backend_buffer_type lm_ggml_backend_buffer_type_amx = {
+        /* .iface = */ {
+                        /* .get_name         = */ lm_ggml_backend_amx_buffer_type_get_name,
+                        /* .alloc_buffer     = */ lm_ggml_backend_amx_buffer_type_alloc_buffer,
+                        /* .get_alignment    = */ lm_ggml_backend_amx_buffer_type_get_alignment,
+                        /* .get_max_size     = */ nullptr,  // defaults to SIZE_MAX
+                        /* .get_alloc_size   = */ lm_ggml_backend_amx_buffer_type_get_alloc_size,
+                        /* .is_host          = */ nullptr,
+                        },
+        /* .device  = */ lm_ggml_backend_reg_dev_get(lm_ggml_backend_cpu_reg(), 0),
+        /* .context = */ new ggml::cpu::amx::extra_buffer_type(),
+    };
+    if (!lm_ggml_amx_init()) {
+        return nullptr;
+    }
+    return &lm_ggml_backend_buffer_type_amx;
+}
+#endif  // defined(__AMX_INT8__) && defined(__AVX512VNNI__)

package/cpp/ggml-cpu/amx/amx.h ADDED Viewed

@@ -0,0 +1,8 @@
+#include "ggml-backend.h"
+#include "ggml-cpu-impl.h"
+// GGML internal header
+#if defined(__AMX_INT8__) && defined(__AVX512VNNI__)
+lm_ggml_backend_buffer_type_t lm_ggml_backend_amx_buffer_type(void);
+#endif