npm - cui-llama.rn - Versions diffs - 1.3.0 → 1.3.3 - Mend

cui-llama.rn 1.3.0 → 1.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (76) hide show

package/android/src/main/CMakeLists.txt +6 -1
package/android/src/main/jni.cpp +6 -6
package/cpp/amx/amx.cpp +196 -0
package/cpp/amx/amx.h +20 -0
package/cpp/amx/common.h +101 -0
package/cpp/amx/mmq.cpp +2524 -0
package/cpp/amx/mmq.h +16 -0
package/cpp/common.cpp +1981 -1682
package/cpp/common.h +636 -600
package/cpp/ggml-aarch64.c +129 -129
package/cpp/ggml-aarch64.h +19 -19
package/cpp/ggml-alloc.c +1038 -1040
package/cpp/ggml-alloc.h +76 -76
package/cpp/ggml-backend-impl.h +238 -216
package/cpp/ggml-backend-reg.cpp +423 -195
package/cpp/ggml-backend.cpp +1999 -1997
package/cpp/ggml-backend.h +351 -328
package/cpp/ggml-common.h +1859 -1853
package/cpp/ggml-cpp.h +38 -38
package/cpp/ggml-cpu-aarch64.c +3823 -3560
package/cpp/ggml-cpu-aarch64.h +32 -30
package/cpp/ggml-cpu-impl.h +386 -371
package/cpp/ggml-cpu-quants.c +10835 -10822
package/cpp/ggml-cpu-quants.h +63 -63
package/cpp/ggml-cpu.c +99 -103
package/cpp/ggml-cpu.cpp +69 -17
package/cpp/ggml-cpu.h +152 -177
package/cpp/ggml-impl.h +556 -550
package/cpp/ggml-metal.h +66 -66
package/cpp/ggml-metal.m +4426 -4294
package/cpp/ggml-quants.c +5247 -5247
package/cpp/ggml-quants.h +100 -100
package/cpp/ggml-threading.cpp +12 -12
package/cpp/ggml-threading.h +12 -12
package/cpp/ggml.c +7618 -8180
package/cpp/ggml.h +2255 -2411
package/cpp/json-schema-to-grammar.cpp +1045 -0
package/cpp/json-schema-to-grammar.h +8 -0
package/cpp/json.hpp +24766 -0
package/cpp/llama-grammar.cpp +1138 -1138
package/cpp/llama-grammar.h +144 -144
package/cpp/llama-impl.h +181 -181
package/cpp/llama-sampling.cpp +2348 -2348
package/cpp/llama-sampling.h +48 -48
package/cpp/llama-vocab.cpp +1984 -1984
package/cpp/llama-vocab.h +170 -170
package/cpp/llama.cpp +22332 -22132
package/cpp/llama.h +1259 -1253
package/cpp/log.cpp +401 -401
package/cpp/log.h +121 -121
package/cpp/rn-llama.hpp +6 -6
package/cpp/sampling.cpp +505 -466
package/cpp/sampling.h +22 -1
package/cpp/sgemm.cpp +1884 -1884
package/cpp/speculative.cpp +270 -0
package/cpp/speculative.h +28 -0
package/cpp/unicode.cpp +11 -0
package/ios/RNLlamaContext.mm +13 -0
package/lib/commonjs/NativeRNLlama.js.map +1 -1
package/lib/commonjs/grammar.js +4 -2
package/lib/commonjs/grammar.js.map +1 -1
package/lib/commonjs/index.js.map +1 -1
package/lib/module/NativeRNLlama.js.map +1 -1
package/lib/module/grammar.js +2 -1
package/lib/module/grammar.js.map +1 -1
package/lib/module/index.js.map +1 -1
package/lib/typescript/NativeRNLlama.d.ts +94 -4
package/lib/typescript/NativeRNLlama.d.ts.map +1 -1
package/lib/typescript/grammar.d.ts +5 -6
package/lib/typescript/grammar.d.ts.map +1 -1
package/lib/typescript/index.d.ts +4 -2
package/lib/typescript/index.d.ts.map +1 -1
package/package.json +2 -1
package/src/NativeRNLlama.ts +97 -10
package/src/grammar.ts +10 -8
package/src/index.ts +22 -1

package/cpp/ggml-alloc.h CHANGED Viewed

@@ -1,76 +1,76 @@
-#pragma once
-#include "ggml.h"
-#ifdef  __cplusplus
-extern "C" {
-#endif
-typedef struct lm_ggml_backend_buffer_type * lm_ggml_backend_buffer_type_t;
-typedef struct      lm_ggml_backend_buffer * lm_ggml_backend_buffer_t;
-typedef struct             lm_ggml_backend * lm_ggml_backend_t;
-// Tensor allocator
-struct lm_ggml_tallocr {
-    lm_ggml_backend_buffer_t buffer;
-    void * base;
-    size_t alignment;
-    size_t offset;
-};
-LM_GGML_API struct lm_ggml_tallocr lm_ggml_tallocr_new(lm_ggml_backend_buffer_t buffer);
-LM_GGML_API void                lm_ggml_tallocr_alloc(struct lm_ggml_tallocr * talloc, struct lm_ggml_tensor * tensor);
-// Graph allocator
-/*
-  Example usage:
-    lm_ggml_gallocr_t galloc = lm_ggml_gallocr_new(lm_ggml_backend_cpu_buffer_type());
-    // optional: create a worst-case graph and reserve the buffers to avoid reallocations
-    lm_ggml_gallocr_reserve(galloc, build_graph(max_batch));
-    // allocate the graph
-    struct lm_ggml_cgraph * graph = build_graph(batch);
-    lm_ggml_gallocr_alloc_graph(galloc, graph);
-    printf("compute buffer size: %zu bytes\n", lm_ggml_gallocr_get_buffer_size(galloc, 0));
-    // evaluate the graph
-    lm_ggml_backend_graph_compute(backend, graph);
-*/
-// special tensor flags for use with the graph allocator:
-//   lm_ggml_set_input(): all input tensors are allocated at the beginning of the graph in non-overlapping addresses
-//   lm_ggml_set_output(): output tensors are never freed and never overwritten
-typedef struct lm_ggml_gallocr * lm_ggml_gallocr_t;
-LM_GGML_API lm_ggml_gallocr_t lm_ggml_gallocr_new(lm_ggml_backend_buffer_type_t buft);
-LM_GGML_API lm_ggml_gallocr_t lm_ggml_gallocr_new_n(lm_ggml_backend_buffer_type_t * bufts, int n_bufs);
-LM_GGML_API void           lm_ggml_gallocr_free(lm_ggml_gallocr_t galloc);
-// pre-allocate buffers from a measure graph - does not allocate or modify the graph
-// call with a worst-case graph to avoid buffer reallocations
-// not strictly required for single buffer usage: lm_ggml_gallocr_alloc_graph will reallocate the buffers automatically if needed
-// returns false if the buffer allocation failed
-LM_GGML_API bool lm_ggml_gallocr_reserve(lm_ggml_gallocr_t galloc, struct lm_ggml_cgraph * graph);
-LM_GGML_API bool lm_ggml_gallocr_reserve_n(
-    lm_ggml_gallocr_t galloc,
-    struct lm_ggml_cgraph * graph,
-    const int * node_buffer_ids,
-    const int * leaf_buffer_ids);
-// automatic reallocation if the topology changes when using a single buffer
-// returns false if using multiple buffers and a re-allocation is needed (call lm_ggml_gallocr_reserve_n first to set the node buffers)
-LM_GGML_API bool lm_ggml_gallocr_alloc_graph(lm_ggml_gallocr_t galloc, struct lm_ggml_cgraph * graph);
-LM_GGML_API size_t lm_ggml_gallocr_get_buffer_size(lm_ggml_gallocr_t galloc, int buffer_id);
-// Utils
-// Create a buffer and allocate all the tensors in a lm_ggml_context
-LM_GGML_API struct lm_ggml_backend_buffer * lm_ggml_backend_alloc_ctx_tensors_from_buft(struct lm_ggml_context * ctx, lm_ggml_backend_buffer_type_t buft);
-LM_GGML_API struct lm_ggml_backend_buffer * lm_ggml_backend_alloc_ctx_tensors(struct lm_ggml_context * ctx, lm_ggml_backend_t backend);
-#ifdef  __cplusplus
-}
-#endif
+#pragma once
+#include "ggml.h"
+#ifdef  __cplusplus
+extern "C" {
+#endif
+typedef struct lm_ggml_backend_buffer_type * lm_ggml_backend_buffer_type_t;
+typedef struct      lm_ggml_backend_buffer * lm_ggml_backend_buffer_t;
+typedef struct             lm_ggml_backend * lm_ggml_backend_t;
+// Tensor allocator
+struct lm_ggml_tallocr {
+    lm_ggml_backend_buffer_t buffer;
+    void * base;
+    size_t alignment;
+    size_t offset;
+};
+LM_GGML_API struct lm_ggml_tallocr lm_ggml_tallocr_new(lm_ggml_backend_buffer_t buffer);
+LM_GGML_API void                lm_ggml_tallocr_alloc(struct lm_ggml_tallocr * talloc, struct lm_ggml_tensor * tensor);
+// Graph allocator
+/*
+  Example usage:
+    lm_ggml_gallocr_t galloc = lm_ggml_gallocr_new(lm_ggml_backend_cpu_buffer_type());
+    // optional: create a worst-case graph and reserve the buffers to avoid reallocations
+    lm_ggml_gallocr_reserve(galloc, build_graph(max_batch));
+    // allocate the graph
+    struct lm_ggml_cgraph * graph = build_graph(batch);
+    lm_ggml_gallocr_alloc_graph(galloc, graph);
+    printf("compute buffer size: %zu bytes\n", lm_ggml_gallocr_get_buffer_size(galloc, 0));
+    // evaluate the graph
+    lm_ggml_backend_graph_compute(backend, graph);
+*/
+// special tensor flags for use with the graph allocator:
+//   lm_ggml_set_input(): all input tensors are allocated at the beginning of the graph in non-overlapping addresses
+//   lm_ggml_set_output(): output tensors are never freed and never overwritten
+typedef struct lm_ggml_gallocr * lm_ggml_gallocr_t;
+LM_GGML_API lm_ggml_gallocr_t lm_ggml_gallocr_new(lm_ggml_backend_buffer_type_t buft);
+LM_GGML_API lm_ggml_gallocr_t lm_ggml_gallocr_new_n(lm_ggml_backend_buffer_type_t * bufts, int n_bufs);
+LM_GGML_API void           lm_ggml_gallocr_free(lm_ggml_gallocr_t galloc);
+// pre-allocate buffers from a measure graph - does not allocate or modify the graph
+// call with a worst-case graph to avoid buffer reallocations
+// not strictly required for single buffer usage: lm_ggml_gallocr_alloc_graph will reallocate the buffers automatically if needed
+// returns false if the buffer allocation failed
+LM_GGML_API bool lm_ggml_gallocr_reserve(lm_ggml_gallocr_t galloc, struct lm_ggml_cgraph * graph);
+LM_GGML_API bool lm_ggml_gallocr_reserve_n(
+    lm_ggml_gallocr_t galloc,
+    struct lm_ggml_cgraph * graph,
+    const int * node_buffer_ids,
+    const int * leaf_buffer_ids);
+// automatic reallocation if the topology changes when using a single buffer
+// returns false if using multiple buffers and a re-allocation is needed (call lm_ggml_gallocr_reserve_n first to set the node buffers)
+LM_GGML_API bool lm_ggml_gallocr_alloc_graph(lm_ggml_gallocr_t galloc, struct lm_ggml_cgraph * graph);
+LM_GGML_API size_t lm_ggml_gallocr_get_buffer_size(lm_ggml_gallocr_t galloc, int buffer_id);
+// Utils
+// Create a buffer and allocate all the tensors in a lm_ggml_context
+LM_GGML_API struct lm_ggml_backend_buffer * lm_ggml_backend_alloc_ctx_tensors_from_buft(struct lm_ggml_context * ctx, lm_ggml_backend_buffer_type_t buft);
+LM_GGML_API struct lm_ggml_backend_buffer * lm_ggml_backend_alloc_ctx_tensors(struct lm_ggml_context * ctx, lm_ggml_backend_t backend);
+#ifdef  __cplusplus
+}
+#endif

package/cpp/ggml-backend-impl.h CHANGED Viewed

@@ -1,216 +1,238 @@
-#pragma once
-// ggml-backend internal header
-#include "ggml-backend.h"
-#ifdef  __cplusplus
-extern "C" {
-#endif
-    //
-    // Backend buffer type
-    //
-    struct lm_ggml_backend_buffer_type_i {
-        const char *          (*get_name)      (lm_ggml_backend_buffer_type_t buft);
-        // allocate a buffer of this type
-        lm_ggml_backend_buffer_t (*alloc_buffer)  (lm_ggml_backend_buffer_type_t buft, size_t size);
-        // tensor alignment
-        size_t                (*get_alignment) (lm_ggml_backend_buffer_type_t buft);
-        // (optional) max buffer size that can be allocated (defaults to SIZE_MAX)
-        size_t                (*get_max_size)  (lm_ggml_backend_buffer_type_t buft);
-        // (optional) data size needed to allocate the tensor, including padding (defaults to lm_ggml_nbytes)
-        size_t                (*get_alloc_size)(lm_ggml_backend_buffer_type_t buft, const struct lm_ggml_tensor * tensor);
-        // (optional) check if tensor data is in host memory and uses standard ggml tensor layout (defaults to false)
-        bool                  (*is_host)       (lm_ggml_backend_buffer_type_t buft);
-    };
-    struct lm_ggml_backend_buffer_type {
-        struct lm_ggml_backend_buffer_type_i  iface;
-        lm_ggml_backend_dev_t device;
-        void * context;
-    };
-    //
-    // Backend buffer
-    //
-    struct lm_ggml_backend_buffer_i {
-        // (optional) free the buffer
-        void         (*free_buffer)  (lm_ggml_backend_buffer_t buffer);
-        // base address of the buffer
-        void *       (*get_base)     (lm_ggml_backend_buffer_t buffer);
-        // (optional) initialize a tensor in the buffer (eg. add tensor extras)
-        void         (*init_tensor)  (lm_ggml_backend_buffer_t buffer, struct lm_ggml_tensor * tensor);
-        // tensor data access
-        void         (*memset_tensor)(lm_ggml_backend_buffer_t buffer,       struct lm_ggml_tensor * tensor,     uint8_t value, size_t offset, size_t size);
-        void         (*set_tensor)   (lm_ggml_backend_buffer_t buffer,       struct lm_ggml_tensor * tensor, const void * data, size_t offset, size_t size);
-        void         (*get_tensor)   (lm_ggml_backend_buffer_t buffer, const struct lm_ggml_tensor * tensor,       void * data, size_t offset, size_t size);
-        // (optional) tensor copy: dst is in the buffer, src may be in any buffer, including buffers from a different backend (return false if not supported)
-        bool         (*cpy_tensor)   (lm_ggml_backend_buffer_t buffer, const struct lm_ggml_tensor * src, struct lm_ggml_tensor * dst);
-        // clear the entire buffer
-        void         (*clear)        (lm_ggml_backend_buffer_t buffer, uint8_t value);
-        // (optional) reset any internal state due to tensor initialization, such as tensor extras
-        void         (*reset)        (lm_ggml_backend_buffer_t buffer);
-    };
-    struct lm_ggml_backend_buffer {
-        struct lm_ggml_backend_buffer_i  iface;
-        lm_ggml_backend_buffer_type_t    buft;
-        void * context;
-        size_t size;
-        enum lm_ggml_backend_buffer_usage usage;
-    };
-    lm_ggml_backend_buffer_t lm_ggml_backend_buffer_init(
-                   lm_ggml_backend_buffer_type_t buft,
-            struct lm_ggml_backend_buffer_i      iface,
-                   void *                     context,
-                   size_t                     size);
-    // do not use directly, use lm_ggml_backend_tensor_copy instead
-    bool lm_ggml_backend_buffer_copy_tensor(const struct lm_ggml_tensor * src, struct lm_ggml_tensor * dst);
-    // multi-buffer
-    // buffer that contains a collection of buffers
-    lm_ggml_backend_buffer_t lm_ggml_backend_multi_buffer_alloc_buffer(lm_ggml_backend_buffer_t * buffers, size_t n_buffers);
-    bool                  lm_ggml_backend_buffer_is_multi_buffer(lm_ggml_backend_buffer_t buffer);
-    void                  lm_ggml_backend_multi_buffer_set_usage(lm_ggml_backend_buffer_t buffer, enum lm_ggml_backend_buffer_usage usage);
-    //
-    // Backend (stream)
-    //
-    struct lm_ggml_backend_i {
-        const char * (*get_name)(lm_ggml_backend_t backend);
-        void (*free)(lm_ggml_backend_t backend);
-        // (optional) asynchronous tensor data access
-        void (*set_tensor_async)(lm_ggml_backend_t backend,       struct lm_ggml_tensor * tensor, const void * data, size_t offset, size_t size);
-        void (*get_tensor_async)(lm_ggml_backend_t backend, const struct lm_ggml_tensor * tensor,       void * data, size_t offset, size_t size);
-        bool (*cpy_tensor_async)(lm_ggml_backend_t backend_src, lm_ggml_backend_t backend_dst, const struct lm_ggml_tensor * src, struct lm_ggml_tensor * dst);
-        // (optional) complete all pending operations (required if the backend supports async operations)
-        void (*synchronize)(lm_ggml_backend_t backend);
-        // (optional) graph plans (not used currently)
-        // compute graph with a plan
-        lm_ggml_backend_graph_plan_t (*graph_plan_create) (lm_ggml_backend_t backend, const struct lm_ggml_cgraph * cgraph);
-        void                      (*graph_plan_free)   (lm_ggml_backend_t backend, lm_ggml_backend_graph_plan_t plan);
-        // update the plan with a new graph - this should be faster than creating a new plan when the graph has the same topology
-        void                      (*graph_plan_update) (lm_ggml_backend_t backend, lm_ggml_backend_graph_plan_t plan, const struct lm_ggml_cgraph * cgraph);
-        // compute the graph with the plan
-        enum lm_ggml_status          (*graph_plan_compute)(lm_ggml_backend_t backend, lm_ggml_backend_graph_plan_t plan);
-        // compute graph (always async if supported by the backend)
-        enum lm_ggml_status          (*graph_compute)     (lm_ggml_backend_t backend, struct lm_ggml_cgraph * cgraph);
-        // (optional) event synchronization
-        // record an event on this stream
-        void (*event_record)(lm_ggml_backend_t backend, lm_ggml_backend_event_t event);
-        // wait for an event on on a different stream
-        void (*event_wait)  (lm_ggml_backend_t backend, lm_ggml_backend_event_t event);
-    };
-    struct lm_ggml_backend {
-        lm_ggml_guid_t guid;
-        struct lm_ggml_backend_i iface;
-        lm_ggml_backend_dev_t device;
-        void * context;
-    };
-    struct lm_ggml_backend_event {
-        struct lm_ggml_backend_device * device;
-        void * context;
-    };
-    //
-    // Backend device
-    //
-    // Note: if additional properties are needed, we should add a struct with all of them
-    //       the current functions to obtain the properties can remain, since they are more convenient for often used properties
-    struct lm_ggml_backend_device_i {
-        // device name: short identifier for this device, such as "CPU" or "CUDA0"
-        const char * (*get_name)(lm_ggml_backend_dev_t dev);
-        // device description: short informative description of the device, could be the model name
-        const char * (*get_description)(lm_ggml_backend_dev_t dev);
-        // device memory in bytes
-        void         (*get_memory)(lm_ggml_backend_dev_t dev, size_t * free, size_t * total);
-        // device type
-        enum lm_ggml_backend_dev_type (*get_type)(lm_ggml_backend_dev_t dev);
-        // device properties
-        void (*get_props)(lm_ggml_backend_dev_t dev, struct lm_ggml_backend_dev_props * props);
-        // backend (stream) initialization
-        lm_ggml_backend_t (*init_backend)(lm_ggml_backend_dev_t dev, const char * params);
-        // preferred buffer type
-        lm_ggml_backend_buffer_type_t (*get_buffer_type)(lm_ggml_backend_dev_t dev);
-        // (optional) host buffer type (in system memory, typically this is a pinned memory buffer for faster transfers between host and device)
-        lm_ggml_backend_buffer_type_t (*get_host_buffer_type)(lm_ggml_backend_dev_t dev);
-        // (optional) buffer from pointer: create a buffer from a host pointer (useful for memory mapped models and importing data from other libraries)
-        lm_ggml_backend_buffer_t (*buffer_from_host_ptr)(lm_ggml_backend_dev_t dev, void * ptr, size_t size, size_t max_tensor_size);
-        // check if the backend can compute an operation
-        bool (*supports_op)(lm_ggml_backend_dev_t dev, const struct lm_ggml_tensor * op);
-        // check if the backend can use tensors allocated in a buffer type
-        bool (*supports_buft)(lm_ggml_backend_dev_t dev, lm_ggml_backend_buffer_type_t buft);
-        // (optional) check if the backend wants to run an operation, even if the weights are allocated in an incompatible buffer
-        // these should be expensive operations that may benefit from running on this backend instead of the CPU backend
-        bool (*offload_op)(lm_ggml_backend_dev_t dev, const struct lm_ggml_tensor * op);
-        // (optional) event synchronization
-        lm_ggml_backend_event_t (*event_new)         (lm_ggml_backend_dev_t dev);
-        void                 (*event_free)        (lm_ggml_backend_dev_t dev, lm_ggml_backend_event_t event);
-        void                 (*event_synchronize) (lm_ggml_backend_dev_t dev, lm_ggml_backend_event_t event);
-    };
-    struct lm_ggml_backend_device {
-        struct lm_ggml_backend_device_i iface;
-        lm_ggml_backend_reg_t reg;
-        void * context;
-    };
-    //
-    // Backend (reg)
-    //
-    struct lm_ggml_backend_reg_i {
-        const char * (*get_name)(lm_ggml_backend_reg_t reg);
-        // enumerate available devices
-        size_t             (*get_device_count)(lm_ggml_backend_reg_t reg);
-        lm_ggml_backend_dev_t (*get_device)(lm_ggml_backend_reg_t reg, size_t index);
-        // (optional) get a pointer to a function in the backend
-        // backends can add custom functions that are not part of the standard ggml-backend interface
-        void * (*get_proc_address)(lm_ggml_backend_reg_t reg, const char * name);
-    };
-    struct lm_ggml_backend_reg {
-        // int api_version; // TODO: for dynamic loading
-        struct lm_ggml_backend_reg_i iface;
-        void * context;
-    };
-    // Internal backend registry API
-    void lm_ggml_backend_register(lm_ggml_backend_reg_t reg);
-    void lm_ggml_backend_device_register(lm_ggml_backend_dev_t device);
-    // TODO: backends can be loaded as a dynamic library, in which case it needs to export this function
-    // typedef lm_ggml_backend_register_t * (*lm_ggml_backend_init)(void);
-#ifdef  __cplusplus
-}
-#endif
+#pragma once
+// ggml-backend internal header
+#include "ggml-backend.h"
+#ifdef  __cplusplus
+extern "C" {
+#endif
+    #define LM_GGML_BACKEND_API_VERSION 1
+    //
+    // Backend buffer type
+    //
+    struct lm_ggml_backend_buffer_type_i {
+        const char *          (*get_name)      (lm_ggml_backend_buffer_type_t buft);
+        // allocate a buffer of this type
+        lm_ggml_backend_buffer_t (*alloc_buffer)  (lm_ggml_backend_buffer_type_t buft, size_t size);
+        // tensor alignment
+        size_t                (*get_alignment) (lm_ggml_backend_buffer_type_t buft);
+        // (optional) max buffer size that can be allocated (defaults to SIZE_MAX)
+        size_t                (*get_max_size)  (lm_ggml_backend_buffer_type_t buft);
+        // (optional) data size needed to allocate the tensor, including padding (defaults to lm_ggml_nbytes)
+        size_t                (*get_alloc_size)(lm_ggml_backend_buffer_type_t buft, const struct lm_ggml_tensor * tensor);
+        // (optional) check if tensor data is in host memory and uses standard ggml tensor layout (defaults to false)
+        bool                  (*is_host)       (lm_ggml_backend_buffer_type_t buft);
+    };
+    struct lm_ggml_backend_buffer_type {
+        struct lm_ggml_backend_buffer_type_i  iface;
+        lm_ggml_backend_dev_t device;
+        void * context;
+    };
+    //
+    // Backend buffer
+    //
+    struct lm_ggml_backend_buffer_i {
+        // (optional) free the buffer
+        void         (*free_buffer)  (lm_ggml_backend_buffer_t buffer);
+        // base address of the buffer
+        void *       (*get_base)     (lm_ggml_backend_buffer_t buffer);
+        // (optional) initialize a tensor in the buffer (eg. add tensor extras)
+        void         (*init_tensor)  (lm_ggml_backend_buffer_t buffer, struct lm_ggml_tensor * tensor);
+        // tensor data access
+        void         (*memset_tensor)(lm_ggml_backend_buffer_t buffer,       struct lm_ggml_tensor * tensor,     uint8_t value, size_t offset, size_t size);
+        void         (*set_tensor)   (lm_ggml_backend_buffer_t buffer,       struct lm_ggml_tensor * tensor, const void * data, size_t offset, size_t size);
+        void         (*get_tensor)   (lm_ggml_backend_buffer_t buffer, const struct lm_ggml_tensor * tensor,       void * data, size_t offset, size_t size);
+        // (optional) tensor copy: dst is in the buffer, src may be in any buffer, including buffers from a different backend (return false if not supported)
+        bool         (*cpy_tensor)   (lm_ggml_backend_buffer_t buffer, const struct lm_ggml_tensor * src, struct lm_ggml_tensor * dst);
+        // clear the entire buffer
+        void         (*clear)        (lm_ggml_backend_buffer_t buffer, uint8_t value);
+        // (optional) reset any internal state due to tensor initialization, such as tensor extras
+        void         (*reset)        (lm_ggml_backend_buffer_t buffer);
+    };
+    struct lm_ggml_backend_buffer {
+        struct lm_ggml_backend_buffer_i  iface;
+        lm_ggml_backend_buffer_type_t    buft;
+        void * context;
+        size_t size;
+        enum lm_ggml_backend_buffer_usage usage;
+    };
+    LM_GGML_API lm_ggml_backend_buffer_t lm_ggml_backend_buffer_init(
+                   lm_ggml_backend_buffer_type_t buft,
+            struct lm_ggml_backend_buffer_i      iface,
+                   void *                     context,
+                   size_t                     size);
+    // do not use directly, use lm_ggml_backend_tensor_copy instead
+    LM_GGML_API bool lm_ggml_backend_buffer_copy_tensor(const struct lm_ggml_tensor * src, struct lm_ggml_tensor * dst);
+    // multi-buffer
+    // buffer that contains a collection of buffers
+    LM_GGML_API lm_ggml_backend_buffer_t lm_ggml_backend_multi_buffer_alloc_buffer(lm_ggml_backend_buffer_t * buffers, size_t n_buffers);
+    LM_GGML_API bool                  lm_ggml_backend_buffer_is_multi_buffer(lm_ggml_backend_buffer_t buffer);
+    LM_GGML_API void                  lm_ggml_backend_multi_buffer_set_usage(lm_ggml_backend_buffer_t buffer, enum lm_ggml_backend_buffer_usage usage);
+    //
+    // Backend (stream)
+    //
+    struct lm_ggml_backend_i {
+        const char * (*get_name)(lm_ggml_backend_t backend);
+        void (*free)(lm_ggml_backend_t backend);
+        // (optional) asynchronous tensor data access
+        void (*set_tensor_async)(lm_ggml_backend_t backend,       struct lm_ggml_tensor * tensor, const void * data, size_t offset, size_t size);
+        void (*get_tensor_async)(lm_ggml_backend_t backend, const struct lm_ggml_tensor * tensor,       void * data, size_t offset, size_t size);
+        bool (*cpy_tensor_async)(lm_ggml_backend_t backend_src, lm_ggml_backend_t backend_dst, const struct lm_ggml_tensor * src, struct lm_ggml_tensor * dst);
+        // (optional) complete all pending operations (required if the backend supports async operations)
+        void (*synchronize)(lm_ggml_backend_t backend);
+        // (optional) graph plans (not used currently)
+        // compute graph with a plan
+        lm_ggml_backend_graph_plan_t (*graph_plan_create) (lm_ggml_backend_t backend, const struct lm_ggml_cgraph * cgraph);
+        void                      (*graph_plan_free)   (lm_ggml_backend_t backend, lm_ggml_backend_graph_plan_t plan);
+        // update the plan with a new graph - this should be faster than creating a new plan when the graph has the same topology
+        void                      (*graph_plan_update) (lm_ggml_backend_t backend, lm_ggml_backend_graph_plan_t plan, const struct lm_ggml_cgraph * cgraph);
+        // compute the graph with the plan
+        enum lm_ggml_status          (*graph_plan_compute)(lm_ggml_backend_t backend, lm_ggml_backend_graph_plan_t plan);
+        // compute graph (always async if supported by the backend)
+        enum lm_ggml_status          (*graph_compute)     (lm_ggml_backend_t backend, struct lm_ggml_cgraph * cgraph);
+        // (optional) event synchronization
+        // record an event on this stream
+        void (*event_record)(lm_ggml_backend_t backend, lm_ggml_backend_event_t event);
+        // wait for an event on on a different stream
+        void (*event_wait)  (lm_ggml_backend_t backend, lm_ggml_backend_event_t event);
+    };
+    struct lm_ggml_backend {
+        lm_ggml_guid_t guid;
+        struct lm_ggml_backend_i iface;
+        lm_ggml_backend_dev_t device;
+        void * context;
+    };
+    struct lm_ggml_backend_event {
+        struct lm_ggml_backend_device * device;
+        void * context;
+    };
+    //
+    // Backend device
+    //
+    // Note: if additional properties are needed, we should add a struct with all of them
+    //       the current functions to obtain the properties can remain, since they are more convenient for often used properties
+    struct lm_ggml_backend_device_i {
+        // device name: short identifier for this device, such as "CPU" or "CUDA0"
+        const char * (*get_name)(lm_ggml_backend_dev_t dev);
+        // device description: short informative description of the device, could be the model name
+        const char * (*get_description)(lm_ggml_backend_dev_t dev);
+        // device memory in bytes
+        void         (*get_memory)(lm_ggml_backend_dev_t dev, size_t * free, size_t * total);
+        // device type
+        enum lm_ggml_backend_dev_type (*get_type)(lm_ggml_backend_dev_t dev);
+        // device properties
+        void (*get_props)(lm_ggml_backend_dev_t dev, struct lm_ggml_backend_dev_props * props);
+        // backend (stream) initialization
+        lm_ggml_backend_t (*init_backend)(lm_ggml_backend_dev_t dev, const char * params);
+        // preferred buffer type
+        lm_ggml_backend_buffer_type_t (*get_buffer_type)(lm_ggml_backend_dev_t dev);
+        // (optional) host buffer type (in system memory, typically this is a pinned memory buffer for faster transfers between host and device)
+        lm_ggml_backend_buffer_type_t (*get_host_buffer_type)(lm_ggml_backend_dev_t dev);
+        // (optional) buffer from pointer: create a buffer from a host pointer (useful for memory mapped models and importing data from other libraries)
+        lm_ggml_backend_buffer_t (*buffer_from_host_ptr)(lm_ggml_backend_dev_t dev, void * ptr, size_t size, size_t max_tensor_size);
+        // check if the backend can compute an operation
+        bool (*supports_op)(lm_ggml_backend_dev_t dev, const struct lm_ggml_tensor * op);
+        // check if the backend can use tensors allocated in a buffer type
+        bool (*supports_buft)(lm_ggml_backend_dev_t dev, lm_ggml_backend_buffer_type_t buft);
+        // (optional) check if the backend wants to run an operation, even if the weights are allocated in an incompatible buffer
+        // these should be expensive operations that may benefit from running on this backend instead of the CPU backend
+        bool (*offload_op)(lm_ggml_backend_dev_t dev, const struct lm_ggml_tensor * op);
+        // (optional) event synchronization
+        lm_ggml_backend_event_t (*event_new)         (lm_ggml_backend_dev_t dev);
+        void                 (*event_free)        (lm_ggml_backend_dev_t dev, lm_ggml_backend_event_t event);
+        void                 (*event_synchronize) (lm_ggml_backend_dev_t dev, lm_ggml_backend_event_t event);
+    };
+    struct lm_ggml_backend_device {
+        struct lm_ggml_backend_device_i iface;
+        lm_ggml_backend_reg_t reg;
+        void * context;
+    };
+    //
+    // Backend (reg)
+    //
+    struct lm_ggml_backend_reg_i {
+        const char * (*get_name)(lm_ggml_backend_reg_t reg);
+        // enumerate available devices
+        size_t             (*get_device_count)(lm_ggml_backend_reg_t reg);
+        lm_ggml_backend_dev_t (*get_device)(lm_ggml_backend_reg_t reg, size_t index);
+        // (optional) get a pointer to a function in the backend
+        // backends can add custom functions that are not part of the standard ggml-backend interface
+        void * (*get_proc_address)(lm_ggml_backend_reg_t reg, const char * name);
+    };
+    struct lm_ggml_backend_reg {
+        int api_version; // initialize to LM_GGML_BACKEND_API_VERSION
+        struct lm_ggml_backend_reg_i iface;
+        void * context;
+    };
+    // Internal backend registry API
+    LM_GGML_API void lm_ggml_backend_register(lm_ggml_backend_reg_t reg);
+    LM_GGML_API void lm_ggml_backend_device_register(lm_ggml_backend_dev_t device);
+    // Add backend dynamic loading support to the backend
+    typedef lm_ggml_backend_reg_t (*lm_ggml_backend_init_t)(void);
+    #ifdef LM_GGML_BACKEND_DL
+        #ifdef __cplusplus
+        #    define LM_GGML_BACKEND_DL_IMPL(reg_fn)                                 \
+                extern "C" {                                                     \
+                    LM_GGML_BACKEND_API lm_ggml_backend_reg_t lm_ggml_backend_init(void); \
+                }                                                                \
+                lm_ggml_backend_reg_t lm_ggml_backend_init(void) {                     \
+                    return reg_fn();                                             \
+                }
+        #else
+        #    define LM_GGML_BACKEND_DL_IMPL(reg_fn)                             \
+                LM_GGML_BACKEND_API lm_ggml_backend_reg_t lm_ggml_backend_init(void); \
+                lm_ggml_backend_reg_t lm_ggml_backend_init(void) {                 \
+                    return reg_fn();                                         \
+                }
+        #endif
+    #else
+    #    define LM_GGML_BACKEND_DL_IMPL(reg_fn)
+    #endif
+#ifdef  __cplusplus
+}
+#endif