@novastera-oss/llamarn 0.2.6 → 0.2.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (192) hide show
  1. package/android/src/main/cpp/include/llama.h +134 -36
  2. package/android/src/main/jniLibs/arm64-v8a/libggml-base.so +0 -0
  3. package/android/src/main/jniLibs/arm64-v8a/libggml-cpu.so +0 -0
  4. package/android/src/main/jniLibs/arm64-v8a/libggml.so +0 -0
  5. package/android/src/main/jniLibs/arm64-v8a/libllama.so +0 -0
  6. package/android/src/main/jniLibs/x86_64/libggml-base.so +0 -0
  7. package/android/src/main/jniLibs/x86_64/libggml-cpu.so +0 -0
  8. package/android/src/main/jniLibs/x86_64/libggml.so +0 -0
  9. package/android/src/main/jniLibs/x86_64/libllama.so +0 -0
  10. package/cpp/LlamaCppModel.cpp +2 -2
  11. package/cpp/LlamaCppModel.h +3 -3
  12. package/cpp/PureCppImpl.cpp +1 -1
  13. package/cpp/PureCppImpl.h +2 -2
  14. package/cpp/build-info.cpp +2 -2
  15. package/cpp/llama.cpp/CMakeLists.txt +15 -4
  16. package/cpp/llama.cpp/Makefile +2 -2
  17. package/cpp/llama.cpp/README.md +32 -13
  18. package/cpp/llama.cpp/common/CMakeLists.txt +10 -20
  19. package/cpp/llama.cpp/common/arg.cpp +30 -6
  20. package/cpp/llama.cpp/common/build-info.cpp.in +2 -2
  21. package/cpp/llama.cpp/common/chat-parser.cpp +5 -0
  22. package/cpp/llama.cpp/common/chat-parser.h +2 -0
  23. package/cpp/llama.cpp/common/chat.cpp +12 -9
  24. package/cpp/llama.cpp/common/chat.h +1 -1
  25. package/cpp/llama.cpp/common/common.cpp +50 -40
  26. package/cpp/llama.cpp/common/common.h +5 -2
  27. package/cpp/llama.cpp/common/speculative.cpp +6 -4
  28. package/cpp/llama.cpp/convert_hf_to_gguf.py +97 -56
  29. package/cpp/llama.cpp/ggml/CMakeLists.txt +47 -2
  30. package/cpp/llama.cpp/ggml/cmake/common.cmake +1 -2
  31. package/cpp/llama.cpp/ggml/src/CMakeLists.txt +47 -13
  32. package/cpp/llama.cpp/ggml/src/ggml-backend-reg.cpp +5 -0
  33. package/cpp/llama.cpp/ggml/src/ggml-cann/common.h +6 -1
  34. package/cpp/llama.cpp/ggml/src/ggml-cann/ggml-cann.cpp +33 -9
  35. package/cpp/llama.cpp/ggml/src/ggml-common.h +4 -0
  36. package/cpp/llama.cpp/ggml/src/ggml-cpu/CMakeLists.txt +93 -24
  37. package/cpp/llama.cpp/ggml/src/ggml-cpu/amx/amx.cpp +1 -1
  38. package/cpp/llama.cpp/ggml/src/ggml-cpu/amx/mmq.cpp +1 -1
  39. package/cpp/llama.cpp/ggml/src/ggml-cpu/arch/arm/cpu-feats.cpp +94 -0
  40. package/cpp/llama.cpp/ggml/src/ggml-cpu/arch/arm/quants.c +4113 -0
  41. package/cpp/llama.cpp/ggml/src/ggml-cpu/arch/arm/repack.cpp +2174 -0
  42. package/cpp/llama.cpp/ggml/src/ggml-cpu/arch/loongarch/quants.c +2638 -0
  43. package/cpp/llama.cpp/ggml/src/ggml-cpu/arch/powerpc/quants.c +2731 -0
  44. package/cpp/llama.cpp/ggml/src/ggml-cpu/arch/riscv/quants.c +2068 -0
  45. package/cpp/llama.cpp/ggml/src/ggml-cpu/arch/riscv/repack.cpp +396 -0
  46. package/cpp/llama.cpp/ggml/src/ggml-cpu/arch/s390/quants.c +1299 -0
  47. package/cpp/llama.cpp/ggml/src/ggml-cpu/arch/wasm/quants.c +1480 -0
  48. package/cpp/llama.cpp/ggml/src/ggml-cpu/arch/x86/quants.c +4310 -0
  49. package/cpp/llama.cpp/ggml/src/ggml-cpu/{ggml-cpu-aarch64.cpp → arch/x86/repack.cpp} +59 -3206
  50. package/cpp/llama.cpp/ggml/src/ggml-cpu/arch-fallback.h +184 -0
  51. package/cpp/llama.cpp/ggml/src/ggml-cpu/common.h +1 -1
  52. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu-impl.h +7 -4
  53. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu.c +10 -2
  54. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu.cpp +8 -8
  55. package/cpp/llama.cpp/ggml/src/ggml-cpu/{ggml-cpu-hbm.cpp → hbm.cpp} +1 -1
  56. package/cpp/llama.cpp/ggml/src/ggml-cpu/kleidiai/kleidiai.cpp +1 -1
  57. package/cpp/llama.cpp/ggml/src/ggml-cpu/llamafile/sgemm.cpp +56 -7
  58. package/cpp/llama.cpp/ggml/src/ggml-cpu/llamafile/sgemm.h +5 -0
  59. package/cpp/llama.cpp/ggml/src/ggml-cpu/ops.cpp +2 -2
  60. package/cpp/llama.cpp/ggml/src/ggml-cpu/quants.c +1157 -0
  61. package/cpp/llama.cpp/ggml/src/ggml-cpu/{ggml-cpu-quants.h → quants.h} +26 -0
  62. package/cpp/llama.cpp/ggml/src/ggml-cpu/repack.cpp +1555 -0
  63. package/cpp/llama.cpp/ggml/src/ggml-cpu/repack.h +98 -0
  64. package/cpp/llama.cpp/ggml/src/ggml-cpu/simd-mappings.h +2 -4
  65. package/cpp/llama.cpp/ggml/src/ggml-cpu/{ggml-cpu-traits.cpp → traits.cpp} +1 -1
  66. package/cpp/llama.cpp/ggml/src/ggml-cuda/common.cuh +5 -8
  67. package/cpp/llama.cpp/ggml/src/ggml-cuda/fattn-mma-f16.cuh +4 -1
  68. package/cpp/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu +6 -8
  69. package/cpp/llama.cpp/ggml/src/ggml-cuda/ssm-scan.cu +6 -4
  70. package/cpp/llama.cpp/ggml/src/ggml-hip/CMakeLists.txt +4 -0
  71. package/cpp/llama.cpp/ggml/src/ggml-metal/CMakeLists.txt +11 -10
  72. package/cpp/llama.cpp/ggml/src/ggml-metal/ggml-metal.m +33 -8
  73. package/cpp/llama.cpp/ggml/src/ggml-metal/ggml-metal.metal +135 -100
  74. package/cpp/llama.cpp/ggml/src/ggml-opencl/CMakeLists.txt +7 -0
  75. package/cpp/llama.cpp/ggml/src/ggml-opencl/ggml-opencl.cpp +908 -3
  76. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/concat.cl +109 -0
  77. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/mul_mv_id_q4_0_f32_8x_flat.cl +283 -0
  78. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/pad.cl +30 -0
  79. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/repeat.cl +39 -0
  80. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/tanh.cl +63 -0
  81. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/tsembd.cl +48 -0
  82. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/upscale.cl +121 -0
  83. package/cpp/llama.cpp/ggml/src/ggml-quants.c +0 -2
  84. package/cpp/llama.cpp/ggml/src/ggml-rpc/ggml-rpc.cpp +18 -15
  85. package/cpp/llama.cpp/ggml/src/ggml-sycl/CMakeLists.txt +1 -1
  86. package/cpp/llama.cpp/ggml/src/ggml-sycl/common.hpp +19 -24
  87. package/cpp/llama.cpp/ggml/src/ggml-sycl/convert.cpp +21 -2
  88. package/cpp/llama.cpp/ggml/src/ggml-sycl/cpy.cpp +121 -4
  89. package/cpp/llama.cpp/ggml/src/ggml-sycl/dequantize.hpp +32 -0
  90. package/cpp/llama.cpp/ggml/src/ggml-sycl/gemm.hpp +3 -0
  91. package/cpp/llama.cpp/ggml/src/ggml-sycl/getrows.cpp +2 -96
  92. package/cpp/llama.cpp/ggml/src/ggml-sycl/ggml-sycl.cpp +164 -38
  93. package/cpp/llama.cpp/ggml/src/ggml-sycl/mmvq.cpp +32 -8
  94. package/cpp/llama.cpp/ggml/src/ggml-sycl/quants.hpp +38 -10
  95. package/cpp/llama.cpp/ggml/src/ggml-sycl/vecdotq.hpp +108 -16
  96. package/cpp/llama.cpp/ggml/src/ggml-vulkan/CMakeLists.txt +26 -29
  97. package/cpp/llama.cpp/ggml/src/ggml-vulkan/ggml-vulkan.cpp +431 -247
  98. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/CMakeLists.txt +0 -12
  99. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/conv_transpose_1d.comp +98 -0
  100. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/vulkan-shaders-gen.cpp +2 -0
  101. package/cpp/llama.cpp/ggml/src/ggml.c +0 -6
  102. package/cpp/llama.cpp/gguf-py/gguf/constants.py +57 -0
  103. package/cpp/llama.cpp/gguf-py/gguf/gguf_writer.py +4 -1
  104. package/cpp/llama.cpp/gguf-py/gguf/tensor_mapping.py +14 -3
  105. package/cpp/llama.cpp/include/llama.h +134 -36
  106. package/cpp/llama.cpp/requirements/requirements-compare-llama-bench.txt +1 -0
  107. package/cpp/llama.cpp/src/CMakeLists.txt +2 -2
  108. package/cpp/llama.cpp/src/llama-arch.cpp +95 -3
  109. package/cpp/llama.cpp/src/llama-arch.h +7 -1
  110. package/cpp/llama.cpp/src/llama-batch.cpp +270 -19
  111. package/cpp/llama.cpp/src/llama-batch.h +36 -11
  112. package/cpp/llama.cpp/src/llama-chat.cpp +19 -2
  113. package/cpp/llama.cpp/src/llama-chat.h +1 -0
  114. package/cpp/llama.cpp/src/llama-context.cpp +313 -213
  115. package/cpp/llama.cpp/src/llama-context.h +16 -12
  116. package/cpp/llama.cpp/src/llama-cparams.cpp +1 -1
  117. package/cpp/llama.cpp/src/llama-cparams.h +1 -1
  118. package/cpp/llama.cpp/src/llama-graph.cpp +249 -129
  119. package/cpp/llama.cpp/src/llama-graph.h +90 -34
  120. package/cpp/llama.cpp/src/llama-hparams.cpp +6 -2
  121. package/cpp/llama.cpp/src/llama-hparams.h +8 -2
  122. package/cpp/llama.cpp/src/llama-kv-cache-unified-iswa.cpp +82 -50
  123. package/cpp/llama.cpp/src/llama-kv-cache-unified-iswa.h +23 -26
  124. package/cpp/llama.cpp/src/llama-kv-cache-unified.cpp +292 -174
  125. package/cpp/llama.cpp/src/llama-kv-cache-unified.h +68 -38
  126. package/cpp/llama.cpp/src/llama-kv-cells.h +18 -13
  127. package/cpp/llama.cpp/src/llama-memory-hybrid.cpp +247 -0
  128. package/cpp/llama.cpp/src/llama-memory-hybrid.h +143 -0
  129. package/cpp/llama.cpp/src/{llama-kv-cache-recurrent.cpp → llama-memory-recurrent.cpp} +266 -282
  130. package/cpp/llama.cpp/src/{llama-kv-cache-recurrent.h → llama-memory-recurrent.h} +54 -57
  131. package/cpp/llama.cpp/src/llama-memory.cpp +41 -0
  132. package/cpp/llama.cpp/src/llama-memory.h +64 -23
  133. package/cpp/llama.cpp/src/llama-mmap.cpp +1 -1
  134. package/cpp/llama.cpp/src/llama-model-loader.cpp +42 -17
  135. package/cpp/llama.cpp/src/llama-model.cpp +726 -141
  136. package/cpp/llama.cpp/src/llama-model.h +4 -0
  137. package/cpp/llama.cpp/src/llama-quant.cpp +2 -1
  138. package/cpp/llama.cpp/src/llama-vocab.cpp +32 -23
  139. package/cpp/llama.cpp/src/llama.cpp +11 -7
  140. package/cpp/llama.cpp/src/unicode.cpp +5 -0
  141. package/cpp/rn-completion.cpp +2 -2
  142. package/cpp/{rn-llama.hpp → rn-llama.h} +1 -1
  143. package/ios/include/chat.h +1 -1
  144. package/ios/include/common.h +5 -2
  145. package/ios/include/llama.h +134 -36
  146. package/ios/libs/llama.xcframework/Info.plist +18 -18
  147. package/ios/libs/llama.xcframework/ios-arm64/dSYMs/llama.dSYM/Contents/Resources/DWARF/llama +0 -0
  148. package/ios/libs/llama.xcframework/ios-arm64/dSYMs/llama.dSYM/Contents/Resources/Relocations/aarch64/llama.yml +4863 -4689
  149. package/ios/libs/llama.xcframework/ios-arm64/llama.framework/Headers/llama.h +134 -36
  150. package/ios/libs/llama.xcframework/ios-arm64/llama.framework/llama +0 -0
  151. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/DWARF/llama +0 -0
  152. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/Relocations/aarch64/llama.yml +4834 -4710
  153. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/Relocations/x86_64/llama.yml +3742 -3622
  154. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/llama.framework/Headers/llama.h +134 -36
  155. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/llama.framework/llama +0 -0
  156. package/ios/libs/llama.xcframework/macos-arm64_x86_64/dSYMs/llama.dSYM/Contents/Resources/DWARF/llama +0 -0
  157. package/ios/libs/llama.xcframework/macos-arm64_x86_64/dSYMs/llama.dSYM/Contents/Resources/Relocations/aarch64/llama.yml +4834 -4710
  158. package/ios/libs/llama.xcframework/macos-arm64_x86_64/dSYMs/llama.dSYM/Contents/Resources/Relocations/x86_64/llama.yml +3744 -3624
  159. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Headers/llama.h +134 -36
  160. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/A/Headers/llama.h +134 -36
  161. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/A/llama +0 -0
  162. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/Current/Headers/llama.h +134 -36
  163. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/Current/llama +0 -0
  164. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/llama +0 -0
  165. package/ios/libs/llama.xcframework/tvos-arm64/dSYMs/llama.dSYM/Contents/Resources/DWARF/llama +0 -0
  166. package/ios/libs/llama.xcframework/tvos-arm64/dSYMs/llama.dSYM/Contents/Resources/Relocations/aarch64/llama.yml +4863 -4689
  167. package/ios/libs/llama.xcframework/tvos-arm64/llama.framework/Headers/llama.h +134 -36
  168. package/ios/libs/llama.xcframework/tvos-arm64/llama.framework/llama +0 -0
  169. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/DWARF/llama +0 -0
  170. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/Relocations/aarch64/llama.yml +4834 -4710
  171. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/Relocations/x86_64/llama.yml +3742 -3622
  172. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/llama.framework/Headers/llama.h +134 -36
  173. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/llama.framework/llama +0 -0
  174. package/ios/libs/llama.xcframework/xros-arm64/dSYMs/llama.dSYM/Contents/Resources/DWARF/llama +0 -0
  175. package/ios/libs/llama.xcframework/xros-arm64/dSYMs/llama.dSYM/Contents/Resources/Relocations/aarch64/llama.yml +4900 -4725
  176. package/ios/libs/llama.xcframework/xros-arm64/llama.framework/Headers/llama.h +134 -36
  177. package/ios/libs/llama.xcframework/xros-arm64/llama.framework/llama +0 -0
  178. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/DWARF/llama +0 -0
  179. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/Relocations/aarch64/llama.yml +4871 -4746
  180. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/Relocations/x86_64/llama.yml +3773 -3652
  181. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/llama.framework/Headers/llama.h +134 -36
  182. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/llama.framework/llama +0 -0
  183. package/package.json +1 -2
  184. package/cpp/llama.cpp/common/cmake/build-info-gen-cpp.cmake +0 -24
  185. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu-aarch64.h +0 -8
  186. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu-quants.c +0 -13891
  187. package/cpp/llama.cpp/src/llama-kv-cache.cpp +0 -1
  188. package/cpp/llama.cpp/src/llama-kv-cache.h +0 -44
  189. /package/cpp/llama.cpp/ggml/src/ggml-cpu/{cpu-feats-x86.cpp → arch/x86/cpu-feats.cpp} +0 -0
  190. /package/cpp/llama.cpp/ggml/src/ggml-cpu/{ggml-cpu-hbm.h → hbm.h} +0 -0
  191. /package/cpp/llama.cpp/ggml/src/ggml-cpu/{ggml-cpu-traits.h → traits.h} +0 -0
  192. /package/cpp/{rn-utils.hpp → rn-utils.h} +0 -0
@@ -1,3 +1,17 @@
1
+ function(ggml_add_cpu_backend_features cpu_name arch)
2
+ # The feature detection code is compiled as a separate target so that
3
+ # it can be built without the architecture flags
4
+ # Since multiple variants of the CPU backend may be included in the same
5
+ # build, using set_source_files_properties() to set the arch flags is not possible
6
+ set(GGML_CPU_FEATS_NAME ${cpu_name}-feats)
7
+ add_library(${GGML_CPU_FEATS_NAME} OBJECT ggml-cpu/arch/${arch}/cpu-feats.cpp)
8
+ target_include_directories(${GGML_CPU_FEATS_NAME} PRIVATE . .. ../include)
9
+ target_compile_definitions(${GGML_CPU_FEATS_NAME} PRIVATE ${ARGN})
10
+ target_compile_definitions(${GGML_CPU_FEATS_NAME} PRIVATE GGML_BACKEND_DL GGML_BACKEND_BUILD GGML_BACKEND_SHARED)
11
+ set_target_properties(${GGML_CPU_FEATS_NAME} PROPERTIES POSITION_INDEPENDENT_CODE ON)
12
+ target_link_libraries(${cpu_name} PRIVATE ${GGML_CPU_FEATS_NAME})
13
+ endfunction()
14
+
1
15
  function(ggml_add_cpu_backend_variant_impl tag_name)
2
16
  if (tag_name)
3
17
  set(GGML_CPU_NAME ggml-cpu-${tag_name})
@@ -10,14 +24,14 @@ function(ggml_add_cpu_backend_variant_impl tag_name)
10
24
  list (APPEND GGML_CPU_SOURCES
11
25
  ggml-cpu/ggml-cpu.c
12
26
  ggml-cpu/ggml-cpu.cpp
13
- ggml-cpu/ggml-cpu-aarch64.cpp
14
- ggml-cpu/ggml-cpu-aarch64.h
15
- ggml-cpu/ggml-cpu-hbm.cpp
16
- ggml-cpu/ggml-cpu-hbm.h
17
- ggml-cpu/ggml-cpu-quants.c
18
- ggml-cpu/ggml-cpu-quants.h
19
- ggml-cpu/ggml-cpu-traits.cpp
20
- ggml-cpu/ggml-cpu-traits.h
27
+ ggml-cpu/repack.cpp
28
+ ggml-cpu/repack.h
29
+ ggml-cpu/hbm.cpp
30
+ ggml-cpu/hbm.h
31
+ ggml-cpu/quants.c
32
+ ggml-cpu/quants.h
33
+ ggml-cpu/traits.cpp
34
+ ggml-cpu/traits.h
21
35
  ggml-cpu/amx/amx.cpp
22
36
  ggml-cpu/amx/amx.h
23
37
  ggml-cpu/amx/mmq.cpp
@@ -84,6 +98,11 @@ function(ggml_add_cpu_backend_variant_impl tag_name)
84
98
 
85
99
  if (GGML_SYSTEM_ARCH STREQUAL "ARM")
86
100
  message(STATUS "ARM detected")
101
+ list(APPEND GGML_CPU_SOURCES
102
+ ggml-cpu/arch/arm/quants.c
103
+ ggml-cpu/arch/arm/repack.cpp
104
+ )
105
+
87
106
  if (MSVC AND NOT CMAKE_C_COMPILER_ID STREQUAL "Clang")
88
107
  message(FATAL_ERROR "MSVC is not supported for ARM, use clang")
89
108
  else()
@@ -138,6 +157,49 @@ function(ggml_add_cpu_backend_variant_impl tag_name)
138
157
  else()
139
158
  if (GGML_CPU_ARM_ARCH)
140
159
  list(APPEND ARCH_FLAGS -march=${GGML_CPU_ARM_ARCH})
160
+ elseif(GGML_CPU_ALL_VARIANTS)
161
+ # Begin with the lowest baseline
162
+ set(ARM_MCPU "armv8-a")
163
+ set(ARCH_TAGS "")
164
+ set(ARCH_DEFINITIONS "")
165
+
166
+ # When a feature is selected, bump the MCPU to the first
167
+ # version that supported it
168
+ if (GGML_INTERNAL_DOTPROD)
169
+ set(ARM_MCPU "armv8.2-a")
170
+ set(ARCH_TAGS "${ARCH_TAGS}+dotprod")
171
+ list(APPEND ARCH_DEFINITIONS GGML_USE_DOTPROD)
172
+ endif()
173
+ if (GGML_INTERNAL_FP16_VECTOR_ARITHMETIC)
174
+ set(ARM_MCPU "armv8.2-a")
175
+ set(ARCH_TAGS "${ARCH_TAGS}+fp16")
176
+ list(APPEND ARCH_DEFINITIONS GGML_USE_FP16_VECTOR_ARITHMETIC)
177
+ endif()
178
+ if (GGML_INTERNAL_SVE)
179
+ set(ARM_MCPU "armv8.2-a")
180
+ set(ARCH_TAGS "${ARCH_TAGS}+sve")
181
+ list(APPEND ARCH_DEFINITIONS GGML_USE_SVE)
182
+ endif()
183
+ if (GGML_INTERNAL_MATMUL_INT8)
184
+ set(ARM_MCPU "armv8.6-a")
185
+ set(ARCH_TAGS "${ARCH_TAGS}+i8mm")
186
+ list(APPEND ARCH_DEFINITIONS GGML_USE_MATMUL_INT8)
187
+ endif()
188
+ if (GGML_INTERNAL_SVE2)
189
+ set(ARM_MCPU "armv8.6-a")
190
+ set(ARCH_TAGS "${ARCH_TAGS}+sve2")
191
+ list(APPEND ARCH_DEFINITIONS GGML_USE_SVE2)
192
+ endif()
193
+ if (GGML_INTERNAL_NOSVE)
194
+ set(ARCH_TAGS "${ARCH_TAGS}+nosve")
195
+ endif()
196
+ if (GGML_INTERNAL_SME)
197
+ set(ARM_MCPU "armv9.2-a")
198
+ set(ARCH_TAGS "${ARCH_TAGS}+sme")
199
+ list(APPEND ARCH_DEFINITIONS GGML_USE_SME)
200
+ endif()
201
+ list(APPEND ARCH_FLAGS "-march=${ARM_MCPU}${ARCH_TAGS}")
202
+ ggml_add_cpu_backend_features(${GGML_CPU_NAME} arm ${ARCH_DEFINITIONS})
141
203
  endif()
142
204
  endif()
143
205
 
@@ -167,6 +229,11 @@ function(ggml_add_cpu_backend_variant_impl tag_name)
167
229
  endif()
168
230
  elseif (GGML_SYSTEM_ARCH STREQUAL "x86")
169
231
  message(STATUS "x86 detected")
232
+ list(APPEND GGML_CPU_SOURCES
233
+ ggml-cpu/arch/x86/quants.c
234
+ ggml-cpu/arch/x86/repack.cpp
235
+ )
236
+
170
237
  if (MSVC)
171
238
  # instruction set detection for MSVC only
172
239
  if (GGML_NATIVE)
@@ -296,21 +363,11 @@ function(ggml_add_cpu_backend_variant_impl tag_name)
296
363
  # the feature check relies on ARCH_DEFINITIONS, but it is not set with GGML_NATIVE
297
364
  message(FATAL_ERROR "GGML_NATIVE is not compatible with GGML_BACKEND_DL, consider using GGML_CPU_ALL_VARIANTS")
298
365
  endif()
299
-
300
- # The feature detection code is compiled as a separate target so that
301
- # it can be built without the architecture flags
302
- # Since multiple variants of the CPU backend may be included in the same
303
- # build, using set_source_files_properties() to set the arch flags is not possible
304
- set(GGML_CPU_FEATS_NAME ${GGML_CPU_NAME}-feats)
305
- add_library(${GGML_CPU_FEATS_NAME} OBJECT ggml-cpu/cpu-feats-x86.cpp)
306
- target_include_directories(${GGML_CPU_FEATS_NAME} PRIVATE . .. ../include)
307
- target_compile_definitions(${GGML_CPU_FEATS_NAME} PRIVATE ${ARCH_DEFINITIONS})
308
- target_compile_definitions(${GGML_CPU_FEATS_NAME} PRIVATE GGML_BACKEND_DL GGML_BACKEND_BUILD GGML_BACKEND_SHARED)
309
- set_target_properties(${GGML_CPU_FEATS_NAME} PROPERTIES POSITION_INDEPENDENT_CODE ON)
310
- target_link_libraries(${GGML_CPU_NAME} PRIVATE ${GGML_CPU_FEATS_NAME})
366
+ ggml_add_cpu_backend_features(${GGML_CPU_NAME} x86 ${ARCH_DEFINITIONS})
311
367
  endif()
312
368
  elseif (GGML_SYSTEM_ARCH STREQUAL "PowerPC")
313
369
  message(STATUS "PowerPC detected")
370
+ list(APPEND GGML_CPU_SOURCES ggml-cpu/arch/powerpc/quants.c)
314
371
  if (GGML_NATIVE)
315
372
  if (${CMAKE_SYSTEM_PROCESSOR} MATCHES "ppc64")
316
373
  file(READ "/proc/cpuinfo" POWER10_M)
@@ -318,7 +375,8 @@ function(ggml_add_cpu_backend_variant_impl tag_name)
318
375
  execute_process(COMMAND bash -c "prtconf |grep 'Implementation' | head -n 1" OUTPUT_VARIABLE POWER10_M)
319
376
  endif()
320
377
 
321
- string(REGEX MATCHALL "POWER *([0-9]+)" MATCHED_STRING "${POWER10_M}")
378
+ string(TOUPPER "${POWER10_M}" POWER10_M_UPPER)
379
+ string(REGEX MATCHALL "POWER *([0-9]+)" MATCHED_STRING "${POWER10_M_UPPER}")
322
380
  string(REGEX REPLACE "POWER *([0-9]+)" "\\1" EXTRACTED_NUMBER "${MATCHED_STRING}")
323
381
 
324
382
  if (EXTRACTED_NUMBER GREATER_EQUAL 10)
@@ -337,6 +395,8 @@ function(ggml_add_cpu_backend_variant_impl tag_name)
337
395
  endif()
338
396
  elseif (GGML_SYSTEM_ARCH STREQUAL "loongarch64")
339
397
  message(STATUS "loongarch64 detected")
398
+ list(APPEND GGML_CPU_SOURCES ggml-cpu/arch/loongarch/quants.c)
399
+
340
400
  list(APPEND ARCH_FLAGS -march=loongarch64)
341
401
  if (GGML_LASX)
342
402
  list(APPEND ARCH_FLAGS -mlasx)
@@ -346,6 +406,10 @@ function(ggml_add_cpu_backend_variant_impl tag_name)
346
406
  endif()
347
407
  elseif (GGML_SYSTEM_ARCH STREQUAL "riscv64")
348
408
  message(STATUS "riscv64 detected")
409
+ list(APPEND GGML_CPU_SOURCES
410
+ ggml-cpu/arch/riscv/quants.c
411
+ ggml-cpu/arch/riscv/repack.cpp
412
+ )
349
413
  if (GGML_RVV)
350
414
  if (GGML_XTHEADVECTOR)
351
415
  list(APPEND ARCH_FLAGS -march=rv64gc_xtheadvector -mabi=lp64d)
@@ -357,6 +421,7 @@ function(ggml_add_cpu_backend_variant_impl tag_name)
357
421
  endif()
358
422
  elseif (GGML_SYSTEM_ARCH STREQUAL "s390x")
359
423
  message(STATUS "s390x detected")
424
+ list(APPEND GGML_CPU_SOURCES ggml-cpu/arch/s390/quants.c)
360
425
  file(READ "/proc/cpuinfo" CPUINFO_CONTENTS)
361
426
  string(REGEX REPLACE "machine[ \t\r\n]*=[ \t\r\n]*([0-9]+)" "\\1" S390X_M ${CPUINFO_CONTENTS})
362
427
 
@@ -380,12 +445,16 @@ function(ggml_add_cpu_backend_variant_impl tag_name)
380
445
  if (GGML_VXE)
381
446
  list(APPEND ARCH_FLAGS -mvx -mzvector)
382
447
  endif()
448
+ elseif (CMAKE_SYSTEM_PROCESSOR MATCHES "wasm")
449
+ message(STATUS "Wasm detected")
450
+ list (APPEND GGML_CPU_SOURCES ggml-cpu/arch/wasm/quants.c)
383
451
  else()
384
- message(STATUS "Unknown architecture")
452
+ message(WARNING "Unknown CPU architecture. Falling back to generic implementations.")
453
+ list(APPEND ARCH_FLAGS -DGGML_CPU_GENERIC)
385
454
  endif()
386
455
 
387
- if (GGML_CPU_AARCH64)
388
- target_compile_definitions(${GGML_CPU_NAME} PRIVATE GGML_USE_CPU_AARCH64)
456
+ if (GGML_CPU_REPACK)
457
+ target_compile_definitions(${GGML_CPU_NAME} PRIVATE GGML_USE_CPU_REPACK)
389
458
  endif()
390
459
 
391
460
  if (GGML_CPU_KLEIDIAI)
@@ -5,7 +5,7 @@
5
5
  #include "ggml-backend.h"
6
6
  #include "ggml-impl.h"
7
7
  #include "ggml-cpu.h"
8
- #include "ggml-cpu-traits.h"
8
+ #include "traits.h"
9
9
 
10
10
  #if defined(__gnu_linux__)
11
11
  #include <sys/syscall.h>
@@ -8,7 +8,7 @@
8
8
  #include "mmq.h"
9
9
  #include "ggml-impl.h"
10
10
  #include "ggml-cpu-impl.h"
11
- #include "ggml-cpu-quants.h"
11
+ #include "quants.h"
12
12
  #include "ggml-quants.h"
13
13
  #include <algorithm>
14
14
  #include <type_traits>
@@ -0,0 +1,94 @@
1
+ #include "ggml-backend-impl.h"
2
+
3
+ #if defined(__aarch64__)
4
+
5
+ #if defined(__linux__)
6
+ #include <sys/auxv.h>
7
+ #elif defined(__APPLE__)
8
+ #include <sys/sysctl.h>
9
+ #endif
10
+
11
+ #if !defined(HWCAP2_I8MM)
12
+ #define HWCAP2_I8MM (1 << 13)
13
+ #endif
14
+
15
+ #if !defined(HWCAP2_SME)
16
+ #define HWCAP2_SME (1 << 23)
17
+ #endif
18
+
19
+ struct aarch64_features {
20
+ // has_neon not needed, aarch64 has NEON guaranteed
21
+ bool has_dotprod = false;
22
+ bool has_fp16_va = false;
23
+ bool has_sve = false;
24
+ bool has_sve2 = false;
25
+ bool has_i8mm = false;
26
+ bool has_sme = false;
27
+
28
+ aarch64_features() {
29
+ #if defined(__linux__)
30
+ uint32_t hwcap = getauxval(AT_HWCAP);
31
+ uint32_t hwcap2 = getauxval(AT_HWCAP2);
32
+
33
+ has_dotprod = !!(hwcap & HWCAP_ASIMDDP);
34
+ has_fp16_va = !!(hwcap & HWCAP_FPHP);
35
+ has_sve = !!(hwcap & HWCAP_SVE);
36
+ has_sve2 = !!(hwcap2 & HWCAP2_SVE2);
37
+ has_i8mm = !!(hwcap2 & HWCAP2_I8MM);
38
+ has_sme = !!(hwcap2 & HWCAP2_SME);
39
+ #elif defined(__APPLE__)
40
+ int oldp = 0;
41
+ size_t size = sizeof(oldp);
42
+
43
+ if (sysctlbyname("hw.optional.arm.FEAT_DotProd", &oldp, &size, NULL, 0) == 0) {
44
+ has_dotprod = static_cast<bool>(oldp);
45
+ }
46
+
47
+ if (sysctlbyname("hw.optional.arm.FEAT_I8MM", &oldp, &size, NULL, 0) == 0) {
48
+ has_i8mm = static_cast<bool>(oldp);
49
+ }
50
+
51
+ if (sysctlbyname("hw.optional.arm.FEAT_SME", &oldp, &size, NULL, 0) == 0) {
52
+ has_sme = static_cast<bool>(oldp);
53
+ }
54
+
55
+ // Apple apparently does not implement SVE yet
56
+ #endif
57
+ }
58
+ };
59
+
60
+ static int ggml_backend_cpu_aarch64_score() {
61
+ int score = 1;
62
+ aarch64_features af;
63
+
64
+ #ifdef GGML_USE_DOTPROD
65
+ if (!af.has_dotprod) { return 0; }
66
+ score += 1<<1;
67
+ #endif
68
+ #ifdef GGML_USE_FP16_VECTOR_ARITHMETIC
69
+ if (!af.has_fp16_va) { return 0; }
70
+ score += 1<<2;
71
+ #endif
72
+ #ifdef GGML_USE_SVE
73
+ if (!af.has_sve) { return 0; }
74
+ score += 1<<3;
75
+ #endif
76
+ #ifdef GGML_USE_MATMUL_INT8
77
+ if (!af.has_i8mm) { return 0; }
78
+ score += 1<<4;
79
+ #endif
80
+ #ifdef GGML_USE_SVE2
81
+ if (!af.has_sve2) { return 0; }
82
+ score += 1<<5;
83
+ #endif
84
+ #ifdef GGML_USE_SME
85
+ if (!af.has_sme) { return 0; }
86
+ score += 1<<6;
87
+ #endif
88
+
89
+ return score;
90
+ }
91
+
92
+ GGML_BACKEND_DL_SCORE_IMPL(ggml_backend_cpu_aarch64_score)
93
+
94
+ # endif // defined(__aarch64__)