@novastera-oss/llamarn 0.0.1-alpha.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (989) hide show
  1. package/INTERFACE.md +389 -0
  2. package/LICENSE +201 -0
  3. package/README.md +235 -0
  4. package/RNLlamaCpp.podspec +69 -0
  5. package/android/CMakeLists.txt +107 -0
  6. package/android/build.gradle +111 -0
  7. package/android/generated/java/com/novastera/llamarn/NativeRNLlamaCppSpec.java +47 -0
  8. package/android/generated/jni/CMakeLists.txt +36 -0
  9. package/android/generated/jni/RNLlamaCppSpec-generated.cpp +44 -0
  10. package/android/generated/jni/RNLlamaCppSpec.h +31 -0
  11. package/android/generated/jni/react/renderer/components/RNLlamaCppSpec/RNLlamaCppSpecJSI-generated.cpp +42 -0
  12. package/android/generated/jni/react/renderer/components/RNLlamaCppSpec/RNLlamaCppSpecJSI.h +336 -0
  13. package/android/gradle.properties +5 -0
  14. package/android/src/main/AndroidManifest.xml +3 -0
  15. package/android/src/main/AndroidManifestNew.xml +2 -0
  16. package/android/src/main/cpp/include/llama-cpp.h +30 -0
  17. package/android/src/main/cpp/include/llama.h +1440 -0
  18. package/android/src/main/java/com/novastera/llamarn/RNLlamaCppPackage.kt +21 -0
  19. package/android/src/main/jniLibs/arm64-v8a/libOpenCL.so +0 -0
  20. package/android/src/main/jniLibs/arm64-v8a/libggml-base.so +0 -0
  21. package/android/src/main/jniLibs/arm64-v8a/libggml-cpu.so +0 -0
  22. package/android/src/main/jniLibs/arm64-v8a/libggml.so +0 -0
  23. package/android/src/main/jniLibs/arm64-v8a/libllama.so +0 -0
  24. package/android/src/main/jniLibs/x86_64/libOpenCL.so +0 -0
  25. package/android/src/main/jniLibs/x86_64/libggml-base.so +0 -0
  26. package/android/src/main/jniLibs/x86_64/libggml-cpu.so +0 -0
  27. package/android/src/main/jniLibs/x86_64/libggml.so +0 -0
  28. package/android/src/main/jniLibs/x86_64/libllama.so +0 -0
  29. package/cpp/LlamaCppModel.cpp +984 -0
  30. package/cpp/LlamaCppModel.h +162 -0
  31. package/cpp/PureCppImpl.cpp +308 -0
  32. package/cpp/PureCppImpl.h +59 -0
  33. package/cpp/SystemUtils.cpp +180 -0
  34. package/cpp/SystemUtils.h +74 -0
  35. package/cpp/build-info.cpp +4 -0
  36. package/cpp/llama.cpp/AUTHORS +1106 -0
  37. package/cpp/llama.cpp/CMakeLists.txt +254 -0
  38. package/cpp/llama.cpp/CMakePresets.json +84 -0
  39. package/cpp/llama.cpp/CODEOWNERS +11 -0
  40. package/cpp/llama.cpp/CONTRIBUTING.md +127 -0
  41. package/cpp/llama.cpp/LICENSE +21 -0
  42. package/cpp/llama.cpp/Makefile +1608 -0
  43. package/cpp/llama.cpp/README.md +575 -0
  44. package/cpp/llama.cpp/SECURITY.md +68 -0
  45. package/cpp/llama.cpp/build-xcframework.sh +540 -0
  46. package/cpp/llama.cpp/cmake/arm64-apple-clang.cmake +16 -0
  47. package/cpp/llama.cpp/cmake/arm64-windows-llvm.cmake +16 -0
  48. package/cpp/llama.cpp/cmake/build-info.cmake +64 -0
  49. package/cpp/llama.cpp/cmake/common.cmake +35 -0
  50. package/cpp/llama.cpp/cmake/git-vars.cmake +22 -0
  51. package/cpp/llama.cpp/cmake/llama-config.cmake.in +30 -0
  52. package/cpp/llama.cpp/cmake/llama.pc.in +10 -0
  53. package/cpp/llama.cpp/cmake/x64-windows-llvm.cmake +5 -0
  54. package/cpp/llama.cpp/common/CMakeLists.txt +170 -0
  55. package/cpp/llama.cpp/common/arg.cpp +3337 -0
  56. package/cpp/llama.cpp/common/arg.h +89 -0
  57. package/cpp/llama.cpp/common/base64.hpp +392 -0
  58. package/cpp/llama.cpp/common/build-info.cpp.in +4 -0
  59. package/cpp/llama.cpp/common/chat.cpp +1781 -0
  60. package/cpp/llama.cpp/common/chat.h +135 -0
  61. package/cpp/llama.cpp/common/cmake/build-info-gen-cpp.cmake +24 -0
  62. package/cpp/llama.cpp/common/common.cpp +1567 -0
  63. package/cpp/llama.cpp/common/common.h +668 -0
  64. package/cpp/llama.cpp/common/console.cpp +504 -0
  65. package/cpp/llama.cpp/common/console.h +19 -0
  66. package/cpp/llama.cpp/common/json-schema-to-grammar.cpp +1027 -0
  67. package/cpp/llama.cpp/common/json-schema-to-grammar.h +21 -0
  68. package/cpp/llama.cpp/common/json.hpp +24766 -0
  69. package/cpp/llama.cpp/common/llguidance.cpp +254 -0
  70. package/cpp/llama.cpp/common/log.cpp +393 -0
  71. package/cpp/llama.cpp/common/log.h +103 -0
  72. package/cpp/llama.cpp/common/minja/chat-template.hpp +537 -0
  73. package/cpp/llama.cpp/common/minja/minja.hpp +2941 -0
  74. package/cpp/llama.cpp/common/ngram-cache.cpp +286 -0
  75. package/cpp/llama.cpp/common/ngram-cache.h +101 -0
  76. package/cpp/llama.cpp/common/sampling.cpp +580 -0
  77. package/cpp/llama.cpp/common/sampling.h +107 -0
  78. package/cpp/llama.cpp/common/speculative.cpp +278 -0
  79. package/cpp/llama.cpp/common/speculative.h +28 -0
  80. package/cpp/llama.cpp/common/stb_image.h +7988 -0
  81. package/cpp/llama.cpp/convert_hf_to_gguf.py +6195 -0
  82. package/cpp/llama.cpp/convert_hf_to_gguf_update.py +393 -0
  83. package/cpp/llama.cpp/convert_llama_ggml_to_gguf.py +450 -0
  84. package/cpp/llama.cpp/convert_lora_to_gguf.py +461 -0
  85. package/cpp/llama.cpp/flake.lock +58 -0
  86. package/cpp/llama.cpp/flake.nix +185 -0
  87. package/cpp/llama.cpp/ggml/CMakeLists.txt +388 -0
  88. package/cpp/llama.cpp/ggml/cmake/GitVars.cmake +22 -0
  89. package/cpp/llama.cpp/ggml/cmake/common.cmake +26 -0
  90. package/cpp/llama.cpp/ggml/cmake/ggml-config.cmake.in +152 -0
  91. package/cpp/llama.cpp/ggml/include/ggml-alloc.h +76 -0
  92. package/cpp/llama.cpp/ggml/include/ggml-backend.h +354 -0
  93. package/cpp/llama.cpp/ggml/include/ggml-blas.h +25 -0
  94. package/cpp/llama.cpp/ggml/include/ggml-cann.h +123 -0
  95. package/cpp/llama.cpp/ggml/include/ggml-cpp.h +39 -0
  96. package/cpp/llama.cpp/ggml/include/ggml-cpu.h +143 -0
  97. package/cpp/llama.cpp/ggml/include/ggml-cuda.h +47 -0
  98. package/cpp/llama.cpp/ggml/include/ggml-kompute.h +50 -0
  99. package/cpp/llama.cpp/ggml/include/ggml-metal.h +66 -0
  100. package/cpp/llama.cpp/ggml/include/ggml-opencl.h +26 -0
  101. package/cpp/llama.cpp/ggml/include/ggml-opt.h +216 -0
  102. package/cpp/llama.cpp/ggml/include/ggml-rpc.h +33 -0
  103. package/cpp/llama.cpp/ggml/include/ggml-sycl.h +49 -0
  104. package/cpp/llama.cpp/ggml/include/ggml-vulkan.h +29 -0
  105. package/cpp/llama.cpp/ggml/include/ggml.h +2192 -0
  106. package/cpp/llama.cpp/ggml/include/gguf.h +202 -0
  107. package/cpp/llama.cpp/ggml/src/CMakeLists.txt +345 -0
  108. package/cpp/llama.cpp/ggml/src/ggml-alloc.c +1042 -0
  109. package/cpp/llama.cpp/ggml/src/ggml-backend-impl.h +255 -0
  110. package/cpp/llama.cpp/ggml/src/ggml-backend-reg.cpp +586 -0
  111. package/cpp/llama.cpp/ggml/src/ggml-backend.cpp +2008 -0
  112. package/cpp/llama.cpp/ggml/src/ggml-blas/CMakeLists.txt +87 -0
  113. package/cpp/llama.cpp/ggml/src/ggml-blas/ggml-blas.cpp +517 -0
  114. package/cpp/llama.cpp/ggml/src/ggml-cann/CMakeLists.txt +74 -0
  115. package/cpp/llama.cpp/ggml/src/ggml-cann/Doxyfile +2579 -0
  116. package/cpp/llama.cpp/ggml/src/ggml-cann/acl_tensor.cpp +179 -0
  117. package/cpp/llama.cpp/ggml/src/ggml-cann/acl_tensor.h +258 -0
  118. package/cpp/llama.cpp/ggml/src/ggml-cann/aclnn_ops.cpp +2589 -0
  119. package/cpp/llama.cpp/ggml/src/ggml-cann/aclnn_ops.h +1083 -0
  120. package/cpp/llama.cpp/ggml/src/ggml-cann/common.h +420 -0
  121. package/cpp/llama.cpp/ggml/src/ggml-cann/ggml-cann.cpp +2554 -0
  122. package/cpp/llama.cpp/ggml/src/ggml-common.h +1857 -0
  123. package/cpp/llama.cpp/ggml/src/ggml-cpu/CMakeLists.txt +495 -0
  124. package/cpp/llama.cpp/ggml/src/ggml-cpu/amx/amx.cpp +221 -0
  125. package/cpp/llama.cpp/ggml/src/ggml-cpu/amx/amx.h +8 -0
  126. package/cpp/llama.cpp/ggml/src/ggml-cpu/amx/common.h +91 -0
  127. package/cpp/llama.cpp/ggml/src/ggml-cpu/amx/mmq.cpp +2511 -0
  128. package/cpp/llama.cpp/ggml/src/ggml-cpu/amx/mmq.h +10 -0
  129. package/cpp/llama.cpp/ggml/src/ggml-cpu/binary-ops.cpp +158 -0
  130. package/cpp/llama.cpp/ggml/src/ggml-cpu/binary-ops.h +16 -0
  131. package/cpp/llama.cpp/ggml/src/ggml-cpu/cmake/FindSIMD.cmake +100 -0
  132. package/cpp/llama.cpp/ggml/src/ggml-cpu/common.h +72 -0
  133. package/cpp/llama.cpp/ggml/src/ggml-cpu/cpu-feats-x86.cpp +327 -0
  134. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu-aarch64.cpp +6431 -0
  135. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu-aarch64.h +8 -0
  136. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu-hbm.cpp +55 -0
  137. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu-hbm.h +8 -0
  138. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu-impl.h +512 -0
  139. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu-quants.c +13131 -0
  140. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu-quants.h +63 -0
  141. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu-traits.cpp +36 -0
  142. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu-traits.h +38 -0
  143. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu.c +3492 -0
  144. package/cpp/llama.cpp/ggml/src/ggml-cpu/ggml-cpu.cpp +671 -0
  145. package/cpp/llama.cpp/ggml/src/ggml-cpu/kleidiai/kernels.cpp +254 -0
  146. package/cpp/llama.cpp/ggml/src/ggml-cpu/kleidiai/kernels.h +60 -0
  147. package/cpp/llama.cpp/ggml/src/ggml-cpu/kleidiai/kleidiai.cpp +287 -0
  148. package/cpp/llama.cpp/ggml/src/ggml-cpu/kleidiai/kleidiai.h +17 -0
  149. package/cpp/llama.cpp/ggml/src/ggml-cpu/llamafile/sgemm.cpp +3544 -0
  150. package/cpp/llama.cpp/ggml/src/ggml-cpu/llamafile/sgemm.h +14 -0
  151. package/cpp/llama.cpp/ggml/src/ggml-cpu/ops.cpp +8796 -0
  152. package/cpp/llama.cpp/ggml/src/ggml-cpu/ops.h +110 -0
  153. package/cpp/llama.cpp/ggml/src/ggml-cpu/simd-mappings.h +892 -0
  154. package/cpp/llama.cpp/ggml/src/ggml-cpu/unary-ops.cpp +186 -0
  155. package/cpp/llama.cpp/ggml/src/ggml-cpu/unary-ops.h +28 -0
  156. package/cpp/llama.cpp/ggml/src/ggml-cpu/vec.cpp +252 -0
  157. package/cpp/llama.cpp/ggml/src/ggml-cpu/vec.h +802 -0
  158. package/cpp/llama.cpp/ggml/src/ggml-cuda/CMakeLists.txt +184 -0
  159. package/cpp/llama.cpp/ggml/src/ggml-cuda/acc.cu +47 -0
  160. package/cpp/llama.cpp/ggml/src/ggml-cuda/acc.cuh +5 -0
  161. package/cpp/llama.cpp/ggml/src/ggml-cuda/arange.cu +34 -0
  162. package/cpp/llama.cpp/ggml/src/ggml-cuda/arange.cuh +5 -0
  163. package/cpp/llama.cpp/ggml/src/ggml-cuda/argmax.cu +91 -0
  164. package/cpp/llama.cpp/ggml/src/ggml-cuda/argmax.cuh +3 -0
  165. package/cpp/llama.cpp/ggml/src/ggml-cuda/argsort.cu +104 -0
  166. package/cpp/llama.cpp/ggml/src/ggml-cuda/argsort.cuh +3 -0
  167. package/cpp/llama.cpp/ggml/src/ggml-cuda/binbcast.cu +363 -0
  168. package/cpp/llama.cpp/ggml/src/ggml-cuda/binbcast.cuh +9 -0
  169. package/cpp/llama.cpp/ggml/src/ggml-cuda/clamp.cu +45 -0
  170. package/cpp/llama.cpp/ggml/src/ggml-cuda/clamp.cuh +5 -0
  171. package/cpp/llama.cpp/ggml/src/ggml-cuda/common.cuh +828 -0
  172. package/cpp/llama.cpp/ggml/src/ggml-cuda/concat.cu +221 -0
  173. package/cpp/llama.cpp/ggml/src/ggml-cuda/concat.cuh +5 -0
  174. package/cpp/llama.cpp/ggml/src/ggml-cuda/conv-transpose-1d.cu +89 -0
  175. package/cpp/llama.cpp/ggml/src/ggml-cuda/conv-transpose-1d.cuh +5 -0
  176. package/cpp/llama.cpp/ggml/src/ggml-cuda/convert.cu +730 -0
  177. package/cpp/llama.cpp/ggml/src/ggml-cuda/convert.cuh +26 -0
  178. package/cpp/llama.cpp/ggml/src/ggml-cuda/count-equal.cu +64 -0
  179. package/cpp/llama.cpp/ggml/src/ggml-cuda/count-equal.cuh +5 -0
  180. package/cpp/llama.cpp/ggml/src/ggml-cuda/cp-async.cuh +57 -0
  181. package/cpp/llama.cpp/ggml/src/ggml-cuda/cpy.cu +695 -0
  182. package/cpp/llama.cpp/ggml/src/ggml-cuda/cpy.cuh +11 -0
  183. package/cpp/llama.cpp/ggml/src/ggml-cuda/cross-entropy-loss.cu +189 -0
  184. package/cpp/llama.cpp/ggml/src/ggml-cuda/cross-entropy-loss.cuh +7 -0
  185. package/cpp/llama.cpp/ggml/src/ggml-cuda/dequantize.cuh +103 -0
  186. package/cpp/llama.cpp/ggml/src/ggml-cuda/diagmask.cu +40 -0
  187. package/cpp/llama.cpp/ggml/src/ggml-cuda/diagmask.cuh +5 -0
  188. package/cpp/llama.cpp/ggml/src/ggml-cuda/fattn-common.cuh +873 -0
  189. package/cpp/llama.cpp/ggml/src/ggml-cuda/fattn-mma-f16.cuh +1269 -0
  190. package/cpp/llama.cpp/ggml/src/ggml-cuda/fattn-tile-f16.cu +357 -0
  191. package/cpp/llama.cpp/ggml/src/ggml-cuda/fattn-tile-f16.cuh +3 -0
  192. package/cpp/llama.cpp/ggml/src/ggml-cuda/fattn-tile-f32.cu +365 -0
  193. package/cpp/llama.cpp/ggml/src/ggml-cuda/fattn-tile-f32.cuh +3 -0
  194. package/cpp/llama.cpp/ggml/src/ggml-cuda/fattn-vec-f16.cuh +437 -0
  195. package/cpp/llama.cpp/ggml/src/ggml-cuda/fattn-vec-f32.cuh +428 -0
  196. package/cpp/llama.cpp/ggml/src/ggml-cuda/fattn-wmma-f16.cu +634 -0
  197. package/cpp/llama.cpp/ggml/src/ggml-cuda/fattn-wmma-f16.cuh +3 -0
  198. package/cpp/llama.cpp/ggml/src/ggml-cuda/fattn.cu +345 -0
  199. package/cpp/llama.cpp/ggml/src/ggml-cuda/fattn.cuh +3 -0
  200. package/cpp/llama.cpp/ggml/src/ggml-cuda/getrows.cu +275 -0
  201. package/cpp/llama.cpp/ggml/src/ggml-cuda/getrows.cuh +15 -0
  202. package/cpp/llama.cpp/ggml/src/ggml-cuda/ggml-cuda.cu +3501 -0
  203. package/cpp/llama.cpp/ggml/src/ggml-cuda/gla.cu +93 -0
  204. package/cpp/llama.cpp/ggml/src/ggml-cuda/gla.cuh +3 -0
  205. package/cpp/llama.cpp/ggml/src/ggml-cuda/im2col.cu +103 -0
  206. package/cpp/llama.cpp/ggml/src/ggml-cuda/im2col.cuh +5 -0
  207. package/cpp/llama.cpp/ggml/src/ggml-cuda/mma.cuh +396 -0
  208. package/cpp/llama.cpp/ggml/src/ggml-cuda/mmq.cu +322 -0
  209. package/cpp/llama.cpp/ggml/src/ggml-cuda/mmq.cuh +3217 -0
  210. package/cpp/llama.cpp/ggml/src/ggml-cuda/mmv.cu +336 -0
  211. package/cpp/llama.cpp/ggml/src/ggml-cuda/mmv.cuh +12 -0
  212. package/cpp/llama.cpp/ggml/src/ggml-cuda/mmvq.cu +595 -0
  213. package/cpp/llama.cpp/ggml/src/ggml-cuda/mmvq.cuh +12 -0
  214. package/cpp/llama.cpp/ggml/src/ggml-cuda/norm.cu +458 -0
  215. package/cpp/llama.cpp/ggml/src/ggml-cuda/norm.cuh +11 -0
  216. package/cpp/llama.cpp/ggml/src/ggml-cuda/opt-step-adamw.cu +78 -0
  217. package/cpp/llama.cpp/ggml/src/ggml-cuda/opt-step-adamw.cuh +5 -0
  218. package/cpp/llama.cpp/ggml/src/ggml-cuda/out-prod.cu +68 -0
  219. package/cpp/llama.cpp/ggml/src/ggml-cuda/out-prod.cuh +3 -0
  220. package/cpp/llama.cpp/ggml/src/ggml-cuda/pad.cu +49 -0
  221. package/cpp/llama.cpp/ggml/src/ggml-cuda/pad.cuh +5 -0
  222. package/cpp/llama.cpp/ggml/src/ggml-cuda/pool2d.cu +94 -0
  223. package/cpp/llama.cpp/ggml/src/ggml-cuda/pool2d.cuh +5 -0
  224. package/cpp/llama.cpp/ggml/src/ggml-cuda/quantize.cu +189 -0
  225. package/cpp/llama.cpp/ggml/src/ggml-cuda/quantize.cuh +27 -0
  226. package/cpp/llama.cpp/ggml/src/ggml-cuda/rope.cu +456 -0
  227. package/cpp/llama.cpp/ggml/src/ggml-cuda/rope.cuh +7 -0
  228. package/cpp/llama.cpp/ggml/src/ggml-cuda/scale.cu +31 -0
  229. package/cpp/llama.cpp/ggml/src/ggml-cuda/scale.cuh +5 -0
  230. package/cpp/llama.cpp/ggml/src/ggml-cuda/softmax.cu +283 -0
  231. package/cpp/llama.cpp/ggml/src/ggml-cuda/softmax.cuh +7 -0
  232. package/cpp/llama.cpp/ggml/src/ggml-cuda/ssm-conv.cu +148 -0
  233. package/cpp/llama.cpp/ggml/src/ggml-cuda/ssm-conv.cuh +3 -0
  234. package/cpp/llama.cpp/ggml/src/ggml-cuda/ssm-scan.cu +153 -0
  235. package/cpp/llama.cpp/ggml/src/ggml-cuda/ssm-scan.cuh +3 -0
  236. package/cpp/llama.cpp/ggml/src/ggml-cuda/sum.cu +45 -0
  237. package/cpp/llama.cpp/ggml/src/ggml-cuda/sum.cuh +5 -0
  238. package/cpp/llama.cpp/ggml/src/ggml-cuda/sumrows.cu +39 -0
  239. package/cpp/llama.cpp/ggml/src/ggml-cuda/sumrows.cuh +5 -0
  240. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_1-ncols2_16.cu +5 -0
  241. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_1-ncols2_8.cu +10 -0
  242. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_16-ncols2_1.cu +10 -0
  243. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_16-ncols2_2.cu +10 -0
  244. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_16-ncols2_4.cu +10 -0
  245. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_2-ncols2_16.cu +5 -0
  246. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_2-ncols2_4.cu +10 -0
  247. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_2-ncols2_8.cu +10 -0
  248. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_32-ncols2_1.cu +10 -0
  249. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_32-ncols2_2.cu +10 -0
  250. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_4-ncols2_16.cu +5 -0
  251. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_4-ncols2_2.cu +10 -0
  252. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_4-ncols2_4.cu +10 -0
  253. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_4-ncols2_8.cu +10 -0
  254. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_64-ncols2_1.cu +10 -0
  255. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_8-ncols2_1.cu +10 -0
  256. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_8-ncols2_2.cu +10 -0
  257. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_8-ncols2_4.cu +10 -0
  258. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_8-ncols2_8.cu +10 -0
  259. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-f16-f16.cu +5 -0
  260. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-f16-q4_0.cu +5 -0
  261. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-f16-q4_1.cu +5 -0
  262. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-f16-q5_0.cu +5 -0
  263. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-f16-q5_1.cu +5 -0
  264. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-f16-q8_0.cu +5 -0
  265. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q4_0-f16.cu +5 -0
  266. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q4_0-q4_0.cu +5 -0
  267. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q4_0-q4_1.cu +5 -0
  268. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q4_0-q5_0.cu +5 -0
  269. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q4_0-q5_1.cu +5 -0
  270. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q4_0-q8_0.cu +5 -0
  271. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q4_1-f16.cu +5 -0
  272. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q4_1-q4_0.cu +5 -0
  273. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q4_1-q4_1.cu +5 -0
  274. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q4_1-q5_0.cu +5 -0
  275. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q4_1-q5_1.cu +5 -0
  276. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q4_1-q8_0.cu +5 -0
  277. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q5_0-f16.cu +5 -0
  278. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q5_0-q4_0.cu +5 -0
  279. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q5_0-q4_1.cu +5 -0
  280. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q5_0-q5_0.cu +5 -0
  281. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q5_0-q5_1.cu +5 -0
  282. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q5_0-q8_0.cu +5 -0
  283. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q5_1-f16.cu +5 -0
  284. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q5_1-q4_0.cu +5 -0
  285. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q5_1-q4_1.cu +5 -0
  286. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q5_1-q5_0.cu +5 -0
  287. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q5_1-q5_1.cu +5 -0
  288. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q5_1-q8_0.cu +5 -0
  289. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q8_0-f16.cu +5 -0
  290. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q8_0-q4_0.cu +5 -0
  291. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q8_0-q4_1.cu +5 -0
  292. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q8_0-q5_0.cu +5 -0
  293. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q8_0-q5_1.cu +5 -0
  294. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs128-q8_0-q8_0.cu +5 -0
  295. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs256-f16-f16.cu +5 -0
  296. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs64-f16-f16.cu +5 -0
  297. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs64-f16-q4_0.cu +5 -0
  298. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs64-f16-q4_1.cu +5 -0
  299. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs64-f16-q5_0.cu +5 -0
  300. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs64-f16-q5_1.cu +5 -0
  301. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f16-instance-hs64-f16-q8_0.cu +5 -0
  302. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-f16-f16.cu +5 -0
  303. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-f16-q4_0.cu +5 -0
  304. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-f16-q4_1.cu +5 -0
  305. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-f16-q5_0.cu +5 -0
  306. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-f16-q5_1.cu +5 -0
  307. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-f16-q8_0.cu +5 -0
  308. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q4_0-f16.cu +5 -0
  309. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q4_0-q4_0.cu +5 -0
  310. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q4_0-q4_1.cu +5 -0
  311. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q4_0-q5_0.cu +5 -0
  312. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q4_0-q5_1.cu +5 -0
  313. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q4_0-q8_0.cu +5 -0
  314. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q4_1-f16.cu +5 -0
  315. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q4_1-q4_0.cu +5 -0
  316. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q4_1-q4_1.cu +5 -0
  317. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q4_1-q5_0.cu +5 -0
  318. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q4_1-q5_1.cu +5 -0
  319. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q4_1-q8_0.cu +5 -0
  320. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q5_0-f16.cu +5 -0
  321. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q5_0-q4_0.cu +5 -0
  322. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q5_0-q4_1.cu +5 -0
  323. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q5_0-q5_0.cu +5 -0
  324. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q5_0-q5_1.cu +5 -0
  325. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q5_0-q8_0.cu +5 -0
  326. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q5_1-f16.cu +5 -0
  327. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q5_1-q4_0.cu +5 -0
  328. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q5_1-q4_1.cu +5 -0
  329. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q5_1-q5_0.cu +5 -0
  330. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q5_1-q5_1.cu +5 -0
  331. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q5_1-q8_0.cu +5 -0
  332. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q8_0-f16.cu +5 -0
  333. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q8_0-q4_0.cu +5 -0
  334. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q8_0-q4_1.cu +5 -0
  335. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q8_0-q5_0.cu +5 -0
  336. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q8_0-q5_1.cu +5 -0
  337. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs128-q8_0-q8_0.cu +5 -0
  338. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs256-f16-f16.cu +5 -0
  339. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs64-f16-f16.cu +5 -0
  340. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs64-f16-q4_0.cu +5 -0
  341. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs64-f16-q4_1.cu +5 -0
  342. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs64-f16-q5_0.cu +5 -0
  343. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs64-f16-q5_1.cu +5 -0
  344. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/fattn-vec-f32-instance-hs64-f16-q8_0.cu +5 -0
  345. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/generate_cu_files.py +78 -0
  346. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-iq1_s.cu +5 -0
  347. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-iq2_s.cu +5 -0
  348. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-iq2_xs.cu +5 -0
  349. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-iq2_xxs.cu +5 -0
  350. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-iq3_s.cu +5 -0
  351. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-iq3_xxs.cu +5 -0
  352. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-iq4_nl.cu +5 -0
  353. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-iq4_xs.cu +5 -0
  354. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-q2_k.cu +5 -0
  355. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-q3_k.cu +5 -0
  356. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-q4_0.cu +5 -0
  357. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-q4_1.cu +5 -0
  358. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-q4_k.cu +5 -0
  359. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-q5_0.cu +5 -0
  360. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-q5_1.cu +5 -0
  361. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-q5_k.cu +5 -0
  362. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-q6_k.cu +5 -0
  363. package/cpp/llama.cpp/ggml/src/ggml-cuda/template-instances/mmq-instance-q8_0.cu +5 -0
  364. package/cpp/llama.cpp/ggml/src/ggml-cuda/tsembd.cu +47 -0
  365. package/cpp/llama.cpp/ggml/src/ggml-cuda/tsembd.cuh +5 -0
  366. package/cpp/llama.cpp/ggml/src/ggml-cuda/unary.cu +279 -0
  367. package/cpp/llama.cpp/ggml/src/ggml-cuda/unary.cuh +57 -0
  368. package/cpp/llama.cpp/ggml/src/ggml-cuda/upscale.cu +51 -0
  369. package/cpp/llama.cpp/ggml/src/ggml-cuda/upscale.cuh +5 -0
  370. package/cpp/llama.cpp/ggml/src/ggml-cuda/vecdotq.cuh +1135 -0
  371. package/cpp/llama.cpp/ggml/src/ggml-cuda/vendors/cuda.h +15 -0
  372. package/cpp/llama.cpp/ggml/src/ggml-cuda/vendors/hip.h +243 -0
  373. package/cpp/llama.cpp/ggml/src/ggml-cuda/vendors/musa.h +140 -0
  374. package/cpp/llama.cpp/ggml/src/ggml-cuda/wkv.cu +199 -0
  375. package/cpp/llama.cpp/ggml/src/ggml-cuda/wkv.cuh +7 -0
  376. package/cpp/llama.cpp/ggml/src/ggml-hip/CMakeLists.txt +131 -0
  377. package/cpp/llama.cpp/ggml/src/ggml-impl.h +601 -0
  378. package/cpp/llama.cpp/ggml/src/ggml-kompute/CMakeLists.txt +166 -0
  379. package/cpp/llama.cpp/ggml/src/ggml-kompute/ggml-kompute.cpp +2251 -0
  380. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/common.comp +112 -0
  381. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_add.comp +58 -0
  382. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_addrow.comp +25 -0
  383. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_cpy_f16_f16.comp +52 -0
  384. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_cpy_f16_f32.comp +52 -0
  385. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_cpy_f32_f16.comp +52 -0
  386. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_cpy_f32_f32.comp +52 -0
  387. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_diagmask.comp +30 -0
  388. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_gelu.comp +22 -0
  389. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_getrows.comp +17 -0
  390. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_getrows_f16.comp +31 -0
  391. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_getrows_f32.comp +31 -0
  392. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_getrows_q4_0.comp +38 -0
  393. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_getrows_q4_1.comp +39 -0
  394. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_getrows_q6_k.comp +44 -0
  395. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_mul.comp +52 -0
  396. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_mul_mat_f16.comp +69 -0
  397. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_mul_mat_mat_f32.comp +51 -0
  398. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_mul_mat_q4_0.comp +33 -0
  399. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_mul_mat_q4_1.comp +35 -0
  400. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_mul_mat_q4_k.comp +140 -0
  401. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_mul_mat_q6_k.comp +106 -0
  402. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_mul_mat_q8_0.comp +73 -0
  403. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_mul_mv_q_n.comp +52 -0
  404. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_mul_mv_q_n_pre.comp +28 -0
  405. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_norm.comp +84 -0
  406. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_relu.comp +21 -0
  407. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_rmsnorm.comp +53 -0
  408. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_rope_neox_f16.comp +52 -0
  409. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_rope_neox_f32.comp +52 -0
  410. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_rope_norm_f16.comp +52 -0
  411. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_rope_norm_f32.comp +52 -0
  412. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_scale.comp +19 -0
  413. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_scale_8.comp +23 -0
  414. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_silu.comp +22 -0
  415. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/op_softmax.comp +72 -0
  416. package/cpp/llama.cpp/ggml/src/ggml-kompute/kompute-shaders/rope_common.comp +71 -0
  417. package/cpp/llama.cpp/ggml/src/ggml-metal/CMakeLists.txt +120 -0
  418. package/cpp/llama.cpp/ggml/src/ggml-metal/ggml-metal-impl.h +618 -0
  419. package/cpp/llama.cpp/ggml/src/ggml-metal/ggml-metal.m +5916 -0
  420. package/cpp/llama.cpp/ggml/src/ggml-metal/ggml-metal.metal +6891 -0
  421. package/cpp/llama.cpp/ggml/src/ggml-musa/CMakeLists.txt +107 -0
  422. package/cpp/llama.cpp/ggml/src/ggml-opencl/CMakeLists.txt +96 -0
  423. package/cpp/llama.cpp/ggml/src/ggml-opencl/ggml-opencl.cpp +4966 -0
  424. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/add.cl +83 -0
  425. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/clamp.cl +20 -0
  426. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/cpy.cl +184 -0
  427. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/cvt.cl +118 -0
  428. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/diag_mask_inf.cl +58 -0
  429. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/embed_kernel.py +26 -0
  430. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/gelu.cl +62 -0
  431. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/gemv_noshuffle.cl +268 -0
  432. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/gemv_noshuffle_general.cl +274 -0
  433. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/get_rows.cl +163 -0
  434. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/im2col_f16.cl +57 -0
  435. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/im2col_f32.cl +57 -0
  436. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/mul.cl +79 -0
  437. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/mul_mat_Ab_Bi_8x4.cl +139 -0
  438. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/mul_mv_f16_f16.cl +118 -0
  439. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/mul_mv_f16_f32.cl +118 -0
  440. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/mul_mv_f16_f32_1row.cl +94 -0
  441. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/mul_mv_f16_f32_l4.cl +84 -0
  442. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/mul_mv_f32_f32.cl +118 -0
  443. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/mul_mv_q4_0_f32.cl +192 -0
  444. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/mul_mv_q4_0_f32_1d_16x_flat.cl +307 -0
  445. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/mul_mv_q4_0_f32_1d_8x_flat.cl +265 -0
  446. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/mul_mv_q4_0_f32_8x_flat.cl +272 -0
  447. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/mul_mv_q4_0_f32_v.cl +254 -0
  448. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/mul_mv_q6_k.cl +190 -0
  449. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/norm.cl +81 -0
  450. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/relu.cl +16 -0
  451. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/rms_norm.cl +96 -0
  452. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/rope.cl +721 -0
  453. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/scale.cl +16 -0
  454. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/silu.cl +30 -0
  455. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/softmax_4_f16.cl +87 -0
  456. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/softmax_4_f32.cl +87 -0
  457. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/softmax_f16.cl +86 -0
  458. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/softmax_f32.cl +86 -0
  459. package/cpp/llama.cpp/ggml/src/ggml-opencl/kernels/transpose.cl +84 -0
  460. package/cpp/llama.cpp/ggml/src/ggml-opt.cpp +854 -0
  461. package/cpp/llama.cpp/ggml/src/ggml-quants.c +5232 -0
  462. package/cpp/llama.cpp/ggml/src/ggml-quants.h +100 -0
  463. package/cpp/llama.cpp/ggml/src/ggml-rpc/CMakeLists.txt +9 -0
  464. package/cpp/llama.cpp/ggml/src/ggml-rpc/ggml-rpc.cpp +1813 -0
  465. package/cpp/llama.cpp/ggml/src/ggml-sycl/CMakeLists.txt +183 -0
  466. package/cpp/llama.cpp/ggml/src/ggml-sycl/backend.hpp +37 -0
  467. package/cpp/llama.cpp/ggml/src/ggml-sycl/binbcast.cpp +350 -0
  468. package/cpp/llama.cpp/ggml/src/ggml-sycl/binbcast.hpp +39 -0
  469. package/cpp/llama.cpp/ggml/src/ggml-sycl/common.cpp +83 -0
  470. package/cpp/llama.cpp/ggml/src/ggml-sycl/common.hpp +493 -0
  471. package/cpp/llama.cpp/ggml/src/ggml-sycl/concat.cpp +197 -0
  472. package/cpp/llama.cpp/ggml/src/ggml-sycl/concat.hpp +20 -0
  473. package/cpp/llama.cpp/ggml/src/ggml-sycl/conv.cpp +100 -0
  474. package/cpp/llama.cpp/ggml/src/ggml-sycl/conv.hpp +20 -0
  475. package/cpp/llama.cpp/ggml/src/ggml-sycl/convert.cpp +596 -0
  476. package/cpp/llama.cpp/ggml/src/ggml-sycl/convert.hpp +34 -0
  477. package/cpp/llama.cpp/ggml/src/ggml-sycl/cpy.cpp +701 -0
  478. package/cpp/llama.cpp/ggml/src/ggml-sycl/cpy.hpp +11 -0
  479. package/cpp/llama.cpp/ggml/src/ggml-sycl/dequantize.hpp +753 -0
  480. package/cpp/llama.cpp/ggml/src/ggml-sycl/dmmv.cpp +1154 -0
  481. package/cpp/llama.cpp/ggml/src/ggml-sycl/dmmv.hpp +27 -0
  482. package/cpp/llama.cpp/ggml/src/ggml-sycl/dpct/helper.hpp +2957 -0
  483. package/cpp/llama.cpp/ggml/src/ggml-sycl/element_wise.cpp +1559 -0
  484. package/cpp/llama.cpp/ggml/src/ggml-sycl/element_wise.hpp +75 -0
  485. package/cpp/llama.cpp/ggml/src/ggml-sycl/gemm.hpp +70 -0
  486. package/cpp/llama.cpp/ggml/src/ggml-sycl/getrows.cpp +311 -0
  487. package/cpp/llama.cpp/ggml/src/ggml-sycl/getrows.hpp +20 -0
  488. package/cpp/llama.cpp/ggml/src/ggml-sycl/ggml-sycl.cpp +4302 -0
  489. package/cpp/llama.cpp/ggml/src/ggml-sycl/gla.cpp +105 -0
  490. package/cpp/llama.cpp/ggml/src/ggml-sycl/gla.hpp +8 -0
  491. package/cpp/llama.cpp/ggml/src/ggml-sycl/im2col.cpp +136 -0
  492. package/cpp/llama.cpp/ggml/src/ggml-sycl/im2col.hpp +21 -0
  493. package/cpp/llama.cpp/ggml/src/ggml-sycl/mmq.cpp +3030 -0
  494. package/cpp/llama.cpp/ggml/src/ggml-sycl/mmq.hpp +33 -0
  495. package/cpp/llama.cpp/ggml/src/ggml-sycl/mmvq.cpp +1081 -0
  496. package/cpp/llama.cpp/ggml/src/ggml-sycl/mmvq.hpp +27 -0
  497. package/cpp/llama.cpp/ggml/src/ggml-sycl/norm.cpp +474 -0
  498. package/cpp/llama.cpp/ggml/src/ggml-sycl/norm.hpp +26 -0
  499. package/cpp/llama.cpp/ggml/src/ggml-sycl/outprod.cpp +46 -0
  500. package/cpp/llama.cpp/ggml/src/ggml-sycl/outprod.hpp +10 -0
  501. package/cpp/llama.cpp/ggml/src/ggml-sycl/presets.hpp +74 -0
  502. package/cpp/llama.cpp/ggml/src/ggml-sycl/quants.hpp +61 -0
  503. package/cpp/llama.cpp/ggml/src/ggml-sycl/rope.cpp +362 -0
  504. package/cpp/llama.cpp/ggml/src/ggml-sycl/rope.hpp +20 -0
  505. package/cpp/llama.cpp/ggml/src/ggml-sycl/softmax.cpp +264 -0
  506. package/cpp/llama.cpp/ggml/src/ggml-sycl/softmax.hpp +20 -0
  507. package/cpp/llama.cpp/ggml/src/ggml-sycl/sycl_hw.cpp +13 -0
  508. package/cpp/llama.cpp/ggml/src/ggml-sycl/sycl_hw.hpp +23 -0
  509. package/cpp/llama.cpp/ggml/src/ggml-sycl/tsembd.cpp +73 -0
  510. package/cpp/llama.cpp/ggml/src/ggml-sycl/tsembd.hpp +20 -0
  511. package/cpp/llama.cpp/ggml/src/ggml-sycl/vecdotq.hpp +1189 -0
  512. package/cpp/llama.cpp/ggml/src/ggml-sycl/wkv.cpp +305 -0
  513. package/cpp/llama.cpp/ggml/src/ggml-sycl/wkv.hpp +10 -0
  514. package/cpp/llama.cpp/ggml/src/ggml-threading.cpp +12 -0
  515. package/cpp/llama.cpp/ggml/src/ggml-threading.h +14 -0
  516. package/cpp/llama.cpp/ggml/src/ggml-vulkan/CMakeLists.txt +202 -0
  517. package/cpp/llama.cpp/ggml/src/ggml-vulkan/cmake/host-toolchain.cmake.in +15 -0
  518. package/cpp/llama.cpp/ggml/src/ggml-vulkan/ggml-vulkan.cpp +10502 -0
  519. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/CMakeLists.txt +22 -0
  520. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/acc.comp +29 -0
  521. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/add.comp +29 -0
  522. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/argmax.comp +51 -0
  523. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/argsort.comp +69 -0
  524. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/clamp.comp +17 -0
  525. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/concat.comp +41 -0
  526. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/contig_copy.comp +49 -0
  527. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/conv2d_dw.comp +105 -0
  528. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/copy.comp +23 -0
  529. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/copy_from_quant.comp +51 -0
  530. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/copy_to_quant.comp +242 -0
  531. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/cos.comp +17 -0
  532. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/count_equal.comp +31 -0
  533. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_f32.comp +20 -0
  534. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs.comp +462 -0
  535. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs_cm2.comp +699 -0
  536. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_head.comp +13 -0
  537. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_iq1_m.comp +42 -0
  538. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_iq1_s.comp +35 -0
  539. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_iq2_s.comp +44 -0
  540. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_iq2_xs.comp +43 -0
  541. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_iq2_xxs.comp +48 -0
  542. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_iq3_s.comp +39 -0
  543. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_iq3_xxs.comp +49 -0
  544. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_iq4_nl.comp +32 -0
  545. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_iq4_xs.comp +34 -0
  546. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q2_k.comp +34 -0
  547. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q3_k.comp +42 -0
  548. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q4_0.comp +30 -0
  549. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q4_1.comp +32 -0
  550. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q4_k.comp +68 -0
  551. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q5_0.comp +34 -0
  552. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q5_1.comp +35 -0
  553. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q5_k.comp +70 -0
  554. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q6_k.comp +33 -0
  555. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q8_0.comp +31 -0
  556. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/diag_mask_inf.comp +34 -0
  557. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/div.comp +27 -0
  558. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn.comp +483 -0
  559. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_cm2.comp +383 -0
  560. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_split_k_reduce.comp +59 -0
  561. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/gelu.comp +25 -0
  562. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/gelu_quick.comp +23 -0
  563. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/generic_binary_head.comp +64 -0
  564. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/generic_head.comp +9 -0
  565. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/generic_unary_head.comp +76 -0
  566. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/get_rows.comp +33 -0
  567. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/get_rows_quant.comp +41 -0
  568. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/group_norm.comp +66 -0
  569. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/im2col.comp +100 -0
  570. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/l2_norm.comp +41 -0
  571. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/leaky_relu.comp +22 -0
  572. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul.comp +27 -0
  573. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_split_k_reduce.comp +48 -0
  574. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec.comp +169 -0
  575. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_base.comp +118 -0
  576. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iq1_m.comp +82 -0
  577. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iq1_s.comp +79 -0
  578. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iq2_s.comp +90 -0
  579. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iq2_xs.comp +87 -0
  580. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iq2_xxs.comp +87 -0
  581. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iq3_s.comp +90 -0
  582. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iq3_xxs.comp +88 -0
  583. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_nc.comp +118 -0
  584. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_p021.comp +154 -0
  585. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_q2_k.comp +130 -0
  586. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_q3_k.comp +132 -0
  587. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_q4_k.comp +136 -0
  588. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_q5_k.comp +167 -0
  589. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_q6_k.comp +130 -0
  590. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm.comp +868 -0
  591. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_cm2.comp +441 -0
  592. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq.comp +442 -0
  593. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_funcs.comp +99 -0
  594. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/norm.comp +44 -0
  595. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/opt_step_adamw.comp +42 -0
  596. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/pad.comp +28 -0
  597. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/pool2d.comp +74 -0
  598. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/quantize_q8_1.comp +77 -0
  599. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/relu.comp +21 -0
  600. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/repeat.comp +26 -0
  601. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/repeat_back.comp +37 -0
  602. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/rms_norm.comp +52 -0
  603. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/rms_norm_back.comp +55 -0
  604. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/rope_head.comp +58 -0
  605. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/rope_multi.comp +60 -0
  606. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/rope_neox.comp +43 -0
  607. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/rope_norm.comp +43 -0
  608. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/rope_vision.comp +47 -0
  609. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/scale.comp +24 -0
  610. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/sigmoid.comp +20 -0
  611. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/silu.comp +22 -0
  612. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/silu_back.comp +26 -0
  613. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/sin.comp +17 -0
  614. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/soft_max.comp +173 -0
  615. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/soft_max_back.comp +50 -0
  616. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/square.comp +17 -0
  617. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/sub.comp +29 -0
  618. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/sum_rows.comp +37 -0
  619. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/tanh.comp +20 -0
  620. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/test_bfloat16_support.comp +7 -0
  621. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/test_coopmat2_support.comp +7 -0
  622. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/test_coopmat_support.comp +7 -0
  623. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/test_integer_dot_support.comp +7 -0
  624. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/timestep_embedding.comp +41 -0
  625. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/types.comp +1373 -0
  626. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/upscale.comp +36 -0
  627. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/vulkan-shaders-gen.cpp +740 -0
  628. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/wkv6.comp +87 -0
  629. package/cpp/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/wkv7.comp +91 -0
  630. package/cpp/llama.cpp/ggml/src/ggml.c +6499 -0
  631. package/cpp/llama.cpp/ggml/src/gguf.cpp +1330 -0
  632. package/cpp/llama.cpp/gguf-py/LICENSE +21 -0
  633. package/cpp/llama.cpp/gguf-py/README.md +99 -0
  634. package/cpp/llama.cpp/gguf-py/examples/reader.py +49 -0
  635. package/cpp/llama.cpp/gguf-py/examples/writer.py +39 -0
  636. package/cpp/llama.cpp/gguf-py/gguf/__init__.py +9 -0
  637. package/cpp/llama.cpp/gguf-py/gguf/constants.py +2296 -0
  638. package/cpp/llama.cpp/gguf-py/gguf/gguf.py +15 -0
  639. package/cpp/llama.cpp/gguf-py/gguf/gguf_reader.py +367 -0
  640. package/cpp/llama.cpp/gguf-py/gguf/gguf_writer.py +1041 -0
  641. package/cpp/llama.cpp/gguf-py/gguf/lazy.py +223 -0
  642. package/cpp/llama.cpp/gguf-py/gguf/metadata.py +642 -0
  643. package/cpp/llama.cpp/gguf-py/gguf/py.typed +0 -0
  644. package/cpp/llama.cpp/gguf-py/gguf/quants.py +1269 -0
  645. package/cpp/llama.cpp/gguf-py/gguf/scripts/gguf_convert_endian.py +182 -0
  646. package/cpp/llama.cpp/gguf-py/gguf/scripts/gguf_dump.py +454 -0
  647. package/cpp/llama.cpp/gguf-py/gguf/scripts/gguf_editor_gui.py +1610 -0
  648. package/cpp/llama.cpp/gguf-py/gguf/scripts/gguf_hash.py +102 -0
  649. package/cpp/llama.cpp/gguf-py/gguf/scripts/gguf_new_metadata.py +207 -0
  650. package/cpp/llama.cpp/gguf-py/gguf/scripts/gguf_set_metadata.py +95 -0
  651. package/cpp/llama.cpp/gguf-py/gguf/tensor_mapping.py +1172 -0
  652. package/cpp/llama.cpp/gguf-py/gguf/utility.py +264 -0
  653. package/cpp/llama.cpp/gguf-py/gguf/vocab.py +492 -0
  654. package/cpp/llama.cpp/gguf-py/pyproject.toml +43 -0
  655. package/cpp/llama.cpp/gguf-py/tests/__init__.py +1 -0
  656. package/cpp/llama.cpp/gguf-py/tests/test_metadata.py +238 -0
  657. package/cpp/llama.cpp/gguf-py/tests/test_quants.py +238 -0
  658. package/cpp/llama.cpp/grammars/README.md +382 -0
  659. package/cpp/llama.cpp/grammars/arithmetic.gbnf +6 -0
  660. package/cpp/llama.cpp/grammars/c.gbnf +42 -0
  661. package/cpp/llama.cpp/grammars/chess.gbnf +13 -0
  662. package/cpp/llama.cpp/grammars/english.gbnf +6 -0
  663. package/cpp/llama.cpp/grammars/japanese.gbnf +7 -0
  664. package/cpp/llama.cpp/grammars/json.gbnf +25 -0
  665. package/cpp/llama.cpp/grammars/json_arr.gbnf +34 -0
  666. package/cpp/llama.cpp/grammars/list.gbnf +4 -0
  667. package/cpp/llama.cpp/include/llama-cpp.h +30 -0
  668. package/cpp/llama.cpp/include/llama.h +1440 -0
  669. package/cpp/llama.cpp/licenses/LICENSE-curl +9 -0
  670. package/cpp/llama.cpp/licenses/LICENSE-httplib +21 -0
  671. package/cpp/llama.cpp/licenses/LICENSE-jsonhpp +21 -0
  672. package/cpp/llama.cpp/licenses/LICENSE-linenoise +26 -0
  673. package/cpp/llama.cpp/media/llama0-banner.png +0 -0
  674. package/cpp/llama.cpp/media/llama0-logo.png +0 -0
  675. package/cpp/llama.cpp/media/llama1-banner.png +0 -0
  676. package/cpp/llama.cpp/media/llama1-logo.png +0 -0
  677. package/cpp/llama.cpp/media/llama1-logo.svg +34 -0
  678. package/cpp/llama.cpp/media/matmul.png +0 -0
  679. package/cpp/llama.cpp/media/matmul.svg +1238 -0
  680. package/cpp/llama.cpp/models/ggml-vocab-aquila.gguf +0 -0
  681. package/cpp/llama.cpp/models/ggml-vocab-baichuan.gguf +0 -0
  682. package/cpp/llama.cpp/models/ggml-vocab-bert-bge.gguf +0 -0
  683. package/cpp/llama.cpp/models/ggml-vocab-bert-bge.gguf.inp +112 -0
  684. package/cpp/llama.cpp/models/ggml-vocab-bert-bge.gguf.out +46 -0
  685. package/cpp/llama.cpp/models/ggml-vocab-chameleon.gguf.inp +112 -0
  686. package/cpp/llama.cpp/models/ggml-vocab-chameleon.gguf.out +46 -0
  687. package/cpp/llama.cpp/models/ggml-vocab-command-r.gguf +0 -0
  688. package/cpp/llama.cpp/models/ggml-vocab-command-r.gguf.inp +112 -0
  689. package/cpp/llama.cpp/models/ggml-vocab-command-r.gguf.out +46 -0
  690. package/cpp/llama.cpp/models/ggml-vocab-deepseek-coder.gguf +0 -0
  691. package/cpp/llama.cpp/models/ggml-vocab-deepseek-coder.gguf.inp +112 -0
  692. package/cpp/llama.cpp/models/ggml-vocab-deepseek-coder.gguf.out +46 -0
  693. package/cpp/llama.cpp/models/ggml-vocab-deepseek-llm.gguf +0 -0
  694. package/cpp/llama.cpp/models/ggml-vocab-deepseek-llm.gguf.inp +112 -0
  695. package/cpp/llama.cpp/models/ggml-vocab-deepseek-llm.gguf.out +46 -0
  696. package/cpp/llama.cpp/models/ggml-vocab-deepseek-r1-qwen.gguf.inp +112 -0
  697. package/cpp/llama.cpp/models/ggml-vocab-deepseek-r1-qwen.gguf.out +46 -0
  698. package/cpp/llama.cpp/models/ggml-vocab-falcon.gguf +0 -0
  699. package/cpp/llama.cpp/models/ggml-vocab-falcon.gguf.inp +112 -0
  700. package/cpp/llama.cpp/models/ggml-vocab-falcon.gguf.out +46 -0
  701. package/cpp/llama.cpp/models/ggml-vocab-gpt-2.gguf +0 -0
  702. package/cpp/llama.cpp/models/ggml-vocab-gpt-2.gguf.inp +112 -0
  703. package/cpp/llama.cpp/models/ggml-vocab-gpt-2.gguf.out +46 -0
  704. package/cpp/llama.cpp/models/ggml-vocab-gpt-4o.gguf.inp +112 -0
  705. package/cpp/llama.cpp/models/ggml-vocab-gpt-4o.gguf.out +46 -0
  706. package/cpp/llama.cpp/models/ggml-vocab-gpt-neox.gguf +0 -0
  707. package/cpp/llama.cpp/models/ggml-vocab-llama-bpe.gguf +0 -0
  708. package/cpp/llama.cpp/models/ggml-vocab-llama-bpe.gguf.inp +112 -0
  709. package/cpp/llama.cpp/models/ggml-vocab-llama-bpe.gguf.out +46 -0
  710. package/cpp/llama.cpp/models/ggml-vocab-llama-spm.gguf +0 -0
  711. package/cpp/llama.cpp/models/ggml-vocab-llama-spm.gguf.inp +112 -0
  712. package/cpp/llama.cpp/models/ggml-vocab-llama-spm.gguf.out +46 -0
  713. package/cpp/llama.cpp/models/ggml-vocab-llama4.gguf.inp +112 -0
  714. package/cpp/llama.cpp/models/ggml-vocab-llama4.gguf.out +46 -0
  715. package/cpp/llama.cpp/models/ggml-vocab-mpt.gguf +0 -0
  716. package/cpp/llama.cpp/models/ggml-vocab-mpt.gguf.inp +112 -0
  717. package/cpp/llama.cpp/models/ggml-vocab-mpt.gguf.out +46 -0
  718. package/cpp/llama.cpp/models/ggml-vocab-phi-3.gguf +0 -0
  719. package/cpp/llama.cpp/models/ggml-vocab-phi-3.gguf.inp +112 -0
  720. package/cpp/llama.cpp/models/ggml-vocab-phi-3.gguf.out +46 -0
  721. package/cpp/llama.cpp/models/ggml-vocab-pixtral.gguf.inp +112 -0
  722. package/cpp/llama.cpp/models/ggml-vocab-pixtral.gguf.out +46 -0
  723. package/cpp/llama.cpp/models/ggml-vocab-qwen2.gguf +0 -0
  724. package/cpp/llama.cpp/models/ggml-vocab-qwen2.gguf.inp +112 -0
  725. package/cpp/llama.cpp/models/ggml-vocab-qwen2.gguf.out +46 -0
  726. package/cpp/llama.cpp/models/ggml-vocab-refact.gguf +0 -0
  727. package/cpp/llama.cpp/models/ggml-vocab-refact.gguf.inp +112 -0
  728. package/cpp/llama.cpp/models/ggml-vocab-refact.gguf.out +46 -0
  729. package/cpp/llama.cpp/models/ggml-vocab-roberta-bpe.gguf.inp +112 -0
  730. package/cpp/llama.cpp/models/ggml-vocab-roberta-bpe.gguf.out +46 -0
  731. package/cpp/llama.cpp/models/ggml-vocab-starcoder.gguf +0 -0
  732. package/cpp/llama.cpp/models/ggml-vocab-starcoder.gguf.inp +112 -0
  733. package/cpp/llama.cpp/models/ggml-vocab-starcoder.gguf.out +46 -0
  734. package/cpp/llama.cpp/models/templates/CohereForAI-c4ai-command-r-plus-tool_use.jinja +202 -0
  735. package/cpp/llama.cpp/models/templates/CohereForAI-c4ai-command-r7b-12-2024-tool_use.jinja +156 -0
  736. package/cpp/llama.cpp/models/templates/NousResearch-Hermes-2-Pro-Llama-3-8B-tool_use.jinja +152 -0
  737. package/cpp/llama.cpp/models/templates/NousResearch-Hermes-3-Llama-3.1-8B-tool_use.jinja +152 -0
  738. package/cpp/llama.cpp/models/templates/Qwen-Qwen2.5-7B-Instruct.jinja +54 -0
  739. package/cpp/llama.cpp/models/templates/README.md +22 -0
  740. package/cpp/llama.cpp/models/templates/deepseek-ai-DeepSeek-R1-Distill-Llama-8B.jinja +1 -0
  741. package/cpp/llama.cpp/models/templates/deepseek-ai-DeepSeek-R1-Distill-Qwen-32B.jinja +1 -0
  742. package/cpp/llama.cpp/models/templates/fireworks-ai-llama-3-firefunction-v2.jinja +57 -0
  743. package/cpp/llama.cpp/models/templates/google-gemma-2-2b-it.jinja +4 -0
  744. package/cpp/llama.cpp/models/templates/llama-cpp-deepseek-r1.jinja +76 -0
  745. package/cpp/llama.cpp/models/templates/meetkai-functionary-medium-v3.1.jinja +58 -0
  746. package/cpp/llama.cpp/models/templates/meetkai-functionary-medium-v3.2.jinja +287 -0
  747. package/cpp/llama.cpp/models/templates/meta-llama-Llama-3.1-8B-Instruct.jinja +109 -0
  748. package/cpp/llama.cpp/models/templates/meta-llama-Llama-3.2-3B-Instruct.jinja +93 -0
  749. package/cpp/llama.cpp/models/templates/meta-llama-Llama-3.3-70B-Instruct.jinja +109 -0
  750. package/cpp/llama.cpp/models/templates/microsoft-Phi-3.5-mini-instruct.jinja +8 -0
  751. package/cpp/llama.cpp/models/templates/mistralai-Mistral-Nemo-Instruct-2407.jinja +87 -0
  752. package/cpp/llama.cpp/mypy.ini +7 -0
  753. package/cpp/llama.cpp/pocs/CMakeLists.txt +14 -0
  754. package/cpp/llama.cpp/pocs/vdot/CMakeLists.txt +9 -0
  755. package/cpp/llama.cpp/pocs/vdot/q8dot.cpp +173 -0
  756. package/cpp/llama.cpp/pocs/vdot/vdot.cpp +311 -0
  757. package/cpp/llama.cpp/poetry.lock +1197 -0
  758. package/cpp/llama.cpp/prompts/LLM-questions.txt +49 -0
  759. package/cpp/llama.cpp/prompts/alpaca.txt +1 -0
  760. package/cpp/llama.cpp/prompts/assistant.txt +31 -0
  761. package/cpp/llama.cpp/prompts/chat-with-baichuan.txt +4 -0
  762. package/cpp/llama.cpp/prompts/chat-with-bob.txt +7 -0
  763. package/cpp/llama.cpp/prompts/chat-with-qwen.txt +1 -0
  764. package/cpp/llama.cpp/prompts/chat-with-vicuna-v0.txt +7 -0
  765. package/cpp/llama.cpp/prompts/chat-with-vicuna-v1.txt +7 -0
  766. package/cpp/llama.cpp/prompts/chat.txt +28 -0
  767. package/cpp/llama.cpp/prompts/dan-modified.txt +1 -0
  768. package/cpp/llama.cpp/prompts/dan.txt +1 -0
  769. package/cpp/llama.cpp/prompts/mnemonics.txt +93 -0
  770. package/cpp/llama.cpp/prompts/parallel-questions.txt +43 -0
  771. package/cpp/llama.cpp/prompts/reason-act.txt +18 -0
  772. package/cpp/llama.cpp/pyproject.toml +45 -0
  773. package/cpp/llama.cpp/pyrightconfig.json +22 -0
  774. package/cpp/llama.cpp/requirements/requirements-all.txt +15 -0
  775. package/cpp/llama.cpp/requirements/requirements-compare-llama-bench.txt +2 -0
  776. package/cpp/llama.cpp/requirements/requirements-convert_hf_to_gguf.txt +3 -0
  777. package/cpp/llama.cpp/requirements/requirements-convert_hf_to_gguf_update.txt +3 -0
  778. package/cpp/llama.cpp/requirements/requirements-convert_legacy_llama.txt +5 -0
  779. package/cpp/llama.cpp/requirements/requirements-convert_llama_ggml_to_gguf.txt +1 -0
  780. package/cpp/llama.cpp/requirements/requirements-convert_lora_to_gguf.txt +2 -0
  781. package/cpp/llama.cpp/requirements/requirements-gguf_editor_gui.txt +3 -0
  782. package/cpp/llama.cpp/requirements/requirements-pydantic.txt +3 -0
  783. package/cpp/llama.cpp/requirements/requirements-test-tokenizer-random.txt +1 -0
  784. package/cpp/llama.cpp/requirements/requirements-tool_bench.txt +12 -0
  785. package/cpp/llama.cpp/requirements.txt +13 -0
  786. package/cpp/llama.cpp/src/CMakeLists.txt +45 -0
  787. package/cpp/llama.cpp/src/llama-adapter.cpp +388 -0
  788. package/cpp/llama.cpp/src/llama-adapter.h +76 -0
  789. package/cpp/llama.cpp/src/llama-arch.cpp +1743 -0
  790. package/cpp/llama.cpp/src/llama-arch.h +437 -0
  791. package/cpp/llama.cpp/src/llama-batch.cpp +372 -0
  792. package/cpp/llama.cpp/src/llama-batch.h +89 -0
  793. package/cpp/llama.cpp/src/llama-chat.cpp +663 -0
  794. package/cpp/llama.cpp/src/llama-chat.h +58 -0
  795. package/cpp/llama.cpp/src/llama-context.cpp +2459 -0
  796. package/cpp/llama.cpp/src/llama-context.h +246 -0
  797. package/cpp/llama.cpp/src/llama-cparams.cpp +1 -0
  798. package/cpp/llama.cpp/src/llama-cparams.h +39 -0
  799. package/cpp/llama.cpp/src/llama-grammar.cpp +1219 -0
  800. package/cpp/llama.cpp/src/llama-grammar.h +173 -0
  801. package/cpp/llama.cpp/src/llama-graph.cpp +1713 -0
  802. package/cpp/llama.cpp/src/llama-graph.h +595 -0
  803. package/cpp/llama.cpp/src/llama-hparams.cpp +79 -0
  804. package/cpp/llama.cpp/src/llama-hparams.h +161 -0
  805. package/cpp/llama.cpp/src/llama-impl.cpp +167 -0
  806. package/cpp/llama.cpp/src/llama-impl.h +61 -0
  807. package/cpp/llama.cpp/src/llama-io.cpp +15 -0
  808. package/cpp/llama.cpp/src/llama-io.h +35 -0
  809. package/cpp/llama.cpp/src/llama-kv-cache.cpp +2486 -0
  810. package/cpp/llama.cpp/src/llama-kv-cache.h +405 -0
  811. package/cpp/llama.cpp/src/llama-memory.cpp +1 -0
  812. package/cpp/llama.cpp/src/llama-memory.h +31 -0
  813. package/cpp/llama.cpp/src/llama-mmap.cpp +600 -0
  814. package/cpp/llama.cpp/src/llama-mmap.h +68 -0
  815. package/cpp/llama.cpp/src/llama-model-loader.cpp +1133 -0
  816. package/cpp/llama.cpp/src/llama-model-loader.h +169 -0
  817. package/cpp/llama.cpp/src/llama-model.cpp +13453 -0
  818. package/cpp/llama.cpp/src/llama-model.h +420 -0
  819. package/cpp/llama.cpp/src/llama-quant.cpp +964 -0
  820. package/cpp/llama.cpp/src/llama-quant.h +1 -0
  821. package/cpp/llama.cpp/src/llama-sampling.cpp +2575 -0
  822. package/cpp/llama.cpp/src/llama-sampling.h +32 -0
  823. package/cpp/llama.cpp/src/llama-vocab.cpp +3313 -0
  824. package/cpp/llama.cpp/src/llama-vocab.h +125 -0
  825. package/cpp/llama.cpp/src/llama.cpp +340 -0
  826. package/cpp/llama.cpp/src/unicode-data.cpp +7034 -0
  827. package/cpp/llama.cpp/src/unicode-data.h +20 -0
  828. package/cpp/llama.cpp/src/unicode.cpp +849 -0
  829. package/cpp/llama.cpp/src/unicode.h +66 -0
  830. package/cpp/rn-completion.cpp +431 -0
  831. package/cpp/rn-llama.hpp +60 -0
  832. package/cpp/rn-utils.hpp +331 -0
  833. package/ios/OnLoad.mm +22 -0
  834. package/ios/generated/RNLlamaCppSpec/RNLlamaCppSpec-generated.mm +64 -0
  835. package/ios/generated/RNLlamaCppSpec/RNLlamaCppSpec.h +251 -0
  836. package/ios/generated/RNLlamaCppSpecJSI-generated.cpp +42 -0
  837. package/ios/generated/RNLlamaCppSpecJSI.h +336 -0
  838. package/ios/include/chat.h +135 -0
  839. package/ios/include/common/base64.hpp +392 -0
  840. package/ios/include/common/json.hpp +24766 -0
  841. package/ios/include/common/minja/chat-template.hpp +537 -0
  842. package/ios/include/common/minja/minja.hpp +2941 -0
  843. package/ios/include/common.h +668 -0
  844. package/ios/include/json-schema-to-grammar.h +21 -0
  845. package/ios/include/llama-cpp.h +30 -0
  846. package/ios/include/llama.h +1440 -0
  847. package/ios/include/log.h +103 -0
  848. package/ios/include/ngram-cache.h +101 -0
  849. package/ios/include/sampling.h +107 -0
  850. package/ios/include/speculative.h +28 -0
  851. package/ios/libs/llama.xcframework/Info.plist +135 -0
  852. package/ios/libs/llama.xcframework/ios-arm64/dSYMs/llama.dSYM/Contents/Info.plist +20 -0
  853. package/ios/libs/llama.xcframework/ios-arm64/dSYMs/llama.dSYM/Contents/Resources/DWARF/llama +0 -0
  854. package/ios/libs/llama.xcframework/ios-arm64/dSYMs/llama.dSYM/Contents/Resources/Relocations/aarch64/llama.yml +4492 -0
  855. package/ios/libs/llama.xcframework/ios-arm64/llama.framework/Headers/ggml-alloc.h +76 -0
  856. package/ios/libs/llama.xcframework/ios-arm64/llama.framework/Headers/ggml-backend.h +354 -0
  857. package/ios/libs/llama.xcframework/ios-arm64/llama.framework/Headers/ggml-blas.h +25 -0
  858. package/ios/libs/llama.xcframework/ios-arm64/llama.framework/Headers/ggml-cpu.h +143 -0
  859. package/ios/libs/llama.xcframework/ios-arm64/llama.framework/Headers/ggml-metal.h +66 -0
  860. package/ios/libs/llama.xcframework/ios-arm64/llama.framework/Headers/ggml.h +2192 -0
  861. package/ios/libs/llama.xcframework/ios-arm64/llama.framework/Headers/gguf.h +202 -0
  862. package/ios/libs/llama.xcframework/ios-arm64/llama.framework/Headers/llama.h +1440 -0
  863. package/ios/libs/llama.xcframework/ios-arm64/llama.framework/Info.plist +36 -0
  864. package/ios/libs/llama.xcframework/ios-arm64/llama.framework/Modules/module.modulemap +17 -0
  865. package/ios/libs/llama.xcframework/ios-arm64/llama.framework/llama +0 -0
  866. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Info.plist +20 -0
  867. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/DWARF/llama +0 -0
  868. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/Relocations/aarch64/llama.yml +4513 -0
  869. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/Relocations/x86_64/llama.yml +3440 -0
  870. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/llama.framework/Headers/ggml-alloc.h +76 -0
  871. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/llama.framework/Headers/ggml-backend.h +354 -0
  872. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/llama.framework/Headers/ggml-blas.h +25 -0
  873. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/llama.framework/Headers/ggml-cpu.h +143 -0
  874. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/llama.framework/Headers/ggml-metal.h +66 -0
  875. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/llama.framework/Headers/ggml.h +2192 -0
  876. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/llama.framework/Headers/gguf.h +202 -0
  877. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/llama.framework/Headers/llama.h +1440 -0
  878. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/llama.framework/Info.plist +36 -0
  879. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/llama.framework/Modules/module.modulemap +17 -0
  880. package/ios/libs/llama.xcframework/ios-arm64_x86_64-simulator/llama.framework/llama +0 -0
  881. package/ios/libs/llama.xcframework/macos-arm64_x86_64/dSYMs/llama.dSYM/Contents/Info.plist +20 -0
  882. package/ios/libs/llama.xcframework/macos-arm64_x86_64/dSYMs/llama.dSYM/Contents/Resources/DWARF/llama +0 -0
  883. package/ios/libs/llama.xcframework/macos-arm64_x86_64/dSYMs/llama.dSYM/Contents/Resources/Relocations/aarch64/llama.yml +4513 -0
  884. package/ios/libs/llama.xcframework/macos-arm64_x86_64/dSYMs/llama.dSYM/Contents/Resources/Relocations/x86_64/llama.yml +3442 -0
  885. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Headers/ggml-alloc.h +76 -0
  886. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Headers/ggml-backend.h +354 -0
  887. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Headers/ggml-blas.h +25 -0
  888. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Headers/ggml-cpu.h +143 -0
  889. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Headers/ggml-metal.h +66 -0
  890. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Headers/ggml.h +2192 -0
  891. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Headers/gguf.h +202 -0
  892. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Headers/llama.h +1440 -0
  893. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Modules/module.modulemap +17 -0
  894. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Resources/Info.plist +32 -0
  895. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/A/Headers/ggml-alloc.h +76 -0
  896. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/A/Headers/ggml-backend.h +354 -0
  897. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/A/Headers/ggml-blas.h +25 -0
  898. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/A/Headers/ggml-cpu.h +143 -0
  899. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/A/Headers/ggml-metal.h +66 -0
  900. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/A/Headers/ggml.h +2192 -0
  901. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/A/Headers/gguf.h +202 -0
  902. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/A/Headers/llama.h +1440 -0
  903. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/A/Modules/module.modulemap +17 -0
  904. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/A/Resources/Info.plist +32 -0
  905. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/A/llama +0 -0
  906. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/Current/Headers/ggml-alloc.h +76 -0
  907. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/Current/Headers/ggml-backend.h +354 -0
  908. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/Current/Headers/ggml-blas.h +25 -0
  909. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/Current/Headers/ggml-cpu.h +143 -0
  910. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/Current/Headers/ggml-metal.h +66 -0
  911. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/Current/Headers/ggml.h +2192 -0
  912. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/Current/Headers/gguf.h +202 -0
  913. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/Current/Headers/llama.h +1440 -0
  914. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/Current/Modules/module.modulemap +17 -0
  915. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/Current/Resources/Info.plist +32 -0
  916. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/Versions/Current/llama +0 -0
  917. package/ios/libs/llama.xcframework/macos-arm64_x86_64/llama.framework/llama +0 -0
  918. package/ios/libs/llama.xcframework/tvos-arm64/dSYMs/llama.dSYM/Contents/Info.plist +20 -0
  919. package/ios/libs/llama.xcframework/tvos-arm64/dSYMs/llama.dSYM/Contents/Resources/DWARF/llama +0 -0
  920. package/ios/libs/llama.xcframework/tvos-arm64/dSYMs/llama.dSYM/Contents/Resources/Relocations/aarch64/llama.yml +4492 -0
  921. package/ios/libs/llama.xcframework/tvos-arm64/llama.framework/Headers/ggml-alloc.h +76 -0
  922. package/ios/libs/llama.xcframework/tvos-arm64/llama.framework/Headers/ggml-backend.h +354 -0
  923. package/ios/libs/llama.xcframework/tvos-arm64/llama.framework/Headers/ggml-blas.h +25 -0
  924. package/ios/libs/llama.xcframework/tvos-arm64/llama.framework/Headers/ggml-cpu.h +143 -0
  925. package/ios/libs/llama.xcframework/tvos-arm64/llama.framework/Headers/ggml-metal.h +66 -0
  926. package/ios/libs/llama.xcframework/tvos-arm64/llama.framework/Headers/ggml.h +2192 -0
  927. package/ios/libs/llama.xcframework/tvos-arm64/llama.framework/Headers/gguf.h +202 -0
  928. package/ios/libs/llama.xcframework/tvos-arm64/llama.framework/Headers/llama.h +1440 -0
  929. package/ios/libs/llama.xcframework/tvos-arm64/llama.framework/Info.plist +35 -0
  930. package/ios/libs/llama.xcframework/tvos-arm64/llama.framework/Modules/module.modulemap +17 -0
  931. package/ios/libs/llama.xcframework/tvos-arm64/llama.framework/llama +0 -0
  932. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Info.plist +20 -0
  933. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/DWARF/llama +0 -0
  934. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/Relocations/aarch64/llama.yml +4513 -0
  935. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/Relocations/x86_64/llama.yml +3440 -0
  936. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/llama.framework/Headers/ggml-alloc.h +76 -0
  937. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/llama.framework/Headers/ggml-backend.h +354 -0
  938. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/llama.framework/Headers/ggml-blas.h +25 -0
  939. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/llama.framework/Headers/ggml-cpu.h +143 -0
  940. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/llama.framework/Headers/ggml-metal.h +66 -0
  941. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/llama.framework/Headers/ggml.h +2192 -0
  942. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/llama.framework/Headers/gguf.h +202 -0
  943. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/llama.framework/Headers/llama.h +1440 -0
  944. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/llama.framework/Info.plist +35 -0
  945. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/llama.framework/Modules/module.modulemap +17 -0
  946. package/ios/libs/llama.xcframework/tvos-arm64_x86_64-simulator/llama.framework/llama +0 -0
  947. package/ios/libs/llama.xcframework/xros-arm64/dSYMs/llama.dSYM/Contents/Info.plist +20 -0
  948. package/ios/libs/llama.xcframework/xros-arm64/dSYMs/llama.dSYM/Contents/Resources/DWARF/llama +0 -0
  949. package/ios/libs/llama.xcframework/xros-arm64/dSYMs/llama.dSYM/Contents/Resources/Relocations/aarch64/llama.yml +4528 -0
  950. package/ios/libs/llama.xcframework/xros-arm64/llama.framework/Headers/ggml-alloc.h +76 -0
  951. package/ios/libs/llama.xcframework/xros-arm64/llama.framework/Headers/ggml-backend.h +354 -0
  952. package/ios/libs/llama.xcframework/xros-arm64/llama.framework/Headers/ggml-blas.h +25 -0
  953. package/ios/libs/llama.xcframework/xros-arm64/llama.framework/Headers/ggml-cpu.h +143 -0
  954. package/ios/libs/llama.xcframework/xros-arm64/llama.framework/Headers/ggml-metal.h +66 -0
  955. package/ios/libs/llama.xcframework/xros-arm64/llama.framework/Headers/ggml.h +2192 -0
  956. package/ios/libs/llama.xcframework/xros-arm64/llama.framework/Headers/gguf.h +202 -0
  957. package/ios/libs/llama.xcframework/xros-arm64/llama.framework/Headers/llama.h +1440 -0
  958. package/ios/libs/llama.xcframework/xros-arm64/llama.framework/Info.plist +32 -0
  959. package/ios/libs/llama.xcframework/xros-arm64/llama.framework/Modules/module.modulemap +17 -0
  960. package/ios/libs/llama.xcframework/xros-arm64/llama.framework/llama +0 -0
  961. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Info.plist +20 -0
  962. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/DWARF/llama +0 -0
  963. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/Relocations/aarch64/llama.yml +4549 -0
  964. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/dSYMs/llama.dSYM/Contents/Resources/Relocations/x86_64/llama.yml +3470 -0
  965. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/llama.framework/Headers/ggml-alloc.h +76 -0
  966. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/llama.framework/Headers/ggml-backend.h +354 -0
  967. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/llama.framework/Headers/ggml-blas.h +25 -0
  968. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/llama.framework/Headers/ggml-cpu.h +143 -0
  969. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/llama.framework/Headers/ggml-metal.h +66 -0
  970. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/llama.framework/Headers/ggml.h +2192 -0
  971. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/llama.framework/Headers/gguf.h +202 -0
  972. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/llama.framework/Headers/llama.h +1440 -0
  973. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/llama.framework/Info.plist +32 -0
  974. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/llama.framework/Modules/module.modulemap +17 -0
  975. package/ios/libs/llama.xcframework/xros-arm64_x86_64-simulator/llama.framework/llama +0 -0
  976. package/lib/module/NativeRNLlamaCpp.js +35 -0
  977. package/lib/module/NativeRNLlamaCpp.js.map +1 -0
  978. package/lib/module/index.js +20 -0
  979. package/lib/module/index.js.map +1 -0
  980. package/lib/module/package.json +1 -0
  981. package/lib/typescript/package.json +1 -0
  982. package/lib/typescript/src/NativeRNLlamaCpp.d.ts +222 -0
  983. package/lib/typescript/src/NativeRNLlamaCpp.d.ts.map +1 -0
  984. package/lib/typescript/src/index.d.ts +5 -0
  985. package/lib/typescript/src/index.d.ts.map +1 -0
  986. package/package.json +161 -0
  987. package/react-native.config.js +15 -0
  988. package/src/NativeRNLlamaCpp.ts +282 -0
  989. package/src/index.tsx +54 -0
@@ -0,0 +1,1559 @@
1
+ #include "common.hpp"
2
+ #include "ggml.h"
3
+ #include "element_wise.hpp"
4
+
5
+ static void acc_f32(const float * x, const float * y, float * dst, const int ne,
6
+ const int ne10, const int ne11, const int ne12,
7
+ const int nb1, const int nb2, int offset, const sycl::nd_item<3> &item_ct1) {
8
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
9
+ item_ct1.get_local_id(2);
10
+ if (i >= ne) {
11
+ return;
12
+ }
13
+ int src1_idx = i - offset;
14
+ int oz = src1_idx / nb2;
15
+ int oy = (src1_idx - (oz * nb2)) / nb1;
16
+ int ox = src1_idx % nb1;
17
+ if (src1_idx >= 0 && ox < ne10 && oy < ne11 && oz < ne12) {
18
+ dst[i] = x[i] + y[ox + oy * ne10 + oz * ne10 * ne11];
19
+ } else {
20
+ dst[i] = x[i];
21
+ }
22
+ }
23
+
24
+ template<typename T>
25
+ static void sgn(const T * x, T * dst, const int k, const sycl::nd_item<3> &item_ct1) {
26
+ for(auto i = item_ct1.get_global_id(2); i < (const size_t)k; i += item_ct1.get_global_range(2)) {
27
+ dst[i] = x[i] > static_cast<T>(0.f) ? static_cast<T>(1.f) : ((x[i] < static_cast<T>(0.f) ? static_cast<T>(-1.f) : static_cast<T>(0.f)));
28
+ }
29
+ }
30
+
31
+ template<typename T>
32
+ static void abs_op(const T * x, T * dst, const int k, const sycl::nd_item<3> &item_ct1) {
33
+ for(auto i = item_ct1.get_global_id(2); i < (const size_t)k; i += item_ct1.get_global_range(2)) {
34
+ dst[i] = sycl::fabs(x[i]);
35
+ }
36
+ }
37
+
38
+ template<typename T>
39
+ static void elu_op(const T * x, T * dst, const int k, const sycl::nd_item<3> &item_ct1) {
40
+ for(auto i = item_ct1.get_global_id(2); i < (const size_t)k; i += item_ct1.get_global_range(2)) {
41
+ dst[i] = (x[i] > static_cast<T>(0.f)) ? x[i] : sycl::expm1(x[i]);
42
+ }
43
+ }
44
+
45
+ template<typename T>
46
+ static void gelu(const T * x, T * dst, const int k,
47
+ const sycl::nd_item<3> &item_ct1) {
48
+ const T GELU_COEF_A = static_cast<T>(0.044715f);
49
+ const T SQRT_2_OVER_PI = static_cast<T>(0.79788456080286535587989211986876f);
50
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
51
+ item_ct1.get_local_id(2);
52
+
53
+ if (i >= k) {
54
+ return;
55
+ }
56
+
57
+ float xi = x[i];
58
+ dst[i] = static_cast<T>(0.5f) * xi *
59
+ (static_cast<T>(1.0f) +
60
+ sycl::tanh(SQRT_2_OVER_PI * xi * (static_cast<T>(1.0f) + GELU_COEF_A * xi * xi)));
61
+ }
62
+
63
+ template<typename T>
64
+ static void silu(const T * x, T * dst, const int k,
65
+ const sycl::nd_item<3> &item_ct1) {
66
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
67
+ item_ct1.get_local_id(2);
68
+
69
+ if (i >= k) {
70
+ return;
71
+ }
72
+ dst[i] = x[i] / (static_cast<T>(1.0f) + sycl::native::exp(-x[i]));
73
+ }
74
+
75
+ template<typename T>
76
+ static void gelu_quick(const T *x, T *dst, int k,
77
+ const sycl::nd_item<3> &item_ct1) {
78
+ const float GELU_QUICK_COEF = -1.702f;
79
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
80
+ item_ct1.get_local_id(2);
81
+ if (i >= k) {
82
+ return;
83
+ }
84
+ dst[i] = x[i] * (static_cast<T>(1.0f) / (static_cast<T>(1.0f) + sycl::native::exp(GELU_QUICK_COEF * x[i])));
85
+ }
86
+
87
+ template<typename T>
88
+ static void tanh(const T *x, T *dst, int k,
89
+ const sycl::nd_item<3> &item_ct1) {
90
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
91
+ item_ct1.get_local_id(2);
92
+ if (i >= k) {
93
+ return;
94
+ }
95
+ dst[i] = sycl::tanh((x[i]));
96
+ }
97
+
98
+ template<typename T>
99
+ static void relu(const T * x, T * dst, const int k,
100
+ const sycl::nd_item<3> &item_ct1) {
101
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
102
+ item_ct1.get_local_id(2);
103
+
104
+ if (i >= k) {
105
+ return;
106
+ }
107
+ dst[i] = sycl::fmax((x[i]), static_cast<T>(0));
108
+ }
109
+
110
+ template<typename T>
111
+ static void sigmoid(const T * x, T * dst, const int k,
112
+ const sycl::nd_item<3> &item_ct1) {
113
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
114
+ item_ct1.get_local_id(2);
115
+
116
+ if (i >= k) {
117
+ return;
118
+ }
119
+ dst[i] = 1.0f / (static_cast<T>(1.0f) + sycl::native::exp(-x[i]));
120
+ }
121
+
122
+ template<typename T>
123
+ static void sqrt(const T * x, T * dst, const int k,
124
+ const sycl::nd_item<3> &item_ct1) {
125
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
126
+ item_ct1.get_local_id(2);
127
+
128
+ if (i >= k) {
129
+ return;
130
+ }
131
+ dst[i] = sycl::sqrt(x[i]);
132
+ }
133
+
134
+ template<typename T>
135
+ static void sin(const T * x, T * dst, const int k,
136
+ const sycl::nd_item<3> &item_ct1) {
137
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
138
+ item_ct1.get_local_id(2);
139
+
140
+ if (i >= k) {
141
+ return;
142
+ }
143
+ dst[i] = sycl::sin(x[i]);
144
+ }
145
+
146
+ template<typename T>
147
+ static void cos(const T * x, T * dst, const int k,
148
+ const sycl::nd_item<3> &item_ct1) {
149
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
150
+ item_ct1.get_local_id(2);
151
+
152
+ if (i >= k) {
153
+ return;
154
+ }
155
+ dst[i] = sycl::cos(x[i]);
156
+ }
157
+
158
+ template<typename T>
159
+ static void hardsigmoid(const T * x, T * dst, const int k,
160
+ const sycl::nd_item<3> &item_ct1) {
161
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
162
+ item_ct1.get_local_id(2);
163
+
164
+ if (i >= k) {
165
+ return;
166
+ }
167
+ dst[i] = sycl::fmin(static_cast<T>(1.0f), sycl::fmax(static_cast<T>(0.0f), (x[i] + static_cast<T>(3.0f)) / static_cast<T>(6.0f)));
168
+ }
169
+
170
+ template<typename T>
171
+ static void hardswish(const T * x, T * dst, const int k,
172
+ const sycl::nd_item<3> &item_ct1) {
173
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
174
+ item_ct1.get_local_id(2);
175
+
176
+ if (i >= k) {
177
+ return;
178
+ }
179
+ dst[i] = x[i] * sycl::fmin(static_cast<T>(1.0f), sycl::fmax(static_cast<T>(0.0f), (x[i] + static_cast<T>(3.0f)) / static_cast<T>(6.0f)));
180
+ }
181
+
182
+ template<typename T>
183
+ static void exp(const T * x, T * dst, const int k,
184
+ const sycl::nd_item<3> &item_ct1) {
185
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
186
+ item_ct1.get_local_id(2);
187
+
188
+ if (i >= k) {
189
+ return;
190
+ }
191
+ dst[i] = sycl::exp(x[i]);
192
+ }
193
+
194
+ template<typename T>
195
+ static void log(const T * x, T * dst, const int k,
196
+ const sycl::nd_item<3> &item_ct1) {
197
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
198
+ item_ct1.get_local_id(2);
199
+
200
+ if (i >= k) {
201
+ return;
202
+ }
203
+ T xi = x[i];
204
+ if (xi <= 0) {
205
+ dst[i] = neg_infinity<T>();
206
+ } else {
207
+ dst[i] = sycl::log(xi);
208
+ }
209
+ }
210
+
211
+ template<typename T>
212
+ static void neg(const T * x, T * dst, const int k,
213
+ const sycl::nd_item<3> &item_ct1) {
214
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
215
+ item_ct1.get_local_id(2);
216
+
217
+ if (i >= k) {
218
+ return;
219
+ }
220
+ dst[i] = -x[i];
221
+ }
222
+
223
+ template<typename T>
224
+ static void step(const T * x, T * dst, const int k,
225
+ const sycl::nd_item<3> &item_ct1) {
226
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
227
+ item_ct1.get_local_id(2);
228
+
229
+ if (i >= k) {
230
+ return;
231
+ }
232
+ dst[i] = x[i] > static_cast<T>(0.0f);
233
+ }
234
+
235
+ template<typename T>
236
+ static void leaky_relu(const T *x, T *dst, const int k, const float negative_slope,
237
+ const sycl::nd_item<3> &item_ct1) {
238
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
239
+ item_ct1.get_local_id(2);
240
+ if (i >= k) {
241
+ return;
242
+ }
243
+ dst[i] = sycl::fmax((x[i]), static_cast<T>(0)) +
244
+ sycl::fmin((x[i]), static_cast<T>(0.0f)) * negative_slope;
245
+ }
246
+
247
+ template<typename T>
248
+ static void sqr(const T * x, T * dst, const int k,
249
+ const sycl::nd_item<3> &item_ct1) {
250
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
251
+ item_ct1.get_local_id(2);
252
+
253
+ if (i >= k) {
254
+ return;
255
+ }
256
+ dst[i] = x[i] * x[i];
257
+ }
258
+
259
+ template<typename T>
260
+ static void upscale(const T *x, T *dst, const int nb00, const int nb01,
261
+ const int nb02, const int nb03, const int ne10, const int ne11,
262
+ const int ne12, const int ne13, const float sf0, const float sf1,
263
+ const float sf2, const float sf3, const sycl::nd_item<1> &item_ct1) {
264
+ int index = item_ct1.get_local_id(0) +
265
+ item_ct1.get_group(0) * item_ct1.get_local_range(0);
266
+ if (index >= ne10 * ne11 * ne12 * ne13) {
267
+ return;
268
+ }
269
+ // operation
270
+ int i10 = index % ne10;
271
+ int i11 = (index / ne10) % ne11;
272
+ int i12 = (index / (ne10 * ne11)) % ne12;
273
+ int i13 = (index / (ne10 * ne11 * ne12)) % ne13;
274
+
275
+ int i00 = i10 / sf0;
276
+ int i01 = i11 / sf1;
277
+ int i02 = i12 / sf2;
278
+ int i03 = i13 / sf3;
279
+
280
+ dst[index] = *(const T *)((const char *)x + i03 * nb03 + i02 * nb02 + i01 * nb01 + i00 * nb00);
281
+ }
282
+
283
+ template <typename T>
284
+ static void pad(const T *x, T *dst, const int ne0, const int ne00, const int ne01, const int ne02,
285
+ const sycl::nd_item<3> &item_ct1) {
286
+ int nidx = item_ct1.get_local_id(2) +
287
+ item_ct1.get_group(2) * item_ct1.get_local_range(2);
288
+ if (nidx >= ne0) {
289
+ return;
290
+ }
291
+
292
+ // operation
293
+ int offset_dst = nidx + item_ct1.get_group(1) * ne0 +
294
+ item_ct1.get_group(0) * ne0 * item_ct1.get_group_range(1);
295
+ if (nidx < ne00 && item_ct1.get_group(1) < (size_t) ne01 && item_ct1.get_group(0) < (size_t) ne02) {
296
+ int offset_src = nidx + item_ct1.get_group(1) * ne00 +
297
+ item_ct1.get_group(0) * ne00 * ne01;
298
+ dst[offset_dst] = x[offset_src];
299
+ } else {
300
+ dst[offset_dst] = static_cast<T>(0.0f);
301
+ }
302
+ }
303
+
304
+
305
+ template<typename T>
306
+ static void clamp(const T * x, T * dst, const float min, const float max, const int k,
307
+ const sycl::nd_item<3> &item_ct1) {
308
+ const int i = item_ct1.get_local_range(2) * item_ct1.get_group(2) +
309
+ item_ct1.get_local_id(2);
310
+
311
+ if (i >= k) {
312
+ return;
313
+ }
314
+
315
+ dst[i] = x[i] < static_cast<T>(min) ? static_cast<T>(min) : (x[i] > static_cast<T>(max) ? static_cast<T>(max) : x[i]);
316
+ }
317
+
318
+ static void acc_f32_sycl(const float *x, const float *y, float *dst,
319
+ const int n_elements, const int ne10, const int ne11,
320
+ const int ne12, const int nb1, const int nb2,
321
+ const int offset, queue_ptr stream) {
322
+ int num_blocks = (n_elements + SYCL_ACC_BLOCK_SIZE - 1) / SYCL_ACC_BLOCK_SIZE;
323
+ stream->parallel_for(
324
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
325
+ sycl::range<3>(1, 1, SYCL_ACC_BLOCK_SIZE),
326
+ sycl::range<3>(1, 1, SYCL_ACC_BLOCK_SIZE)),
327
+ [=](sycl::nd_item<3> item_ct1) {
328
+ acc_f32(x, y, dst, n_elements, ne10, ne11, ne12, nb1, nb2, offset,
329
+ item_ct1);
330
+ });
331
+ }
332
+
333
+ template<typename T>
334
+ static void gelu_sycl(const T *x, T *dst, const int k,
335
+ queue_ptr stream) {
336
+ const int num_blocks = (k + SYCL_GELU_BLOCK_SIZE - 1) / SYCL_GELU_BLOCK_SIZE;
337
+ stream->parallel_for(
338
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
339
+ sycl::range<3>(1, 1, SYCL_GELU_BLOCK_SIZE),
340
+ sycl::range<3>(1, 1, SYCL_GELU_BLOCK_SIZE)),
341
+ [=](sycl::nd_item<3> item_ct1) {
342
+ gelu(x, dst, k, item_ct1);
343
+ });
344
+ }
345
+
346
+ template<typename T>
347
+ static void silu_sycl(const T *x, T *dst, const int k,
348
+ queue_ptr stream) {
349
+ const int num_blocks = (k + SYCL_SILU_BLOCK_SIZE - 1) / SYCL_SILU_BLOCK_SIZE;
350
+ stream->parallel_for(
351
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
352
+ sycl::range<3>(1, 1, SYCL_SILU_BLOCK_SIZE),
353
+ sycl::range<3>(1, 1, SYCL_SILU_BLOCK_SIZE)),
354
+ [=](sycl::nd_item<3> item_ct1) {
355
+ silu(x, dst, k, item_ct1);
356
+ });
357
+ }
358
+
359
+ template<typename T>
360
+ static void sgn_sycl(const T * x, T * dst, const int k, queue_ptr stream) {
361
+ // hard code for now
362
+ const int num_blocks = ceil_div(k, 256);
363
+ stream->parallel_for(
364
+ sycl::nd_range<3>((sycl::range<3>(1, 1, num_blocks) * sycl::range(1, 1, 256)), sycl::range(1, 1, 256)), [=](sycl::nd_item<3> item_ct1) {
365
+ sgn(x, dst, k, item_ct1);
366
+ });
367
+ }
368
+
369
+ template<typename T>
370
+ static void abs_sycl(const T * x, T * dst, const int k, queue_ptr stream) {
371
+ // hard code for now
372
+ const int num_blocks = ceil_div(k, 256);
373
+ stream->parallel_for(
374
+ sycl::nd_range<3>((sycl::range<3>(1, 1, num_blocks) * sycl::range<3>(1, 1, 256)), sycl::range<3>(1, 1, 256)), [=](sycl::nd_item<3> item_ct1) {
375
+ abs_op(x, dst, k, item_ct1);
376
+ });
377
+ }
378
+
379
+
380
+ template<typename T>
381
+ static void elu_sycl(const T * x, T * dst, const int k, queue_ptr stream) {
382
+ // hard code for now
383
+ const int num_blocks = ceil_div(k, 256);
384
+ stream->parallel_for(
385
+ sycl::nd_range<3>((sycl::range<3>(1, 1, num_blocks) * sycl::range<3>(1, 1, 256)), sycl::range<3>(1, 1, 256)), [=](sycl::nd_item<3> item_ct1) {
386
+ elu_op(x, dst, k, item_ct1);
387
+ });
388
+ }
389
+
390
+ template<typename T>
391
+ static void gelu_quick_sycl(const T *x, T *dst, const int k,
392
+ queue_ptr stream) {
393
+ const int num_blocks = (k + SYCL_GELU_BLOCK_SIZE - 1) / SYCL_GELU_BLOCK_SIZE;
394
+ stream->parallel_for(
395
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
396
+ sycl::range<3>(1, 1, SYCL_GELU_BLOCK_SIZE),
397
+ sycl::range<3>(1, 1, SYCL_GELU_BLOCK_SIZE)),
398
+ [=](sycl::nd_item<3> item_ct1) {
399
+ gelu_quick(x, dst, k, item_ct1);
400
+ });
401
+ }
402
+
403
+ template<typename T>
404
+ static void tanh_sycl(const T *x, T *dst, const int k,
405
+ queue_ptr stream) {
406
+ const int num_blocks = (k + SYCL_TANH_BLOCK_SIZE - 1) / SYCL_TANH_BLOCK_SIZE;
407
+ stream->parallel_for(
408
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
409
+ sycl::range<3>(1, 1, SYCL_TANH_BLOCK_SIZE),
410
+ sycl::range<3>(1, 1, SYCL_TANH_BLOCK_SIZE)),
411
+ [=](sycl::nd_item<3> item_ct1) {
412
+ tanh(x, dst, k, item_ct1);
413
+ });
414
+ }
415
+
416
+ template<typename T>
417
+ static void relu_sycl(const T *x, T *dst, const int k,
418
+ queue_ptr stream) {
419
+ const int num_blocks = (k + SYCL_RELU_BLOCK_SIZE - 1) / SYCL_RELU_BLOCK_SIZE;
420
+ stream->parallel_for(
421
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
422
+ sycl::range<3>(1, 1, SYCL_RELU_BLOCK_SIZE),
423
+ sycl::range<3>(1, 1, SYCL_RELU_BLOCK_SIZE)),
424
+ [=](sycl::nd_item<3> item_ct1) {
425
+ relu(x, dst, k, item_ct1);
426
+ });
427
+ }
428
+
429
+ template<typename T>
430
+ static void hardsigmoid_sycl(const T *x, T *dst, const int k,
431
+ queue_ptr stream) {
432
+ const int num_blocks = (k + SYCL_HARDSIGMOID_BLOCK_SIZE - 1) / SYCL_HARDSIGMOID_BLOCK_SIZE;
433
+ stream->parallel_for(
434
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
435
+ sycl::range<3>(1, 1, SYCL_HARDSIGMOID_BLOCK_SIZE),
436
+ sycl::range<3>(1, 1, SYCL_HARDSIGMOID_BLOCK_SIZE)),
437
+ [=](sycl::nd_item<3> item_ct1) {
438
+ hardsigmoid(x, dst, k, item_ct1);
439
+ });
440
+ }
441
+
442
+ template<typename T>
443
+ static void hardswish_sycl(const T *x, T *dst, const int k,
444
+ queue_ptr stream) {
445
+ const int num_blocks = (k + SYCL_HARDSWISH_BLOCK_SIZE - 1) / SYCL_HARDSWISH_BLOCK_SIZE;
446
+ stream->parallel_for(
447
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
448
+ sycl::range<3>(1, 1, SYCL_HARDSWISH_BLOCK_SIZE),
449
+ sycl::range<3>(1, 1, SYCL_HARDSWISH_BLOCK_SIZE)),
450
+ [=](sycl::nd_item<3> item_ct1) {
451
+ hardswish(x, dst, k, item_ct1);
452
+ });
453
+ }
454
+
455
+ template<typename T>
456
+ static void exp_sycl(const T *x, T *dst, const int k,
457
+ queue_ptr stream) {
458
+ const int num_blocks = (k + SYCL_EXP_BLOCK_SIZE - 1) / SYCL_EXP_BLOCK_SIZE;
459
+ stream->parallel_for(
460
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
461
+ sycl::range<3>(1, 1, SYCL_EXP_BLOCK_SIZE),
462
+ sycl::range<3>(1, 1, SYCL_EXP_BLOCK_SIZE)),
463
+ [=](sycl::nd_item<3> item_ct1) {
464
+ exp(x, dst, k, item_ct1);
465
+ });
466
+ }
467
+
468
+ template<typename T>
469
+ static void log_sycl(const T *x, T *dst, const int k,
470
+ queue_ptr stream) {
471
+ const int num_blocks = (k + SYCL_EXP_BLOCK_SIZE - 1) / SYCL_EXP_BLOCK_SIZE;
472
+ stream->parallel_for(
473
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
474
+ sycl::range<3>(1, 1, SYCL_EXP_BLOCK_SIZE),
475
+ sycl::range<3>(1, 1, SYCL_EXP_BLOCK_SIZE)),
476
+ [=](sycl::nd_item<3> item_ct1) {
477
+ log(x, dst, k, item_ct1);
478
+ });
479
+ }
480
+
481
+ template<typename T>
482
+ static void neg_sycl(const T *x, T *dst, const int k,
483
+ queue_ptr stream) {
484
+ const int num_blocks = (k + SYCL_NEG_BLOCK_SIZE - 1) / SYCL_NEG_BLOCK_SIZE;
485
+ stream->parallel_for(
486
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
487
+ sycl::range<3>(1, 1, SYCL_NEG_BLOCK_SIZE),
488
+ sycl::range<3>(1, 1, SYCL_NEG_BLOCK_SIZE)),
489
+ [=](sycl::nd_item<3> item_ct1) {
490
+ neg(x, dst, k, item_ct1);
491
+ });
492
+ }
493
+
494
+ template<typename T>
495
+ static void step_sycl(const T *x, T *dst, const int k,
496
+ queue_ptr stream) {
497
+ const int num_blocks = (k + SYCL_NEG_BLOCK_SIZE - 1) / SYCL_NEG_BLOCK_SIZE;
498
+ stream->parallel_for(
499
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
500
+ sycl::range<3>(1, 1, SYCL_NEG_BLOCK_SIZE),
501
+ sycl::range<3>(1, 1, SYCL_NEG_BLOCK_SIZE)),
502
+ [=](sycl::nd_item<3> item_ct1) {
503
+ step(x, dst, k, item_ct1);
504
+ });
505
+ }
506
+
507
+ template<typename T>
508
+ static void sigmoid_sycl(const T *x, T *dst, const int k,
509
+ queue_ptr stream) {
510
+ const int num_blocks = (k + SYCL_SIGMOID_BLOCK_SIZE - 1) / SYCL_SIGMOID_BLOCK_SIZE;
511
+ stream->parallel_for(
512
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
513
+ sycl::range<3>(1, 1, SYCL_SIGMOID_BLOCK_SIZE),
514
+ sycl::range<3>(1, 1, SYCL_SIGMOID_BLOCK_SIZE)),
515
+ [=](sycl::nd_item<3> item_ct1) {
516
+ sigmoid(x, dst, k, item_ct1);
517
+ });
518
+ }
519
+
520
+ template<typename T>
521
+ static void sqrt_sycl(const T *x, T *dst, const int k,
522
+ queue_ptr stream) {
523
+ const int num_blocks = (k + SYCL_SQRT_BLOCK_SIZE - 1) / SYCL_SQRT_BLOCK_SIZE;
524
+ stream->parallel_for(
525
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
526
+ sycl::range<3>(1, 1, SYCL_SQRT_BLOCK_SIZE),
527
+ sycl::range<3>(1, 1, SYCL_SQRT_BLOCK_SIZE)),
528
+ [=](sycl::nd_item<3> item_ct1) {
529
+ sqrt(x, dst, k, item_ct1);
530
+ });
531
+ }
532
+
533
+ template<typename T>
534
+ static void sin_sycl(const T *x, T *dst, const int k,
535
+ queue_ptr stream) {
536
+ const int num_blocks = (k + SYCL_SIN_BLOCK_SIZE - 1) / SYCL_SIN_BLOCK_SIZE;
537
+ stream->parallel_for(
538
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
539
+ sycl::range<3>(1, 1, SYCL_SIN_BLOCK_SIZE),
540
+ sycl::range<3>(1, 1, SYCL_SIN_BLOCK_SIZE)),
541
+ [=](sycl::nd_item<3> item_ct1) {
542
+ sin(x, dst, k, item_ct1);
543
+ });
544
+ }
545
+
546
+ template<typename T>
547
+ static void cos_sycl(const T *x, T *dst, const int k,
548
+ queue_ptr stream) {
549
+ const int num_blocks = (k + SYCL_SIN_BLOCK_SIZE - 1) / SYCL_SIN_BLOCK_SIZE;
550
+ stream->parallel_for(
551
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
552
+ sycl::range<3>(1, 1, SYCL_SIN_BLOCK_SIZE),
553
+ sycl::range<3>(1, 1, SYCL_SIN_BLOCK_SIZE)),
554
+ [=](sycl::nd_item<3> item_ct1) {
555
+ cos(x, dst, k, item_ct1);
556
+ });
557
+ }
558
+
559
+ template<typename T>
560
+ static void leaky_relu_sycl(const T *x, T *dst, const int k,
561
+ const float negative_slope,
562
+ queue_ptr stream) {
563
+ const int num_blocks = (k + SYCL_RELU_BLOCK_SIZE - 1) / SYCL_RELU_BLOCK_SIZE;
564
+ stream->parallel_for(
565
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
566
+ sycl::range<3>(1, 1, SYCL_RELU_BLOCK_SIZE),
567
+ sycl::range<3>(1, 1, SYCL_RELU_BLOCK_SIZE)),
568
+ [=](sycl::nd_item<3> item_ct1) {
569
+ leaky_relu(x, dst, k, negative_slope, item_ct1);
570
+ });
571
+ }
572
+
573
+ template<typename T>
574
+ static void sqr_sycl(const T *x, T *dst, const int k,
575
+ queue_ptr stream) {
576
+ const int num_blocks = (k + SYCL_SQR_BLOCK_SIZE - 1) / SYCL_SQR_BLOCK_SIZE;
577
+ stream->parallel_for(
578
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
579
+ sycl::range<3>(1, 1, SYCL_SQR_BLOCK_SIZE),
580
+ sycl::range<3>(1, 1, SYCL_SQR_BLOCK_SIZE)),
581
+ [=](sycl::nd_item<3> item_ct1) {
582
+ sqr(x, dst, k, item_ct1);
583
+ });
584
+ }
585
+
586
+ template<typename T>
587
+ static void upscale_sycl(const T *x, T *dst, const int nb00, const int nb01,
588
+ const int nb02, const int nb03, const int ne10, const int ne11,
589
+ const int ne12, const int ne13, const float sf0, const float sf1,
590
+ const float sf2, const float sf3, queue_ptr stream) {
591
+ int dst_size = ne10 * ne11 * ne12 * ne13;
592
+ int num_blocks = (dst_size + SYCL_UPSCALE_BLOCK_SIZE - 1) / SYCL_UPSCALE_BLOCK_SIZE;
593
+ sycl::range<1> gridDim(num_blocks * SYCL_UPSCALE_BLOCK_SIZE);
594
+ stream->parallel_for(
595
+ sycl::nd_range<1>(gridDim, sycl::range<1>(SYCL_UPSCALE_BLOCK_SIZE)),
596
+ [=](sycl::nd_item<1> item_ct1) {
597
+ upscale(x, dst, nb00, nb01, nb02, nb03, ne10, ne11, ne12, ne13, sf0, sf1, sf2, sf3, item_ct1);
598
+ });
599
+ }
600
+
601
+ template<typename T>
602
+ static void pad_sycl(const T *x, T *dst, const int ne00,
603
+ const int ne01, const int ne02, const int ne0,
604
+ const int ne1, const int ne2, queue_ptr stream) {
605
+ int num_blocks = (ne0 + SYCL_PAD_BLOCK_SIZE - 1) / SYCL_PAD_BLOCK_SIZE;
606
+ sycl::range<3> gridDim(ne2, ne1, num_blocks);
607
+ stream->parallel_for(
608
+ sycl::nd_range<3>(gridDim * sycl::range<3>(1, 1, SYCL_PAD_BLOCK_SIZE),
609
+ sycl::range<3>(1, 1, SYCL_PAD_BLOCK_SIZE)),
610
+ [=](sycl::nd_item<3> item_ct1) {
611
+ pad(x, dst, ne0, ne00, ne01, ne02, item_ct1);
612
+ });
613
+ }
614
+
615
+ template<typename T>
616
+ static void clamp_sycl(const T *x, T *dst, const float min,
617
+ const float max, const int k,
618
+ queue_ptr stream) {
619
+ const int num_blocks = (k + SYCL_CLAMP_BLOCK_SIZE - 1) / SYCL_CLAMP_BLOCK_SIZE;
620
+ stream->parallel_for(
621
+ sycl::nd_range<3>(sycl::range<3>(1, 1, num_blocks) *
622
+ sycl::range<3>(1, 1, SYCL_CLAMP_BLOCK_SIZE),
623
+ sycl::range<3>(1, 1, SYCL_CLAMP_BLOCK_SIZE)),
624
+ [=](sycl::nd_item<3> item_ct1) {
625
+ clamp(x, dst, min, max, k, item_ct1);
626
+ });
627
+ }
628
+
629
+ inline void ggml_sycl_op_sgn(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
630
+ #if defined (GGML_SYCL_F16)
631
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
632
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
633
+
634
+ #else
635
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
636
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
637
+ #endif
638
+ GGML_ASSERT(dst->src[0]->type == dst->type);
639
+ dpct::queue_ptr main_stream = ctx.stream();
640
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
641
+ switch (dst->type) {
642
+ #if defined (GGML_SYCL_F16)
643
+ case GGML_TYPE_F16:
644
+ {
645
+ auto data_pts = cast_data<sycl::half>(dst);
646
+ sgn_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
647
+ break;
648
+ }
649
+ #endif
650
+ case GGML_TYPE_F32:
651
+ {
652
+ auto data_pts = cast_data<float>(dst);
653
+ sgn_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
654
+ break;
655
+ }
656
+ default:
657
+ GGML_ABORT("GGML tensor type not supported!\n");
658
+ break;
659
+ }
660
+ }
661
+
662
+ inline void ggml_sycl_op_abs(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
663
+ #if defined (GGML_SYCL_F16)
664
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
665
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
666
+
667
+ #else
668
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
669
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
670
+ #endif
671
+ GGML_ASSERT(dst->src[0]->type == dst->type);
672
+ dpct::queue_ptr main_stream = ctx.stream();
673
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
674
+ switch (dst->type) {
675
+ #if defined (GGML_SYCL_F16)
676
+ case GGML_TYPE_F16:
677
+ {
678
+ auto data_pts = cast_data<sycl::half>(dst);
679
+ abs_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
680
+ break;
681
+ }
682
+ #endif
683
+ case GGML_TYPE_F32:
684
+ {
685
+ auto data_pts = cast_data<float>(dst);
686
+ abs_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
687
+ break;
688
+ }
689
+ default:
690
+ GGML_ABORT("GGML tensor type not supported!\n");
691
+ break;
692
+ }
693
+ }
694
+
695
+
696
+ inline void ggml_sycl_op_elu(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
697
+ #if defined (GGML_SYCL_F16)
698
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
699
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
700
+
701
+ #else
702
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
703
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
704
+ #endif
705
+ GGML_ASSERT(dst->src[0]->type == dst->type);
706
+ dpct::queue_ptr main_stream = ctx.stream();
707
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
708
+ switch (dst->type) {
709
+ #if defined (GGML_SYCL_F16)
710
+ case GGML_TYPE_F16:
711
+ {
712
+ auto data_pts = cast_data<sycl::half>(dst);
713
+ elu_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
714
+ break;
715
+ }
716
+ #endif
717
+ case GGML_TYPE_F32:
718
+ {
719
+ auto data_pts = cast_data<float>(dst);
720
+ elu_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
721
+ break;
722
+ }
723
+ default:
724
+ GGML_ABORT("GGML tensor type not supported!\n");
725
+ break;
726
+ }
727
+ }
728
+
729
+ inline void ggml_sycl_op_silu(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
730
+ #if defined (GGML_SYCL_F16)
731
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
732
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
733
+ #else
734
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
735
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
736
+ #endif
737
+ GGML_ASSERT(dst->src[0]->type == dst->type);
738
+ dpct::queue_ptr main_stream = ctx.stream();
739
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
740
+ switch (dst->type) {
741
+ #if defined (GGML_SYCL_F16)
742
+ case GGML_TYPE_F16:
743
+ {
744
+ auto data_pts = cast_data<sycl::half>(dst);
745
+ silu_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
746
+ break;
747
+ }
748
+ #endif
749
+ case GGML_TYPE_F32:
750
+ {
751
+ auto data_pts = cast_data<float>(dst);
752
+ silu_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
753
+ break;
754
+ }
755
+ default:
756
+ GGML_ABORT("GGML tensor type not supported!\n");
757
+ break;
758
+ }
759
+ }
760
+
761
+ inline void ggml_sycl_op_gelu(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
762
+ #if defined (GGML_SYCL_F16)
763
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
764
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
765
+ #else
766
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
767
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
768
+ #endif
769
+ GGML_ASSERT(dst->src[0]->type == dst->type);
770
+ dpct::queue_ptr main_stream = ctx.stream();
771
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
772
+ switch (dst->type) {
773
+ #if defined (GGML_SYCL_F16)
774
+ case GGML_TYPE_F16:
775
+ {
776
+ auto data_pts = cast_data<sycl::half>(dst);
777
+ gelu_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
778
+ break;
779
+ }
780
+ #endif
781
+ case GGML_TYPE_F32:
782
+ {
783
+ auto data_pts = cast_data<float>(dst);
784
+ gelu_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
785
+ break;
786
+ }
787
+ default:
788
+ GGML_ABORT("GGML tensor type not supported!\n");
789
+ break;
790
+ }
791
+ }
792
+
793
+ inline void ggml_sycl_op_gelu_quick(ggml_backend_sycl_context & ctx, ggml_tensor *dst) {
794
+ #if defined (GGML_SYCL_F16)
795
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
796
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
797
+ #else
798
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
799
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
800
+ #endif
801
+ GGML_ASSERT(dst->src[0]->type == dst->type);
802
+ dpct::queue_ptr main_stream = ctx.stream();
803
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
804
+ switch (dst->type) {
805
+ #if defined (GGML_SYCL_F16)
806
+ case GGML_TYPE_F16:
807
+ {
808
+ auto data_pts = cast_data<sycl::half>(dst);
809
+ gelu_quick_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
810
+ break;
811
+ }
812
+ #endif
813
+ case GGML_TYPE_F32:
814
+ {
815
+ auto data_pts = cast_data<float>(dst);
816
+ gelu_quick_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
817
+ break;
818
+ }
819
+ default:
820
+ GGML_ABORT("GGML tensor type not supported!\n");
821
+ break;
822
+ }
823
+ }
824
+
825
+ inline void ggml_sycl_op_tanh(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
826
+ #if defined (GGML_SYCL_F16)
827
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
828
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
829
+ #else
830
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
831
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
832
+ #endif
833
+ GGML_ASSERT(dst->src[0]->type == dst->type);
834
+ dpct::queue_ptr main_stream = ctx.stream();
835
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
836
+ switch (dst->type) {
837
+ #if defined (GGML_SYCL_F16)
838
+ case GGML_TYPE_F16:
839
+ {
840
+ auto data_pts = cast_data<sycl::half>(dst);
841
+ tanh_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
842
+ break;
843
+ }
844
+ #endif
845
+ case GGML_TYPE_F32:
846
+ {
847
+ auto data_pts = cast_data<float>(dst);
848
+ tanh_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
849
+ break;
850
+ }
851
+ default:
852
+ GGML_ABORT("GGML tensor type not supported!\n");
853
+ break;
854
+ }
855
+ }
856
+
857
+ inline void ggml_sycl_op_relu(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
858
+ #if defined (GGML_SYCL_F16)
859
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
860
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
861
+ #else
862
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
863
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
864
+ #endif
865
+ GGML_ASSERT(dst->src[0]->type == dst->type);
866
+ dpct::queue_ptr main_stream = ctx.stream();
867
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
868
+
869
+ switch (dst->type) {
870
+ #if defined (GGML_SYCL_F16)
871
+ case GGML_TYPE_F16:
872
+ {
873
+ auto data_pts = cast_data<sycl::half>(dst);
874
+ relu_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
875
+ break;
876
+ }
877
+ #endif
878
+ case GGML_TYPE_F32:
879
+ {
880
+ auto data_pts = cast_data<float>(dst);
881
+ relu_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
882
+ break;
883
+ }
884
+ default:
885
+ GGML_ABORT("GGML tensor type not supported!\n");
886
+ break;
887
+ }
888
+ }
889
+
890
+ inline void ggml_sycl_op_hardsigmoid(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
891
+ #if defined (GGML_SYCL_F16)
892
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
893
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
894
+ #else
895
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
896
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
897
+ #endif
898
+ GGML_ASSERT(dst->src[0]->type == dst->type);
899
+
900
+ dpct::queue_ptr main_stream = ctx.stream();
901
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
902
+
903
+ switch (dst->type) {
904
+ #if defined (GGML_SYCL_F16)
905
+ case GGML_TYPE_F16:
906
+ {
907
+ auto data_pts = cast_data<sycl::half>(dst);
908
+ hardsigmoid_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
909
+ break;
910
+ }
911
+ #endif
912
+ case GGML_TYPE_F32:
913
+ {
914
+ auto data_pts = cast_data<float>(dst);
915
+ hardsigmoid_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
916
+ break;
917
+ }
918
+ default:
919
+ GGML_ABORT("GGML tensor type not supported!\n");
920
+ break;
921
+ }
922
+ }
923
+
924
+ inline void ggml_sycl_op_hardswish(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
925
+ #if defined (GGML_SYCL_F16)
926
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
927
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
928
+ #else
929
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
930
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
931
+ #endif
932
+ GGML_ASSERT(dst->src[0]->type == dst->type);
933
+ dpct::queue_ptr main_stream = ctx.stream();
934
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
935
+ switch (dst->type) {
936
+ #if defined (GGML_SYCL_F16)
937
+ case GGML_TYPE_F16:
938
+ {
939
+ auto data_pts = cast_data<sycl::half>(dst);
940
+ hardswish_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
941
+ break;
942
+ }
943
+ #endif
944
+ case GGML_TYPE_F32:
945
+ {
946
+ auto data_pts = cast_data<float>(dst);
947
+ hardswish_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
948
+ break;
949
+ }
950
+ default:
951
+ GGML_ABORT("GGML tensor type not supported!\n");
952
+ break;
953
+ }
954
+ }
955
+
956
+ inline void ggml_sycl_op_exp(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
957
+ #if defined (GGML_SYCL_F16)
958
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
959
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
960
+ #else
961
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
962
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
963
+ #endif
964
+ GGML_ASSERT(dst->src[0]->type == dst->type);
965
+ dpct::queue_ptr main_stream = ctx.stream();
966
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
967
+ switch (dst->type) {
968
+ #if defined (GGML_SYCL_F16)
969
+ case GGML_TYPE_F16:
970
+ {
971
+ auto data_pts = cast_data<sycl::half>(dst);
972
+ exp_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
973
+ break;
974
+ }
975
+ #endif
976
+ case GGML_TYPE_F32:
977
+ {
978
+ auto data_pts = cast_data<float>(dst);
979
+ exp_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
980
+ break;
981
+ }
982
+ default:
983
+ GGML_ABORT("GGML tensor type not supported!\n");
984
+ break;
985
+ }
986
+ }
987
+
988
+ inline void ggml_sycl_op_log(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
989
+ #if defined (GGML_SYCL_F16)
990
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
991
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
992
+ #else
993
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
994
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
995
+ #endif
996
+ GGML_ASSERT(dst->src[0]->type == dst->type);
997
+ dpct::queue_ptr main_stream = ctx.stream();
998
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
999
+ switch (dst->type) {
1000
+ #if defined (GGML_SYCL_F16)
1001
+ case GGML_TYPE_F16:
1002
+ {
1003
+ auto data_pts = cast_data<sycl::half>(dst);
1004
+ log_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1005
+ break;
1006
+ }
1007
+ #endif
1008
+ case GGML_TYPE_F32:
1009
+ {
1010
+ auto data_pts = cast_data<float>(dst);
1011
+ log_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1012
+ break;
1013
+ }
1014
+ default:
1015
+ GGML_ABORT("GGML tensor type not supported!\n");
1016
+ break;
1017
+ }
1018
+ }
1019
+
1020
+ inline void ggml_sycl_op_sigmoid(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1021
+ #if defined (GGML_SYCL_F16)
1022
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
1023
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
1024
+ #else
1025
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
1026
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
1027
+ #endif
1028
+ GGML_ASSERT(dst->src[0]->type == dst->type);
1029
+ dpct::queue_ptr main_stream = ctx.stream();
1030
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
1031
+ switch (dst->type) {
1032
+ #if defined (GGML_SYCL_F16)
1033
+ case GGML_TYPE_F16:
1034
+ {
1035
+ auto data_pts = cast_data<sycl::half>(dst);
1036
+ sigmoid_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1037
+ break;
1038
+ }
1039
+ #endif
1040
+ case GGML_TYPE_F32:
1041
+ {
1042
+ auto data_pts = cast_data<float>(dst);
1043
+ sigmoid_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1044
+ break;
1045
+ }
1046
+ default:
1047
+ GGML_ABORT("GGML tensor type not supported!\n");
1048
+ break;
1049
+ }
1050
+ }
1051
+
1052
+ inline void ggml_sycl_op_sqrt(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1053
+ #if defined (GGML_SYCL_F16)
1054
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
1055
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
1056
+ #else
1057
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
1058
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
1059
+ #endif
1060
+ GGML_ASSERT(dst->src[0]->type == dst->type);
1061
+
1062
+ dpct::queue_ptr main_stream = ctx.stream();
1063
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
1064
+ switch (dst->type) {
1065
+ #if defined (GGML_SYCL_F16)
1066
+ case GGML_TYPE_F16:
1067
+ {
1068
+ auto data_pts = cast_data<sycl::half>(dst);
1069
+ sqrt_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1070
+ break;
1071
+ }
1072
+ #endif
1073
+ case GGML_TYPE_F32:
1074
+ {
1075
+ auto data_pts = cast_data<float>(dst);
1076
+ sqrt_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1077
+ break;
1078
+ }
1079
+ default:
1080
+ GGML_ABORT("GGML tensor type not supported!\n");
1081
+ break;
1082
+ }
1083
+ }
1084
+
1085
+ inline void ggml_sycl_op_sin(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1086
+ #if defined (GGML_SYCL_F16)
1087
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
1088
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
1089
+ #else
1090
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
1091
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
1092
+ #endif
1093
+ GGML_ASSERT(dst->src[0]->type == dst->type);
1094
+ dpct::queue_ptr main_stream = ctx.stream();
1095
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
1096
+ switch (dst->type) {
1097
+ #if defined (GGML_SYCL_F16)
1098
+ case GGML_TYPE_F16:
1099
+ {
1100
+ auto data_pts = cast_data<sycl::half>(dst);
1101
+ sin_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1102
+ break;
1103
+ }
1104
+ #endif
1105
+ case GGML_TYPE_F32:
1106
+ {
1107
+ auto data_pts = cast_data<float>(dst);
1108
+ sin_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1109
+ break;
1110
+ }
1111
+ default:
1112
+ GGML_ABORT("GGML tensor type not supported!\n");
1113
+ break;
1114
+ }
1115
+ }
1116
+
1117
+ inline void ggml_sycl_op_cos(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1118
+ #if defined (GGML_SYCL_F16)
1119
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
1120
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
1121
+ #else
1122
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
1123
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
1124
+ #endif
1125
+ GGML_ASSERT(dst->src[0]->type == dst->type);
1126
+ dpct::queue_ptr main_stream = ctx.stream();
1127
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
1128
+ switch (dst->type) {
1129
+ #if defined (GGML_SYCL_F16)
1130
+ case GGML_TYPE_F16:
1131
+ {
1132
+ auto data_pts = cast_data<sycl::half>(dst);
1133
+ cos_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1134
+ break;
1135
+ }
1136
+ #endif
1137
+ case GGML_TYPE_F32:
1138
+ {
1139
+ auto data_pts = cast_data<float>(dst);
1140
+ cos_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1141
+ break;
1142
+ }
1143
+ default:
1144
+ GGML_ABORT("GGML tensor type not supported!\n");
1145
+ break;
1146
+ }
1147
+ }
1148
+
1149
+ inline void ggml_sycl_op_step(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1150
+ #if defined (GGML_SYCL_F16)
1151
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
1152
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
1153
+ #else
1154
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
1155
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
1156
+ #endif
1157
+ GGML_ASSERT(dst->src[0]->type == dst->type);
1158
+ dpct::queue_ptr main_stream = ctx.stream();
1159
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
1160
+ switch (dst->type) {
1161
+ #if defined (GGML_SYCL_F16)
1162
+ case GGML_TYPE_F16:
1163
+ {
1164
+ auto data_pts = cast_data<sycl::half>(dst);
1165
+ step_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1166
+ break;
1167
+ }
1168
+ #endif
1169
+ case GGML_TYPE_F32:
1170
+ {
1171
+ auto data_pts = cast_data<float>(dst);
1172
+ step_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1173
+ break;
1174
+ }
1175
+ default:
1176
+ GGML_ABORT("GGML tensor type not supported!\n");
1177
+ break;
1178
+ }
1179
+ }
1180
+
1181
+ inline void ggml_sycl_op_neg(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1182
+ #if defined (GGML_SYCL_F16)
1183
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
1184
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
1185
+ #else
1186
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
1187
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
1188
+ #endif
1189
+ GGML_ASSERT(dst->src[0]->type == dst->type);
1190
+ dpct::queue_ptr main_stream = ctx.stream();
1191
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
1192
+ switch (dst->type) {
1193
+ #if defined (GGML_SYCL_F16)
1194
+ case GGML_TYPE_F16:
1195
+ {
1196
+ auto data_pts = cast_data<sycl::half>(dst);
1197
+ neg_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1198
+ break;
1199
+ }
1200
+ #endif
1201
+ case GGML_TYPE_F32:
1202
+ {
1203
+ auto data_pts = cast_data<float>(dst);
1204
+ neg_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1205
+ break;
1206
+ }
1207
+ default:
1208
+ GGML_ABORT("GGML tensor type not supported!\n");
1209
+ break;
1210
+ }
1211
+ }
1212
+
1213
+ inline void ggml_sycl_op_leaky_relu(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1214
+ #if defined (GGML_SYCL_F16)
1215
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
1216
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
1217
+ #else
1218
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
1219
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
1220
+ #endif
1221
+
1222
+ GGML_ASSERT(dst->src[0]->type == dst->type);
1223
+ float negative_slope;
1224
+ memcpy(&negative_slope, dst->op_params, sizeof(float));
1225
+ dpct::queue_ptr main_stream = ctx.stream();
1226
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
1227
+ switch (dst->type) {
1228
+ #if defined (GGML_SYCL_F16)
1229
+ case GGML_TYPE_F16:
1230
+ {
1231
+ auto data_pts = cast_data<sycl::half>(dst);
1232
+ leaky_relu_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), negative_slope, main_stream);
1233
+ break;
1234
+ }
1235
+ #endif
1236
+ case GGML_TYPE_F32:
1237
+ {
1238
+ auto data_pts = cast_data<float>(dst);
1239
+ leaky_relu_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), negative_slope, main_stream);
1240
+ break;
1241
+ }
1242
+ default:
1243
+ GGML_ABORT("GGML tensor type not supported!\n");
1244
+ break;
1245
+ }
1246
+ }
1247
+
1248
+ inline void ggml_sycl_op_sqr(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1249
+ #if defined (GGML_SYCL_F16)
1250
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
1251
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
1252
+ #else
1253
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
1254
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
1255
+ #endif
1256
+ GGML_ASSERT(dst->src[0]->type == dst->type);
1257
+ dpct::queue_ptr main_stream = ctx.stream();
1258
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
1259
+ switch (dst->type) {
1260
+ #if defined (GGML_SYCL_F16)
1261
+ case GGML_TYPE_F16:
1262
+ {
1263
+ auto data_pts = cast_data<sycl::half>(dst);
1264
+ sqr_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1265
+ break;
1266
+ }
1267
+ #endif
1268
+ case GGML_TYPE_F32:
1269
+ {
1270
+ auto data_pts = cast_data<float>(dst);
1271
+ sqr_sycl(data_pts.src, data_pts.dst, ggml_nelements(dst->src[0]), main_stream);
1272
+ break;
1273
+ }
1274
+ default:
1275
+ GGML_ABORT("GGML tensor type not supported!\n");
1276
+ break;
1277
+ }
1278
+ }
1279
+
1280
+ inline void ggml_sycl_op_upscale(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1281
+ #if defined (GGML_SYCL_F16)
1282
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
1283
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
1284
+ #else
1285
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
1286
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
1287
+ #endif
1288
+ GGML_ASSERT(dst->src[0]->type == dst->type);
1289
+
1290
+ dpct::queue_ptr main_stream = ctx.stream();
1291
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
1292
+
1293
+ const float sf0 = (float) dst->ne[0] / dst->src[0]->ne[0];
1294
+ const float sf1 = (float) dst->ne[1] / dst->src[0]->ne[1];
1295
+ const float sf2 = (float) dst->ne[2] / dst->src[0]->ne[2];
1296
+ const float sf3 = (float) dst->ne[3] / dst->src[0]->ne[3];
1297
+ switch (dst->type) {
1298
+ #if defined (GGML_SYCL_F16)
1299
+ case GGML_TYPE_F16:
1300
+ {
1301
+ auto data_pts = cast_data<sycl::half>(dst);
1302
+ upscale_sycl(data_pts.src, data_pts.dst, dst->src[0]->nb[0], dst->src[0]->nb[1], dst->src[0]->nb[2],
1303
+ dst->src[0]->nb[3], dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], sf0, sf1, sf2, sf3,
1304
+ main_stream);
1305
+ break;
1306
+ }
1307
+ #endif
1308
+ case GGML_TYPE_F32:
1309
+ {
1310
+ auto data_pts = cast_data<float>(dst);
1311
+ upscale_sycl(data_pts.src, data_pts.dst, dst->src[0]->nb[0], dst->src[0]->nb[1], dst->src[0]->nb[2],
1312
+ dst->src[0]->nb[3], dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], sf0, sf1, sf2, sf3,
1313
+ main_stream);
1314
+ break;
1315
+ }
1316
+ default:
1317
+ GGML_ABORT("GGML tensor type not supported!\n");
1318
+ break;
1319
+ }
1320
+ }
1321
+
1322
+ inline void ggml_sycl_op_pad(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1323
+ #if defined (GGML_SYCL_F16)
1324
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
1325
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
1326
+ #else
1327
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
1328
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
1329
+ #endif
1330
+ GGML_ASSERT(dst->src[0]->type == dst->type);
1331
+ GGML_ASSERT(dst->src[0]->ne[3] == 1 && dst->ne[3] == 1); // just 3D tensors
1332
+ dpct::queue_ptr main_stream = ctx.stream();
1333
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
1334
+ switch (dst->type) {
1335
+ #if defined (GGML_SYCL_F16)
1336
+ case GGML_TYPE_F16:
1337
+ {
1338
+ auto data_pts = cast_data<sycl::half>(dst);
1339
+ pad_sycl(data_pts.src, data_pts.dst, dst->src[0]->ne[0], dst->src[0]->ne[1], dst->src[0]->ne[2], dst->ne[0],
1340
+ dst->ne[1], dst->ne[2], main_stream);
1341
+ break;
1342
+ }
1343
+ #endif
1344
+ case GGML_TYPE_F32:
1345
+ {
1346
+ auto data_pts = cast_data<float>(dst);
1347
+ pad_sycl(data_pts.src, data_pts.dst, dst->src[0]->ne[0], dst->src[0]->ne[1], dst->src[0]->ne[2], dst->ne[0],
1348
+ dst->ne[1], dst->ne[2], main_stream);
1349
+ break;
1350
+ }
1351
+ default:
1352
+ GGML_ABORT("GGML tensor type not supported!\n");
1353
+ break;
1354
+ }
1355
+ }
1356
+
1357
+ inline void ggml_sycl_op_clamp(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1358
+ #if defined(GGML_SYCL_F16)
1359
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32 || dst->src[0]->type == GGML_TYPE_F16);
1360
+ GGML_ASSERT(dst->type == GGML_TYPE_F32 || dst->type == GGML_TYPE_F16);
1361
+ #else
1362
+
1363
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
1364
+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
1365
+ #endif
1366
+ GGML_ASSERT(dst->src[0]->type == dst->type);
1367
+ dpct::queue_ptr main_stream = ctx.stream();
1368
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
1369
+ float min;
1370
+ float max;
1371
+ memcpy(&min, dst->op_params, sizeof(float));
1372
+ memcpy(&max, (float *) dst->op_params + 1, sizeof(float));
1373
+
1374
+ switch (dst->type) {
1375
+ #if defined(GGML_SYCL_F16)
1376
+ case GGML_TYPE_F16:
1377
+ {
1378
+ auto data_pts = cast_data<sycl::half>(dst);
1379
+ clamp_sycl(data_pts.src, data_pts.dst, min, max, ggml_nelements(dst->src[0]), main_stream);
1380
+ break;
1381
+ }
1382
+ #endif
1383
+ case GGML_TYPE_F32:
1384
+ {
1385
+ auto data_pts = cast_data<float>(dst);
1386
+ clamp_sycl(data_pts.src, data_pts.dst, min, max, ggml_nelements(dst->src[0]), main_stream);
1387
+ break;
1388
+ }
1389
+ default:
1390
+ GGML_ABORT("GGML tensor type not supported!\n");
1391
+ break;
1392
+ }
1393
+ }
1394
+
1395
+ inline void ggml_sycl_op_acc(ggml_backend_sycl_context & ctx, ggml_tensor *dst) {
1396
+
1397
+ GGML_ASSERT(dst->src[0]->type == GGML_TYPE_F32);
1398
+ GGML_ASSERT(dst->src[1]->type == GGML_TYPE_F32);
1399
+ GGML_ASSERT( dst->type == GGML_TYPE_F32);
1400
+ GGML_ASSERT(dst->ne[3] == 1); // just 3D tensors supported
1401
+ dpct::queue_ptr main_stream = ctx.stream();
1402
+ SYCL_CHECK(ggml_sycl_set_device(ctx.device));
1403
+ const float * src0_dd = static_cast<const float *>(dst->src[0]->data);
1404
+ const float * src1_dd = static_cast<const float*>(dst->src[1]->data);
1405
+ float * dst_dd = static_cast<float *>(dst->data);
1406
+
1407
+ int nb1 = dst->op_params[0] / 4; // 4 bytes of float32
1408
+ int nb2 = dst->op_params[1] / 4; // 4 bytes of float32
1409
+ // int nb3 = dst->op_params[2] / 4; // 4 bytes of float32 - unused
1410
+ int offset = dst->op_params[3] / 4; // offset in bytes
1411
+
1412
+ acc_f32_sycl(src0_dd, src1_dd, dst_dd, ggml_nelements(dst), dst->src[1]->ne[0], dst->src[1]->ne[1], dst->src[1]->ne[2], nb1, nb2, offset, main_stream);
1413
+ }
1414
+
1415
+
1416
+ void ggml_sycl_sqrt(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1417
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1418
+ ggml_sycl_op_sqrt(ctx, dst);
1419
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1420
+ }
1421
+
1422
+ void ggml_sycl_sin(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1423
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1424
+ ggml_sycl_op_sin(ctx, dst);
1425
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1426
+ }
1427
+
1428
+ void ggml_sycl_cos(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1429
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1430
+ ggml_sycl_op_cos(ctx, dst);
1431
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1432
+ }
1433
+
1434
+ void ggml_sycl_acc(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1435
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1436
+ ggml_sycl_op_acc(ctx, dst);
1437
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1438
+ }
1439
+
1440
+ void ggml_sycl_gelu(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1441
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1442
+ ggml_sycl_op_gelu(ctx, dst);
1443
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1444
+ }
1445
+
1446
+ void ggml_sycl_silu(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1447
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1448
+ ggml_sycl_op_silu(ctx, dst);
1449
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1450
+ }
1451
+
1452
+ void ggml_sycl_gelu_quick(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1453
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1454
+ ggml_sycl_op_gelu_quick(ctx, dst);
1455
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1456
+ }
1457
+
1458
+ void ggml_sycl_tanh(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1459
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1460
+ ggml_sycl_op_tanh(ctx, dst);
1461
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1462
+ }
1463
+
1464
+ void ggml_sycl_relu(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1465
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1466
+ ggml_sycl_op_relu(ctx, dst);
1467
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1468
+ }
1469
+
1470
+ void ggml_sycl_sigmoid(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1471
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1472
+ ggml_sycl_op_sigmoid(ctx, dst);
1473
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1474
+ }
1475
+
1476
+ void ggml_sycl_hardsigmoid(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1477
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1478
+ ggml_sycl_op_hardsigmoid(ctx, dst);
1479
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1480
+ }
1481
+
1482
+ void ggml_sycl_hardswish(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1483
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1484
+ ggml_sycl_op_hardswish(ctx, dst);
1485
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1486
+ }
1487
+
1488
+
1489
+ void ggml_sycl_exp(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1490
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1491
+ ggml_sycl_op_exp(ctx, dst);
1492
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1493
+ }
1494
+
1495
+ void ggml_sycl_log(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1496
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1497
+ ggml_sycl_op_log(ctx, dst);
1498
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1499
+ }
1500
+
1501
+ void ggml_sycl_neg(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1502
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1503
+ ggml_sycl_op_neg(ctx, dst);
1504
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1505
+ }
1506
+
1507
+ void ggml_sycl_step(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1508
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1509
+ ggml_sycl_op_step(ctx, dst);
1510
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1511
+ }
1512
+
1513
+ void ggml_sycl_leaky_relu(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1514
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1515
+ ggml_sycl_op_leaky_relu(ctx, dst);
1516
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1517
+ }
1518
+
1519
+ void ggml_sycl_sqr(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1520
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1521
+ ggml_sycl_op_sqr(ctx, dst);
1522
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1523
+ }
1524
+
1525
+ void ggml_sycl_upscale(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1526
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1527
+ ggml_sycl_op_upscale(ctx, dst);
1528
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1529
+ }
1530
+
1531
+ void ggml_sycl_pad(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1532
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1533
+ ggml_sycl_op_pad(ctx, dst);
1534
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1535
+ }
1536
+
1537
+ void ggml_sycl_clamp(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1538
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1539
+ ggml_sycl_op_clamp(ctx, dst);
1540
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1541
+ }
1542
+
1543
+ void ggml_sycl_sgn(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1544
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1545
+ ggml_sycl_op_sgn(ctx, dst);
1546
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1547
+ }
1548
+
1549
+ void ggml_sycl_abs(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1550
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1551
+ ggml_sycl_op_abs(ctx, dst);
1552
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1553
+ }
1554
+
1555
+ void ggml_sycl_elu(ggml_backend_sycl_context & ctx, ggml_tensor * dst) {
1556
+ GGML_SYCL_DEBUG("call %s: DST Tensor type: %s\n", __func__, ggml_type_name(dst->type));
1557
+ ggml_sycl_op_elu(ctx, dst);
1558
+ GGML_SYCL_DEBUG("call %s done\n", __func__);
1559
+ }