whispercpp 1.3.6 → 1.3.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (965) hide show
  1. checksums.yaml +4 -4
  2. data/.document +3 -0
  3. data/.rdoc_options +2 -0
  4. data/README.md +43 -9
  5. data/Rakefile +18 -3
  6. data/ext/dependencies.rb +10 -4
  7. data/ext/dependencies_for_windows.rb +17 -0
  8. data/ext/extconf.rb +20 -8
  9. data/ext/options.rb +54 -14
  10. data/ext/options_for_windows.rb +51 -0
  11. data/ext/ruby_whisper.c +35 -42
  12. data/ext/ruby_whisper.h +141 -0
  13. data/ext/ruby_whisper_context.c +157 -29
  14. data/ext/ruby_whisper_log_queue.c +180 -0
  15. data/ext/ruby_whisper_log_settable.h +46 -0
  16. data/ext/ruby_whisper_parakeet.c +49 -0
  17. data/ext/ruby_whisper_parakeet_context.c +304 -0
  18. data/ext/ruby_whisper_parakeet_context_params.c +117 -0
  19. data/ext/ruby_whisper_parakeet_model.c +84 -0
  20. data/ext/ruby_whisper_parakeet_params.c +548 -0
  21. data/ext/ruby_whisper_parakeet_segment.c +157 -0
  22. data/ext/ruby_whisper_parakeet_token.c +188 -0
  23. data/ext/ruby_whisper_parakeet_transcribe.cpp +58 -0
  24. data/ext/ruby_whisper_params.c +265 -73
  25. data/ext/ruby_whisper_segment.c +6 -6
  26. data/ext/ruby_whisper_transcribe.cpp +23 -15
  27. data/ext/ruby_whisper_vad_context.c +30 -10
  28. data/ext/ruby_whisper_vad_context_detect.cpp +8 -9
  29. data/ext/ruby_whisper_vad_params.c +4 -4
  30. data/ext/ruby_whisper_vad_segment.c +2 -2
  31. data/ext/sources/CMakeLists.txt +42 -3
  32. data/ext/sources/CMakePresets.json +95 -0
  33. data/ext/sources/cmake/parakeet-config.cmake.in +30 -0
  34. data/ext/sources/cmake/parakeet.pc.in +10 -0
  35. data/ext/sources/cmake/whisper.pc.in +2 -2
  36. data/ext/sources/examples/CMakeLists.txt +4 -2
  37. data/ext/sources/examples/bench/bench.cpp +1 -1
  38. data/ext/sources/examples/cli/cli.cpp +52 -10
  39. data/ext/sources/examples/common-ggml.cpp +4 -0
  40. data/ext/sources/examples/common-whisper.cpp +139 -67
  41. data/ext/sources/examples/common-whisper.h +11 -0
  42. data/ext/sources/examples/ffmpeg-transcode.cpp +211 -341
  43. data/ext/sources/examples/parakeet-cli/CMakeLists.txt +8 -0
  44. data/ext/sources/examples/parakeet-cli/parakeet-cli.cpp +243 -0
  45. data/ext/sources/examples/parakeet-quantize/CMakeLists.txt +7 -0
  46. data/ext/sources/examples/parakeet-quantize/parakeet-quantize.cpp +230 -0
  47. data/ext/sources/examples/server/server.cpp +199 -163
  48. data/ext/sources/examples/vad-speech-segments/speech.cpp +3 -2
  49. data/ext/sources/ggml/CMakeLists.txt +21 -14
  50. data/ext/sources/ggml/cmake/FindNCCL.cmake +36 -0
  51. data/ext/sources/ggml/cmake/ggml-config.cmake.in +12 -2
  52. data/ext/sources/ggml/include/ggml-alloc.h +1 -0
  53. data/ext/sources/ggml/include/ggml-backend.h +72 -10
  54. data/ext/sources/ggml/include/ggml-cuda.h +2 -2
  55. data/ext/sources/ggml/include/ggml-rpc.h +3 -3
  56. data/ext/sources/ggml/include/ggml-sycl.h +8 -0
  57. data/ext/sources/ggml/include/ggml.h +103 -9
  58. data/ext/sources/ggml/include/gguf.h +10 -2
  59. data/ext/sources/ggml/src/CMakeLists.txt +30 -6
  60. data/ext/sources/ggml/src/ggml-alloc.c +5 -1
  61. data/ext/sources/ggml/src/ggml-backend-impl.h +22 -2
  62. data/ext/sources/ggml/src/ggml-backend-meta.cpp +2266 -0
  63. data/ext/sources/ggml/src/ggml-backend-reg.cpp +12 -0
  64. data/ext/sources/ggml/src/ggml-backend.cpp +110 -9
  65. data/ext/sources/ggml/src/ggml-blas/ggml-blas.cpp +4 -0
  66. data/ext/sources/ggml/src/ggml-cann/aclnn_ops.cpp +672 -257
  67. data/ext/sources/ggml/src/ggml-cann/aclnn_ops.h +71 -0
  68. data/ext/sources/ggml/src/ggml-cann/common.h +20 -10
  69. data/ext/sources/ggml/src/ggml-cann/ggml-cann.cpp +211 -30
  70. data/ext/sources/ggml/src/ggml-common.h +24 -2
  71. data/ext/sources/ggml/src/ggml-cpu/CMakeLists.txt +59 -30
  72. data/ext/sources/ggml/src/ggml-cpu/amx/amx.cpp +2 -0
  73. data/ext/sources/ggml/src/ggml-cpu/amx/mmq.cpp +21 -22
  74. data/ext/sources/ggml/src/ggml-cpu/arch/arm/quants.c +194 -11
  75. data/ext/sources/ggml/src/ggml-cpu/arch/arm/repack.cpp +65 -0
  76. data/ext/sources/ggml/src/ggml-cpu/arch/loongarch/quants.c +151 -1
  77. data/ext/sources/ggml/src/ggml-cpu/arch/powerpc/quants.c +0 -1
  78. data/ext/sources/ggml/src/ggml-cpu/arch/riscv/quants.c +4279 -1292
  79. data/ext/sources/ggml/src/ggml-cpu/arch/riscv/repack.cpp +5 -35
  80. data/ext/sources/ggml/src/ggml-cpu/arch/s390/quants.c +0 -1
  81. data/ext/sources/ggml/src/ggml-cpu/arch/wasm/quants.c +72 -1
  82. data/ext/sources/ggml/src/ggml-cpu/arch/x86/quants.c +319 -31
  83. data/ext/sources/ggml/src/ggml-cpu/arch/x86/repack.cpp +1 -1
  84. data/ext/sources/ggml/src/ggml-cpu/arch-fallback.h +12 -2
  85. data/ext/sources/ggml/src/ggml-cpu/cmake/FindSMTIME.cmake +32 -0
  86. data/ext/sources/ggml/src/ggml-cpu/ggml-cpu-impl.h +10 -0
  87. data/ext/sources/ggml/src/ggml-cpu/ggml-cpu.c +109 -5
  88. data/ext/sources/ggml/src/ggml-cpu/ggml-cpu.cpp +2 -0
  89. data/ext/sources/ggml/src/ggml-cpu/kleidiai/kleidiai.cpp +146 -134
  90. data/ext/sources/ggml/src/ggml-cpu/llamafile/sgemm.cpp +107 -82
  91. data/ext/sources/ggml/src/ggml-cpu/ops.cpp +501 -119
  92. data/ext/sources/ggml/src/ggml-cpu/ops.h +3 -0
  93. data/ext/sources/ggml/src/ggml-cpu/quants.c +106 -0
  94. data/ext/sources/ggml/src/ggml-cpu/quants.h +6 -0
  95. data/ext/sources/ggml/src/ggml-cpu/repack.cpp +3 -0
  96. data/ext/sources/ggml/src/ggml-cpu/simd-gemm.h +91 -1
  97. data/ext/sources/ggml/src/ggml-cpu/simd-mappings.h +14 -16
  98. data/ext/sources/ggml/src/ggml-cpu/spacemit/ime.cpp +1402 -687
  99. data/ext/sources/ggml/src/ggml-cpu/spacemit/ime.h +8 -0
  100. data/ext/sources/ggml/src/ggml-cpu/spacemit/ime1_kernels.cpp +597 -2766
  101. data/ext/sources/ggml/src/ggml-cpu/spacemit/ime2_kernels.cpp +5768 -0
  102. data/ext/sources/ggml/src/ggml-cpu/spacemit/ime_env.cpp +320 -0
  103. data/ext/sources/ggml/src/ggml-cpu/spacemit/ime_env.h +55 -0
  104. data/ext/sources/ggml/src/ggml-cpu/spacemit/ime_kernels.h +182 -19
  105. data/ext/sources/ggml/src/ggml-cpu/spacemit/repack.cpp +1795 -0
  106. data/ext/sources/ggml/src/ggml-cpu/spacemit/repack.h +14 -0
  107. data/ext/sources/ggml/src/ggml-cpu/spacemit/rvv_kernels.cpp +3178 -0
  108. data/ext/sources/ggml/src/ggml-cpu/spacemit/rvv_kernels.h +95 -0
  109. data/ext/sources/ggml/src/ggml-cpu/spacemit/spine_barrier.h +34 -0
  110. data/ext/sources/ggml/src/ggml-cpu/spacemit/spine_mem_pool.cpp +760 -0
  111. data/ext/sources/ggml/src/ggml-cpu/spacemit/spine_mem_pool.h +32 -0
  112. data/ext/sources/ggml/src/ggml-cpu/spacemit/spine_tcm.h +409 -0
  113. data/ext/sources/ggml/src/ggml-cpu/vec.cpp +39 -55
  114. data/ext/sources/ggml/src/ggml-cpu/vec.h +225 -240
  115. data/ext/sources/ggml/src/ggml-cuda/CMakeLists.txt +17 -7
  116. data/ext/sources/ggml/src/ggml-cuda/allreduce.cu +971 -0
  117. data/ext/sources/ggml/src/ggml-cuda/allreduce.cuh +29 -0
  118. data/ext/sources/ggml/src/ggml-cuda/argsort.cu +62 -26
  119. data/ext/sources/ggml/src/ggml-cuda/binbcast.cu +134 -64
  120. data/ext/sources/ggml/src/ggml-cuda/binbcast.cuh +1 -0
  121. data/ext/sources/ggml/src/ggml-cuda/col2im-1d.cu +81 -0
  122. data/ext/sources/ggml/src/ggml-cuda/col2im-1d.cuh +3 -0
  123. data/ext/sources/ggml/src/ggml-cuda/common.cuh +246 -28
  124. data/ext/sources/ggml/src/ggml-cuda/concat.cu +134 -116
  125. data/ext/sources/ggml/src/ggml-cuda/conv-transpose-1d.cu +14 -12
  126. data/ext/sources/ggml/src/ggml-cuda/conv2d-transpose.cu +45 -21
  127. data/ext/sources/ggml/src/ggml-cuda/conv2d-transpose.cuh +1 -0
  128. data/ext/sources/ggml/src/ggml-cuda/convert.cu +139 -34
  129. data/ext/sources/ggml/src/ggml-cuda/convert.cuh +10 -0
  130. data/ext/sources/ggml/src/ggml-cuda/cpy.cu +88 -29
  131. data/ext/sources/ggml/src/ggml-cuda/dequantize.cuh +22 -0
  132. data/ext/sources/ggml/src/ggml-cuda/fattn-common.cuh +287 -49
  133. data/ext/sources/ggml/src/ggml-cuda/fattn-mma-f16.cuh +335 -130
  134. data/ext/sources/ggml/src/ggml-cuda/fattn-tile.cu +12 -0
  135. data/ext/sources/ggml/src/ggml-cuda/fattn-tile.cuh +127 -24
  136. data/ext/sources/ggml/src/ggml-cuda/fattn-vec.cuh +40 -15
  137. data/ext/sources/ggml/src/ggml-cuda/fattn-wmma-f16.cu +18 -9
  138. data/ext/sources/ggml/src/ggml-cuda/fattn.cu +169 -60
  139. data/ext/sources/ggml/src/ggml-cuda/fattn.cuh +2 -0
  140. data/ext/sources/ggml/src/ggml-cuda/fwht.cu +101 -0
  141. data/ext/sources/ggml/src/ggml-cuda/fwht.cuh +4 -0
  142. data/ext/sources/ggml/src/ggml-cuda/gated_delta_net.cu +109 -45
  143. data/ext/sources/ggml/src/ggml-cuda/gated_delta_net.cuh +10 -0
  144. data/ext/sources/ggml/src/ggml-cuda/getrows.cu +48 -23
  145. data/ext/sources/ggml/src/ggml-cuda/ggml-cuda.cu +2034 -2104
  146. data/ext/sources/ggml/src/ggml-cuda/im2col.cu +32 -29
  147. data/ext/sources/ggml/src/ggml-cuda/mean.cu +4 -2
  148. data/ext/sources/ggml/src/ggml-cuda/mma.cuh +242 -195
  149. data/ext/sources/ggml/src/ggml-cuda/mmf.cuh +3 -3
  150. data/ext/sources/ggml/src/ggml-cuda/mmq.cu +25 -12
  151. data/ext/sources/ggml/src/ggml-cuda/mmq.cuh +502 -423
  152. data/ext/sources/ggml/src/ggml-cuda/mmvf.cu +19 -12
  153. data/ext/sources/ggml/src/ggml-cuda/mmvq.cu +562 -97
  154. data/ext/sources/ggml/src/ggml-cuda/mmvq.cuh +6 -1
  155. data/ext/sources/ggml/src/ggml-cuda/norm.cu +36 -10
  156. data/ext/sources/ggml/src/ggml-cuda/out-prod.cu +66 -7
  157. data/ext/sources/ggml/src/ggml-cuda/quantize.cu +133 -26
  158. data/ext/sources/ggml/src/ggml-cuda/quantize.cuh +1 -1
  159. data/ext/sources/ggml/src/ggml-cuda/reduce_rows.cuh +5 -1
  160. data/ext/sources/ggml/src/ggml-cuda/rope.cu +11 -4
  161. data/ext/sources/ggml/src/ggml-cuda/scale.cu +4 -1
  162. data/ext/sources/ggml/src/ggml-cuda/set-rows.cu +78 -10
  163. data/ext/sources/ggml/src/ggml-cuda/snake.cu +72 -0
  164. data/ext/sources/ggml/src/ggml-cuda/snake.cuh +8 -0
  165. data/ext/sources/ggml/src/ggml-cuda/softcap.cu +4 -1
  166. data/ext/sources/ggml/src/ggml-cuda/ssm-conv.cu +45 -13
  167. data/ext/sources/ggml/src/ggml-cuda/ssm-conv.cuh +1 -1
  168. data/ext/sources/ggml/src/ggml-cuda/ssm-scan.cu +40 -18
  169. data/ext/sources/ggml/src/ggml-cuda/sumrows.cu +8 -4
  170. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_1-ncols2_16.cu +1 -0
  171. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_1-ncols2_32.cu +1 -0
  172. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_1-ncols2_8.cu +2 -0
  173. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_16-ncols2_2.cu +1 -0
  174. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_16-ncols2_4.cu +1 -0
  175. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_2-ncols2_16.cu +1 -0
  176. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_2-ncols2_32.cu +1 -0
  177. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_2-ncols2_4.cu +1 -0
  178. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_2-ncols2_8.cu +2 -0
  179. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_32-ncols2_2.cu +1 -0
  180. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_4-ncols2_16.cu +1 -0
  181. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_4-ncols2_2.cu +1 -0
  182. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_4-ncols2_4.cu +1 -0
  183. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_4-ncols2_8.cu +2 -0
  184. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_8-ncols2_2.cu +1 -0
  185. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_8-ncols2_4.cu +1 -0
  186. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_8-ncols2_8.cu +2 -0
  187. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-tile-instance-dkq192-dv128.cu +5 -0
  188. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-tile-instance-dkq320-dv256.cu +5 -0
  189. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-tile-instance-dkq512-dv512.cu +5 -0
  190. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-bf16-bf16.cu +7 -0
  191. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-bf16-f16.cu +7 -0
  192. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-bf16-q4_0.cu +7 -0
  193. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-bf16-q4_1.cu +7 -0
  194. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-bf16-q5_0.cu +7 -0
  195. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-bf16-q5_1.cu +7 -0
  196. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-bf16-q8_0.cu +7 -0
  197. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-f16-bf16.cu +7 -0
  198. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q4_0-bf16.cu +7 -0
  199. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q4_1-bf16.cu +7 -0
  200. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q5_0-bf16.cu +7 -0
  201. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q5_1-bf16.cu +7 -0
  202. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-bf16.cu +7 -0
  203. data/ext/sources/ggml/src/ggml-cuda/template-instances/mmq-instance-nvfp4.cu +5 -0
  204. data/ext/sources/ggml/src/ggml-cuda/template-instances/mmq-instance-q1_0.cu +5 -0
  205. data/ext/sources/ggml/src/ggml-cuda/top-k.cu +5 -4
  206. data/ext/sources/ggml/src/ggml-cuda/topk-moe.cu +33 -24
  207. data/ext/sources/ggml/src/ggml-cuda/unary.cu +31 -2
  208. data/ext/sources/ggml/src/ggml-cuda/unary.cuh +2 -0
  209. data/ext/sources/ggml/src/ggml-cuda/vecdotq.cuh +80 -0
  210. data/ext/sources/ggml/src/ggml-cuda/vendors/cuda.h +7 -2
  211. data/ext/sources/ggml/src/ggml-cuda/vendors/hip.h +23 -4
  212. data/ext/sources/ggml/src/ggml-cuda/vendors/musa.h +4 -0
  213. data/ext/sources/ggml/src/ggml-hexagon/CMakeLists.txt +1 -5
  214. data/ext/sources/ggml/src/ggml-hexagon/ggml-hexagon.cpp +2788 -1762
  215. data/ext/sources/ggml/src/ggml-hexagon/htp/CMakeLists.txt +13 -4
  216. data/ext/sources/ggml/src/ggml-hexagon/htp/act-ops.c +53 -84
  217. data/ext/sources/ggml/src/ggml-hexagon/htp/argsort-ops.c +25 -12
  218. data/ext/sources/ggml/src/ggml-hexagon/htp/binary-ops.c +165 -184
  219. data/ext/sources/ggml/src/ggml-hexagon/htp/cmake-toolchain.cmake +17 -19
  220. data/ext/sources/ggml/src/ggml-hexagon/htp/concat-ops.c +277 -0
  221. data/ext/sources/ggml/src/ggml-hexagon/htp/cpy-ops.c +170 -127
  222. data/ext/sources/ggml/src/ggml-hexagon/htp/cumsum-ops.c +270 -0
  223. data/ext/sources/ggml/src/ggml-hexagon/htp/diag-ops.c +216 -0
  224. data/ext/sources/ggml/src/ggml-hexagon/htp/fill-ops.c +123 -0
  225. data/ext/sources/ggml/src/ggml-hexagon/htp/flash-attn-ops.c +1774 -396
  226. data/ext/sources/ggml/src/ggml-hexagon/htp/flash-attn-ops.h +303 -0
  227. data/ext/sources/ggml/src/ggml-hexagon/htp/gated-delta-net-ops.c +1148 -0
  228. data/ext/sources/ggml/src/ggml-hexagon/htp/get-rows-ops.c +148 -42
  229. data/ext/sources/ggml/src/ggml-hexagon/htp/hex-common.h +80 -0
  230. data/ext/sources/ggml/src/ggml-hexagon/htp/hex-dma.c +2 -2
  231. data/ext/sources/ggml/src/ggml-hexagon/htp/hex-dma.h +255 -62
  232. data/ext/sources/ggml/src/ggml-hexagon/htp/hex-dump.h +9 -0
  233. data/ext/sources/ggml/src/ggml-hexagon/htp/hex-profile.h +64 -0
  234. data/ext/sources/ggml/src/ggml-hexagon/htp/hex-utils.h +25 -21
  235. data/ext/sources/ggml/src/ggml-hexagon/htp/hmx-fa-kernels.h +555 -0
  236. data/ext/sources/ggml/src/ggml-hexagon/htp/hmx-mm-kernels-tiled.h +1303 -0
  237. data/ext/sources/ggml/src/ggml-hexagon/htp/hmx-queue.c +167 -0
  238. data/ext/sources/ggml/src/ggml-hexagon/htp/hmx-queue.h +157 -0
  239. data/ext/sources/ggml/src/ggml-hexagon/htp/hmx-utils.h +222 -0
  240. data/ext/sources/ggml/src/ggml-hexagon/htp/htp-ctx.h +104 -13
  241. data/ext/sources/ggml/src/ggml-hexagon/htp/htp-ops.h +222 -57
  242. data/ext/sources/ggml/src/ggml-hexagon/htp/htp-vtcm.h +19 -0
  243. data/ext/sources/ggml/src/ggml-hexagon/htp/htp_iface.idl +10 -3
  244. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-base.h +78 -26
  245. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-copy.h +27 -10
  246. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-div.h +63 -23
  247. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-exp.h +48 -8
  248. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-fa-kernels.h +232 -0
  249. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-flash-attn.h +47 -0
  250. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-log.h +65 -0
  251. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-mm-kernels-flat.h +1511 -0
  252. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-mm-kernels-tiled.h +1200 -0
  253. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-pow.h +42 -0
  254. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-repl.h +74 -0
  255. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-sigmoid.h +40 -0
  256. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-sin-cos.h +90 -0
  257. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-utils.h +5 -8
  258. data/ext/sources/ggml/src/ggml-hexagon/htp/main.c +625 -816
  259. data/ext/sources/ggml/src/ggml-hexagon/htp/matmul-ops.c +3052 -2166
  260. data/ext/sources/ggml/src/ggml-hexagon/htp/matmul-ops.h +650 -0
  261. data/ext/sources/ggml/src/ggml-hexagon/htp/pad-ops.c +547 -0
  262. data/ext/sources/ggml/src/ggml-hexagon/htp/repeat-ops.c +148 -0
  263. data/ext/sources/ggml/src/ggml-hexagon/htp/rope-ops.c +337 -106
  264. data/ext/sources/ggml/src/ggml-hexagon/htp/set-rows-ops.c +59 -37
  265. data/ext/sources/ggml/src/ggml-hexagon/htp/softmax-ops.c +121 -133
  266. data/ext/sources/ggml/src/ggml-hexagon/htp/solve-tri-ops.c +267 -0
  267. data/ext/sources/ggml/src/ggml-hexagon/htp/ssm-conv.c +245 -151
  268. data/ext/sources/ggml/src/ggml-hexagon/htp/sum-rows-ops.c +6 -6
  269. data/ext/sources/ggml/src/ggml-hexagon/htp/unary-ops.c +719 -45
  270. data/ext/sources/ggml/src/ggml-hexagon/htp/worker-pool.c +15 -3
  271. data/ext/sources/ggml/src/ggml-hexagon/htp/worker-pool.h +8 -0
  272. data/ext/sources/ggml/src/ggml-hexagon/htp-opnode.h +390 -0
  273. data/ext/sources/ggml/src/ggml-hexagon/libggml-htp.inf +3 -5
  274. data/ext/sources/ggml/src/ggml-hip/CMakeLists.txt +27 -9
  275. data/ext/sources/ggml/src/ggml-impl.h +6 -1
  276. data/ext/sources/ggml/src/ggml-metal/ggml-metal-device.cpp +207 -18
  277. data/ext/sources/ggml/src/ggml-metal/ggml-metal-device.h +36 -2
  278. data/ext/sources/ggml/src/ggml-metal/ggml-metal-device.m +186 -29
  279. data/ext/sources/ggml/src/ggml-metal/ggml-metal-impl.h +118 -0
  280. data/ext/sources/ggml/src/ggml-metal/ggml-metal-ops.cpp +322 -21
  281. data/ext/sources/ggml/src/ggml-metal/ggml-metal-ops.h +4 -0
  282. data/ext/sources/ggml/src/ggml-metal/ggml-metal.cpp +39 -26
  283. data/ext/sources/ggml/src/ggml-metal/ggml-metal.metal +1226 -467
  284. data/ext/sources/ggml/src/ggml-musa/CMakeLists.txt +5 -6
  285. data/ext/sources/ggml/src/ggml-opencl/CMakeLists.txt +67 -5
  286. data/ext/sources/ggml/src/ggml-opencl/fa_tune.h +92 -0
  287. data/ext/sources/ggml/src/ggml-opencl/ggml-opencl.cpp +16290 -6246
  288. data/ext/sources/ggml/src/ggml-opencl/kernels/concat.cl +67 -0
  289. data/ext/sources/ggml/src/ggml-opencl/kernels/cpy.cl +59 -0
  290. data/ext/sources/ggml/src/ggml-opencl/kernels/cvt.cl +1997 -92
  291. data/ext/sources/ggml/src/ggml-opencl/kernels/flash_attn_f16.cl +81 -41
  292. data/ext/sources/ggml/src/ggml-opencl/kernels/flash_attn_f32.cl +88 -39
  293. data/ext/sources/ggml/src/ggml-opencl/kernels/flash_attn_f32_f16.cl +1995 -96
  294. data/ext/sources/ggml/src/ggml-opencl/kernels/flash_attn_f32_q4_0.cl +1615 -0
  295. data/ext/sources/ggml/src/ggml-opencl/kernels/flash_attn_f32_q8_0.cl +1486 -0
  296. data/ext/sources/ggml/src/ggml-opencl/kernels/flash_attn_pre_f16.cl +156 -0
  297. data/ext/sources/ggml/src/ggml-opencl/kernels/gated_delta_net.cl +249 -0
  298. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_mxfp4_f32_ns.cl +374 -0
  299. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_q4_0_f32_ns.cl +324 -0
  300. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_q4_1_f32_ns.cl +326 -0
  301. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_q4_k_f32_ns.cl +348 -0
  302. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_q5_0_f32_ns.cl +328 -0
  303. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_q5_1_f32_ns.cl +330 -0
  304. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_q5_k_f32_ns.cl +356 -0
  305. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_q6_k_f32_ns.cl +335 -0
  306. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_iq4_nl_f32.cl +150 -0
  307. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q1_0_f32.cl +94 -0
  308. data/ext/sources/ggml/src/ggml-opencl/kernels/{mul_mat_Ab_Bi_8x4.cl → gemm_noshuffle_q4_0_f32.cl} +1 -1
  309. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q4_k_f32.cl +172 -0
  310. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q5_0_f32.cl +131 -0
  311. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q5_1_f32.cl +134 -0
  312. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q5_k_f32.cl +176 -0
  313. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q6_k_f32.cl +140 -0
  314. data/ext/sources/ggml/src/ggml-opencl/kernels/{mul_mm_q8_0_f32_8x4.cl → gemm_noshuffle_q8_0_f32.cl} +1 -1
  315. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_xmem_f16_f32_os8.cl +233 -0
  316. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_mxfp4_f32_ns.cl +165 -0
  317. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_q4_0_f32_ns.cl +120 -0
  318. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_q4_1_f32_ns.cl +123 -0
  319. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_q4_k_f32_ns.cl +155 -0
  320. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_q5_0_f32_ns.cl +123 -0
  321. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_q5_1_f32_ns.cl +125 -0
  322. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_q5_k_f32_ns.cl +160 -0
  323. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_q6_k_f32_ns.cl +141 -0
  324. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_iq4_nl_f32.cl +302 -0
  325. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q1_0_f32.cl +121 -0
  326. data/ext/sources/ggml/src/ggml-opencl/kernels/{gemv_noshuffle_general.cl → gemv_noshuffle_q4_0_f32.cl} +5 -5
  327. data/ext/sources/ggml/src/ggml-opencl/kernels/{gemv_noshuffle.cl → gemv_noshuffle_q4_0_f32_spec.cl} +5 -5
  328. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32.cl +318 -0
  329. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q5_0_f32.cl +291 -0
  330. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q5_1_f32.cl +294 -0
  331. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q5_k_f32.cl +326 -0
  332. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32.cl +293 -0
  333. data/ext/sources/ggml/src/ggml-opencl/kernels/{gemv_noshuffle_general_q8_0_f32.cl → gemv_noshuffle_q8_0_f32.cl} +1 -1
  334. data/ext/sources/ggml/src/ggml-opencl/kernels/get_rows.cl +15 -9
  335. data/ext/sources/ggml/src/ggml-opencl/kernels/moe_reorder_b.cl +30 -0
  336. data/ext/sources/ggml/src/ggml-opencl/kernels/moe_sort_by_expert.cl +82 -0
  337. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mm_iq4_nl_f32_l4_lm.cl +171 -0
  338. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mm_q1_0_f32_l4_lm.cl +156 -0
  339. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mm_q4_k_f32_l4_lm.cl +179 -0
  340. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mm_q5_0_f32_l4_lm.cl +173 -0
  341. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mm_q5_1_f32_l4_lm.cl +175 -0
  342. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mm_q5_k_f32_l4_lm.cl +192 -0
  343. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_f16_f32_l4.cl +1149 -0
  344. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_iq4_nl_f32.cl +164 -0
  345. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_iq4_nl_f32_flat.cl +202 -0
  346. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q1_0_f32.cl +141 -0
  347. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q1_0_f32_flat.cl +190 -0
  348. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q4_k_f32_flat.cl +196 -0
  349. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q5_0_f32.cl +241 -0
  350. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q5_0_f32_flat.cl +243 -0
  351. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q5_1_f32.cl +243 -0
  352. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q5_1_f32_flat.cl +247 -0
  353. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q5_k_f32.cl +187 -0
  354. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q5_k_f32_flat.cl +203 -0
  355. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q6_k_f32_flat.cl +48 -64
  356. data/ext/sources/ggml/src/ggml-opencl/kernels/norm.cl +5 -2
  357. data/ext/sources/ggml/src/ggml-opencl/kernels/set_rows.cl +500 -0
  358. data/ext/sources/ggml/src/ggml-opencl/libdl.h +79 -0
  359. data/ext/sources/ggml/src/ggml-openvino/.clang-format +0 -5
  360. data/ext/sources/ggml/src/ggml-openvino/CMakeLists.txt +2 -4
  361. data/ext/sources/ggml/src/ggml-openvino/ggml-decoder.cpp +740 -127
  362. data/ext/sources/ggml/src/ggml-openvino/ggml-decoder.h +76 -23
  363. data/ext/sources/ggml/src/ggml-openvino/ggml-openvino-extra.cpp +75 -14
  364. data/ext/sources/ggml/src/ggml-openvino/ggml-openvino-extra.h +29 -8
  365. data/ext/sources/ggml/src/ggml-openvino/ggml-openvino.cpp +339 -69
  366. data/ext/sources/ggml/src/ggml-openvino/ggml-quants.cpp +330 -192
  367. data/ext/sources/ggml/src/ggml-openvino/ggml-quants.h +10 -4
  368. data/ext/sources/ggml/src/ggml-openvino/openvino/decoder.h +56 -16
  369. data/ext/sources/ggml/src/ggml-openvino/openvino/frontend.h +1 -1
  370. data/ext/sources/ggml/src/ggml-openvino/openvino/input_model.h +4 -4
  371. data/ext/sources/ggml/src/ggml-openvino/openvino/node_context.h +94 -37
  372. data/ext/sources/ggml/src/ggml-openvino/openvino/op/add_id.cpp +76 -0
  373. data/ext/sources/ggml/src/ggml-openvino/openvino/op/argsort.cpp +47 -0
  374. data/ext/sources/ggml/src/ggml-openvino/openvino/op/clamp.cpp +33 -0
  375. data/ext/sources/ggml/src/ggml-openvino/openvino/op/concat.cpp +48 -0
  376. data/ext/sources/ggml/src/ggml-openvino/openvino/op/cont.cpp +8 -16
  377. data/ext/sources/ggml/src/ggml-openvino/openvino/op/cpy.cpp +14 -1
  378. data/ext/sources/ggml/src/ggml-openvino/openvino/op/div.cpp +146 -0
  379. data/ext/sources/ggml/src/ggml-openvino/openvino/op/flash_attn_ext.cpp +108 -21
  380. data/ext/sources/ggml/src/ggml-openvino/openvino/op/gated_delta_net.cpp +282 -0
  381. data/ext/sources/ggml/src/ggml-openvino/openvino/op/gated_delta_net.hpp +65 -0
  382. data/ext/sources/ggml/src/ggml-openvino/openvino/op/get_rows.cpp +2 -9
  383. data/ext/sources/ggml/src/ggml-openvino/openvino/op/glu_geglu.cpp +21 -7
  384. data/ext/sources/ggml/src/ggml-openvino/openvino/op/glu_swiglu.cpp +41 -8
  385. data/ext/sources/ggml/src/ggml-openvino/openvino/op/im2col.cpp +120 -0
  386. data/ext/sources/ggml/src/ggml-openvino/openvino/op/l2_norm.cpp +44 -0
  387. data/ext/sources/ggml/src/ggml-openvino/openvino/op/mul_mat_id.cpp +226 -0
  388. data/ext/sources/ggml/src/ggml-openvino/openvino/op/mulmat.cpp +19 -9
  389. data/ext/sources/ggml/src/ggml-openvino/openvino/op/norm.cpp +58 -0
  390. data/ext/sources/ggml/src/ggml-openvino/openvino/op/pad.cpp +95 -0
  391. data/ext/sources/ggml/src/ggml-openvino/openvino/op/permute.cpp +58 -13
  392. data/ext/sources/ggml/src/ggml-openvino/openvino/op/repeat.cpp +74 -0
  393. data/ext/sources/ggml/src/ggml-openvino/openvino/op/reshape.cpp +13 -6
  394. data/ext/sources/ggml/src/ggml-openvino/openvino/op/rms_norm.cpp +1 -1
  395. data/ext/sources/ggml/src/ggml-openvino/openvino/op/rope.cpp +161 -39
  396. data/ext/sources/ggml/src/ggml-openvino/openvino/op/set_rows.cpp +3 -3
  397. data/ext/sources/ggml/src/ggml-openvino/openvino/op/softmax.cpp +126 -49
  398. data/ext/sources/ggml/src/ggml-openvino/openvino/op/ssm_conv.cpp +59 -0
  399. data/ext/sources/ggml/src/ggml-openvino/openvino/op/sum_rows.cpp +27 -0
  400. data/ext/sources/ggml/src/ggml-openvino/openvino/op/transpose.cpp +32 -1
  401. data/ext/sources/ggml/src/ggml-openvino/openvino/op/unary_silu.cpp +1 -1
  402. data/ext/sources/ggml/src/ggml-openvino/openvino/op/unary_softplus.cpp +38 -0
  403. data/ext/sources/ggml/src/ggml-openvino/openvino/op/view.cpp +90 -25
  404. data/ext/sources/ggml/src/ggml-openvino/openvino/op_table.cpp +41 -22
  405. data/ext/sources/ggml/src/ggml-openvino/openvino/op_table.h +18 -4
  406. data/ext/sources/ggml/src/ggml-openvino/openvino/pass/mark_decompression_convert_constant_folding.h +1 -1
  407. data/ext/sources/ggml/src/ggml-openvino/openvino/rt_info/weightless_caching_attributes.hpp +41 -0
  408. data/ext/sources/ggml/src/ggml-openvino/openvino/translate_session.cpp +70 -43
  409. data/ext/sources/ggml/src/ggml-openvino/openvino/translate_session.h +5 -4
  410. data/ext/sources/ggml/src/ggml-openvino/openvino/utils.cpp +612 -36
  411. data/ext/sources/ggml/src/ggml-openvino/openvino/utils.h +29 -26
  412. data/ext/sources/ggml/src/ggml-openvino/utils.cpp +460 -114
  413. data/ext/sources/ggml/src/ggml-openvino/utils.h +32 -9
  414. data/ext/sources/ggml/src/ggml-opt.cpp +1 -0
  415. data/ext/sources/ggml/src/ggml-quants.c +365 -114
  416. data/ext/sources/ggml/src/ggml-quants.h +6 -0
  417. data/ext/sources/ggml/src/ggml-rpc/CMakeLists.txt +24 -0
  418. data/ext/sources/ggml/src/ggml-rpc/ggml-rpc.cpp +167 -311
  419. data/ext/sources/ggml/src/ggml-rpc/transport.cpp +683 -0
  420. data/ext/sources/ggml/src/ggml-rpc/transport.h +34 -0
  421. data/ext/sources/ggml/src/ggml-sycl/CMakeLists.txt +50 -4
  422. data/ext/sources/ggml/src/ggml-sycl/add-id.cpp +1 -1
  423. data/ext/sources/ggml/src/ggml-sycl/backend.hpp +5 -1
  424. data/ext/sources/ggml/src/ggml-sycl/binbcast.cpp +12 -0
  425. data/ext/sources/ggml/src/ggml-sycl/col2im-1d.cpp +102 -0
  426. data/ext/sources/ggml/src/ggml-sycl/col2im-1d.hpp +8 -0
  427. data/ext/sources/ggml/src/ggml-sycl/common.cpp +72 -2
  428. data/ext/sources/ggml/src/ggml-sycl/common.hpp +59 -2
  429. data/ext/sources/ggml/src/ggml-sycl/concat.cpp +21 -1
  430. data/ext/sources/ggml/src/ggml-sycl/conv2d-dw.cpp +158 -0
  431. data/ext/sources/ggml/src/ggml-sycl/conv2d-dw.hpp +10 -0
  432. data/ext/sources/ggml/src/ggml-sycl/conv2d-transpose.cpp +125 -0
  433. data/ext/sources/ggml/src/ggml-sycl/conv2d-transpose.hpp +10 -0
  434. data/ext/sources/ggml/src/ggml-sycl/conv2d.cpp +150 -0
  435. data/ext/sources/ggml/src/ggml-sycl/conv2d.hpp +10 -0
  436. data/ext/sources/ggml/src/ggml-sycl/conv3d.cpp +224 -0
  437. data/ext/sources/ggml/src/ggml-sycl/conv3d.hpp +8 -0
  438. data/ext/sources/ggml/src/ggml-sycl/convert.cpp +121 -13
  439. data/ext/sources/ggml/src/ggml-sycl/convert.hpp +9 -0
  440. data/ext/sources/ggml/src/ggml-sycl/cpy.cpp +706 -0
  441. data/ext/sources/ggml/src/ggml-sycl/cpy.hpp +281 -0
  442. data/ext/sources/ggml/src/ggml-sycl/cross_entropy_loss.cpp +255 -0
  443. data/ext/sources/ggml/src/ggml-sycl/cross_entropy_loss.hpp +7 -0
  444. data/ext/sources/ggml/src/ggml-sycl/cumsum.cpp +148 -0
  445. data/ext/sources/ggml/src/ggml-sycl/cumsum.hpp +5 -0
  446. data/ext/sources/ggml/src/ggml-sycl/dequantize.hpp +678 -0
  447. data/ext/sources/ggml/src/ggml-sycl/diag.cpp +67 -0
  448. data/ext/sources/ggml/src/ggml-sycl/diag.hpp +5 -0
  449. data/ext/sources/ggml/src/ggml-sycl/dmmv.cpp +997 -244
  450. data/ext/sources/ggml/src/ggml-sycl/dpct/helper.hpp +15 -7
  451. data/ext/sources/ggml/src/ggml-sycl/element_wise.cpp +215 -204
  452. data/ext/sources/ggml/src/ggml-sycl/element_wise.hpp +2 -2
  453. data/ext/sources/ggml/src/ggml-sycl/fattn-buffers.cpp +56 -0
  454. data/ext/sources/ggml/src/ggml-sycl/fattn-buffers.hpp +63 -0
  455. data/ext/sources/ggml/src/ggml-sycl/fattn-common.hpp +7 -5
  456. data/ext/sources/ggml/src/ggml-sycl/fattn-tile.cpp +4 -0
  457. data/ext/sources/ggml/src/ggml-sycl/fattn-tile.hpp +76 -168
  458. data/ext/sources/ggml/src/ggml-sycl/fattn-vec.hpp +7 -0
  459. data/ext/sources/ggml/src/ggml-sycl/fattn.cpp +3 -1
  460. data/ext/sources/ggml/src/ggml-sycl/fill.cpp +55 -0
  461. data/ext/sources/ggml/src/ggml-sycl/fill.hpp +5 -0
  462. data/ext/sources/ggml/src/ggml-sycl/gated_delta_net.cpp +69 -31
  463. data/ext/sources/ggml/src/ggml-sycl/gated_delta_net.hpp +1 -0
  464. data/ext/sources/ggml/src/ggml-sycl/gemm.hpp +3 -0
  465. data/ext/sources/ggml/src/ggml-sycl/getrows.cpp +79 -3
  466. data/ext/sources/ggml/src/ggml-sycl/ggml-sycl.cpp +1758 -455
  467. data/ext/sources/ggml/src/ggml-sycl/im2col.cpp +353 -89
  468. data/ext/sources/ggml/src/ggml-sycl/im2col.hpp +5 -3
  469. data/ext/sources/ggml/src/ggml-sycl/mmvq.cpp +1542 -39
  470. data/ext/sources/ggml/src/ggml-sycl/mmvq.hpp +33 -0
  471. data/ext/sources/ggml/src/ggml-sycl/norm.cpp +103 -49
  472. data/ext/sources/ggml/src/ggml-sycl/outprod.cpp +45 -9
  473. data/ext/sources/ggml/src/ggml-sycl/pad.cpp +27 -27
  474. data/ext/sources/ggml/src/ggml-sycl/pool.cpp +185 -0
  475. data/ext/sources/ggml/src/ggml-sycl/pool.hpp +22 -0
  476. data/ext/sources/ggml/src/ggml-sycl/presets.hpp +3 -1
  477. data/ext/sources/ggml/src/ggml-sycl/quants.hpp +71 -0
  478. data/ext/sources/ggml/src/ggml-sycl/set_rows.cpp +17 -3
  479. data/ext/sources/ggml/src/ggml-sycl/softmax.cpp +9 -10
  480. data/ext/sources/ggml/src/ggml-sycl/solve_tri.cpp +172 -0
  481. data/ext/sources/ggml/src/ggml-sycl/solve_tri.hpp +8 -0
  482. data/ext/sources/ggml/src/ggml-sycl/ssm_conv.cpp +6 -1
  483. data/ext/sources/ggml/src/ggml-sycl/ssm_scan.cpp +156 -0
  484. data/ext/sources/ggml/src/ggml-sycl/ssm_scan.hpp +5 -0
  485. data/ext/sources/ggml/src/ggml-sycl/sycl_hw.cpp +62 -10
  486. data/ext/sources/ggml/src/ggml-sycl/sycl_hw.hpp +18 -6
  487. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-tile-instance-dkq512-dv512.cpp +6 -0
  488. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-f16-f16.cpp +1 -0
  489. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-f16-q4_0.cpp +1 -0
  490. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-f16-q4_1.cpp +1 -0
  491. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-f16-q5_0.cpp +1 -0
  492. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-f16-q5_1.cpp +1 -0
  493. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-f16-q8_0.cpp +1 -0
  494. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_0-f16.cpp +1 -0
  495. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_0-q4_0.cpp +1 -0
  496. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_0-q4_1.cpp +1 -0
  497. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_0-q5_0.cpp +1 -0
  498. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_0-q5_1.cpp +1 -0
  499. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_0-q8_0.cpp +1 -0
  500. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_1-f16.cpp +1 -0
  501. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_1-q4_0.cpp +1 -0
  502. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_1-q4_1.cpp +1 -0
  503. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_1-q5_0.cpp +1 -0
  504. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_1-q5_1.cpp +1 -0
  505. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_1-q8_0.cpp +1 -0
  506. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_0-f16.cpp +1 -0
  507. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_0-q4_0.cpp +1 -0
  508. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_0-q4_1.cpp +1 -0
  509. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_0-q5_0.cpp +1 -0
  510. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_0-q5_1.cpp +1 -0
  511. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_0-q8_0.cpp +1 -0
  512. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_1-f16.cpp +1 -0
  513. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_1-q4_0.cpp +1 -0
  514. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_1-q4_1.cpp +1 -0
  515. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_1-q5_0.cpp +1 -0
  516. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_1-q5_1.cpp +1 -0
  517. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_1-q8_0.cpp +1 -0
  518. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q8_0-f16.cpp +1 -0
  519. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q8_0-q4_0.cpp +1 -0
  520. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q8_0-q4_1.cpp +1 -0
  521. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q8_0-q5_0.cpp +1 -0
  522. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q8_0-q5_1.cpp +1 -0
  523. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q8_0-q8_0.cpp +1 -0
  524. data/ext/sources/ggml/src/ggml-sycl/type.hpp +112 -0
  525. data/ext/sources/ggml/src/ggml-sycl/upscale.cpp +410 -0
  526. data/ext/sources/ggml/src/ggml-sycl/upscale.hpp +9 -0
  527. data/ext/sources/ggml/src/ggml-sycl/vecdotq.hpp +242 -45
  528. data/ext/sources/ggml/src/ggml-virtgpu/ggml-backend-buffer.cpp +4 -0
  529. data/ext/sources/ggml/src/ggml-virtgpu/ggml-backend-device.cpp +2 -0
  530. data/ext/sources/ggml/src/ggml-virtgpu/ggml-backend.cpp +2 -0
  531. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu-shm.cpp +1 -0
  532. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu.cpp +1 -0
  533. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu.h +0 -2
  534. data/ext/sources/ggml/src/ggml-vulkan/CMakeLists.txt +16 -0
  535. data/ext/sources/ggml/src/ggml-vulkan/ggml-vulkan.cpp +2843 -700
  536. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/CMakeLists.txt +4 -0
  537. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/col2im_1d.comp +61 -0
  538. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/contig_copy.comp +6 -2
  539. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/conv2d_mm.comp +146 -13
  540. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/conv3d_mm.comp +431 -0
  541. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/copy.comp +3 -1
  542. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/copy_from_quant.comp +1 -1
  543. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/copy_to_quant.comp +25 -1
  544. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs.glsl +88 -0
  545. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs_cm2.glsl +643 -1
  546. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/dequant_nvfp4.comp +32 -0
  547. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q1_0.comp +29 -0
  548. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/diag.comp +3 -4
  549. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/dot_product_funcs.glsl +27 -0
  550. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/feature-tests/coopmat2_decode_vector.comp +7 -0
  551. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn.comp +198 -48
  552. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_base.glsl +60 -59
  553. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_cm1.comp +116 -113
  554. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_cm2.comp +122 -31
  555. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_dequant.glsl +131 -0
  556. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_mmq_funcs.glsl +203 -0
  557. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/fwht.comp +115 -0
  558. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/gated_delta_net.comp +125 -64
  559. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/generic_binary_head.glsl +0 -1
  560. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/generic_unary_head.glsl +21 -19
  561. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/get_rows_back.comp +25 -0
  562. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/glu_head.glsl +29 -1
  563. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/glu_main.glsl +17 -11
  564. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/im2col.comp +76 -54
  565. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/im2col_3d.comp +0 -1
  566. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/l2_norm.comp +4 -7
  567. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/log.comp +0 -1
  568. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec.comp +122 -27
  569. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iface.glsl +6 -6
  570. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_q2_k.comp +1 -1
  571. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_q4_k.comp +1 -1
  572. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_q5_k.comp +1 -1
  573. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vecq.comp +22 -24
  574. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vecq_funcs.glsl +88 -55
  575. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm.comp +42 -40
  576. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_cm2.comp +49 -15
  577. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_funcs.glsl +222 -171
  578. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_funcs.glsl +8 -8
  579. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_shmem_types.glsl +24 -9
  580. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/multi_add.comp +0 -1
  581. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/norm.comp +10 -10
  582. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/repeat_back.comp +3 -3
  583. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/roll.comp +3 -3
  584. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/rope_funcs.glsl +5 -2
  585. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/rope_head.glsl +0 -1
  586. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/rope_params.glsl +3 -2
  587. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/snake.comp +49 -0
  588. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/ssm_conv.comp +11 -1
  589. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/tri.comp +3 -4
  590. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/types.glsl +79 -2
  591. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/unary.comp +168 -0
  592. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/vulkan-shaders-gen.cpp +282 -211
  593. data/ext/sources/ggml/src/ggml-webgpu/CMakeLists.txt +5 -2
  594. data/ext/sources/ggml/src/ggml-webgpu/ggml-webgpu-shader-lib.hpp +2209 -283
  595. data/ext/sources/ggml/src/ggml-webgpu/ggml-webgpu.cpp +2618 -1416
  596. data/ext/sources/ggml/src/ggml-webgpu/pre_wgsl.hpp +37 -7
  597. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/add_id.wgsl +64 -0
  598. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/binary.wgsl +8 -7
  599. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/common_decls.tmpl +90 -95
  600. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/concat.wgsl +19 -1
  601. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/conv2d.wgsl +165 -0
  602. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/{cpy.tmpl.wgsl → cpy.wgsl} +25 -50
  603. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn.wgsl +107 -184
  604. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_quant_staging.tmpl +124 -0
  605. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_tile.wgsl +397 -0
  606. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_vec_blk.wgsl +101 -0
  607. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_vec_reduce.wgsl +84 -0
  608. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_vec_split.wgsl +619 -0
  609. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/gated_delta_net.wgsl +149 -0
  610. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/get_rows.wgsl +204 -78
  611. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/glu.wgsl +155 -0
  612. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/im2col.wgsl +101 -0
  613. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_decls.tmpl +805 -526
  614. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_id.wgsl +195 -0
  615. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_id_gather.wgsl +52 -0
  616. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_id_vec.wgsl +154 -0
  617. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_reg_tile.wgsl +8 -6
  618. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_subgroup_matrix.wgsl +5 -1
  619. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec.wgsl +90 -413
  620. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec_acc.tmpl +1553 -0
  621. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec_q_acc.tmpl +297 -0
  622. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/quant_inner_loops.tmpl +21 -0
  623. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/quantize_q8.wgsl +178 -0
  624. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/rms_norm_mul.wgsl +152 -0
  625. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/{rope.tmpl.wgsl → rope.wgsl} +71 -142
  626. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/row_norm.wgsl +153 -0
  627. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/scale.wgsl +6 -4
  628. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/set.wgsl +109 -0
  629. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/set_rows.wgsl +2 -3
  630. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/set_rows_quant.wgsl +224 -0
  631. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/{soft_max.tmpl.wgsl → soft_max.wgsl} +106 -206
  632. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/solve_tri.wgsl +121 -0
  633. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/ssm_conv.wgsl +65 -0
  634. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/ssm_scan.wgsl +193 -0
  635. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/unary.wgsl +68 -48
  636. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/upscale.wgsl +240 -0
  637. data/ext/sources/ggml/src/ggml-zdnn/ggml-zdnn.cpp +18 -14
  638. data/ext/sources/ggml/src/ggml-zendnn/CMakeLists.txt +1 -1
  639. data/ext/sources/ggml/src/ggml-zendnn/ggml-zendnn.cpp +244 -10
  640. data/ext/sources/ggml/src/ggml.c +146 -42
  641. data/ext/sources/ggml/src/gguf.cpp +173 -28
  642. data/ext/sources/include/parakeet.h +342 -0
  643. data/ext/sources/include/whisper.h +31 -0
  644. data/ext/sources/media/matmul.png +0 -0
  645. data/ext/sources/src/CMakeLists.txt +23 -0
  646. data/ext/sources/src/parakeet-arch.h +188 -0
  647. data/ext/sources/src/parakeet.cpp +3838 -0
  648. data/ext/sources/src/whisper.cpp +220 -26
  649. data/extsources.rb +26 -10
  650. data/lib/whisper/log_settable.rb +33 -0
  651. data/lib/whisper/model/uri.rb +13 -8
  652. data/lib/whisper/output.rb +74 -0
  653. data/sig/whisper.rbs +417 -62
  654. data/test/helper.rb +2 -0
  655. data/test/jfk_reader/jfk_reader.c +50 -7
  656. data/test/test_callback.rb +1 -0
  657. data/test/test_package.rb +6 -5
  658. data/test/test_parakeet.rb +28 -0
  659. data/test/test_parakeet_callback.rb +107 -0
  660. data/test/test_parakeet_context.rb +116 -0
  661. data/test/test_parakeet_context_params.rb +24 -0
  662. data/test/test_parakeet_model.rb +21 -0
  663. data/test/test_parakeet_params.rb +78 -0
  664. data/test/test_parakeet_segment.rb +42 -0
  665. data/test/test_parakeet_token.rb +73 -0
  666. data/test/test_params.rb +2 -0
  667. data/test/test_vad.rb +9 -0
  668. data/test/test_vad_context.rb +2 -2
  669. data/test/test_vad_segment.rb +1 -1
  670. data/test/test_whisper.rb +24 -6
  671. data/whispercpp.gemspec +2 -2
  672. metadata +263 -304
  673. data/ext/sources/bindings/javascript/CMakeLists.txt +0 -41
  674. data/ext/sources/bindings/javascript/emscripten.cpp +0 -93
  675. data/ext/sources/bindings/javascript/libwhisper.worker.js +0 -1
  676. data/ext/sources/bindings/javascript/package.json +0 -26
  677. data/ext/sources/bindings/javascript/whisper.js +0 -19
  678. data/ext/sources/examples/addon.node/CMakeLists.txt +0 -31
  679. data/ext/sources/examples/addon.node/__test__/whisper.spec.js +0 -133
  680. data/ext/sources/examples/addon.node/addon.cpp +0 -557
  681. data/ext/sources/examples/addon.node/index.js +0 -59
  682. data/ext/sources/examples/addon.node/package.json +0 -16
  683. data/ext/sources/examples/addon.node/vad-example.js +0 -132
  684. data/ext/sources/examples/bench.wasm/CMakeLists.txt +0 -49
  685. data/ext/sources/examples/bench.wasm/emscripten.cpp +0 -87
  686. data/ext/sources/examples/bench.wasm/index-tmpl.html +0 -285
  687. data/ext/sources/examples/coi-serviceworker.js +0 -146
  688. data/ext/sources/examples/command/CMakeLists.txt +0 -10
  689. data/ext/sources/examples/command/command.cpp +0 -802
  690. data/ext/sources/examples/command/commands.txt +0 -9
  691. data/ext/sources/examples/command.wasm/CMakeLists.txt +0 -50
  692. data/ext/sources/examples/command.wasm/emscripten.cpp +0 -327
  693. data/ext/sources/examples/command.wasm/index-tmpl.html +0 -415
  694. data/ext/sources/examples/generate-karaoke.sh +0 -57
  695. data/ext/sources/examples/helpers.js +0 -191
  696. data/ext/sources/examples/livestream.sh +0 -112
  697. data/ext/sources/examples/lsp/CMakeLists.txt +0 -10
  698. data/ext/sources/examples/lsp/lsp.cpp +0 -471
  699. data/ext/sources/examples/lsp/whisper.vim +0 -362
  700. data/ext/sources/examples/python/test_whisper_processor.py +0 -7
  701. data/ext/sources/examples/python/whisper_processor.py +0 -54
  702. data/ext/sources/examples/server/bench.js +0 -29
  703. data/ext/sources/examples/server.py +0 -120
  704. data/ext/sources/examples/stream/CMakeLists.txt +0 -10
  705. data/ext/sources/examples/stream/stream.cpp +0 -437
  706. data/ext/sources/examples/stream.wasm/CMakeLists.txt +0 -49
  707. data/ext/sources/examples/stream.wasm/emscripten.cpp +0 -216
  708. data/ext/sources/examples/stream.wasm/index-tmpl.html +0 -491
  709. data/ext/sources/examples/sycl/CMakeLists.txt +0 -9
  710. data/ext/sources/examples/sycl/build.sh +0 -22
  711. data/ext/sources/examples/sycl/ls-sycl-device.cpp +0 -11
  712. data/ext/sources/examples/sycl/run-whisper.sh +0 -17
  713. data/ext/sources/examples/talk-llama/CMakeLists.txt +0 -48
  714. data/ext/sources/examples/talk-llama/eleven-labs.py +0 -80
  715. data/ext/sources/examples/talk-llama/llama-adapter.cpp +0 -488
  716. data/ext/sources/examples/talk-llama/llama-adapter.h +0 -89
  717. data/ext/sources/examples/talk-llama/llama-arch.cpp +0 -2877
  718. data/ext/sources/examples/talk-llama/llama-arch.h +0 -628
  719. data/ext/sources/examples/talk-llama/llama-batch.cpp +0 -919
  720. data/ext/sources/examples/talk-llama/llama-batch.h +0 -173
  721. data/ext/sources/examples/talk-llama/llama-chat.cpp +0 -896
  722. data/ext/sources/examples/talk-llama/llama-chat.h +0 -71
  723. data/ext/sources/examples/talk-llama/llama-context.cpp +0 -3633
  724. data/ext/sources/examples/talk-llama/llama-context.h +0 -359
  725. data/ext/sources/examples/talk-llama/llama-cparams.cpp +0 -5
  726. data/ext/sources/examples/talk-llama/llama-cparams.h +0 -47
  727. data/ext/sources/examples/talk-llama/llama-ext.h +0 -12
  728. data/ext/sources/examples/talk-llama/llama-grammar.cpp +0 -1464
  729. data/ext/sources/examples/talk-llama/llama-grammar.h +0 -194
  730. data/ext/sources/examples/talk-llama/llama-graph.cpp +0 -2735
  731. data/ext/sources/examples/talk-llama/llama-graph.h +0 -1031
  732. data/ext/sources/examples/talk-llama/llama-hparams.cpp +0 -258
  733. data/ext/sources/examples/talk-llama/llama-hparams.h +0 -353
  734. data/ext/sources/examples/talk-llama/llama-impl.cpp +0 -171
  735. data/ext/sources/examples/talk-llama/llama-impl.h +0 -75
  736. data/ext/sources/examples/talk-llama/llama-io.cpp +0 -15
  737. data/ext/sources/examples/talk-llama/llama-io.h +0 -35
  738. data/ext/sources/examples/talk-llama/llama-kv-cache-iswa.cpp +0 -330
  739. data/ext/sources/examples/talk-llama/llama-kv-cache-iswa.h +0 -137
  740. data/ext/sources/examples/talk-llama/llama-kv-cache.cpp +0 -2285
  741. data/ext/sources/examples/talk-llama/llama-kv-cache.h +0 -389
  742. data/ext/sources/examples/talk-llama/llama-kv-cells.h +0 -533
  743. data/ext/sources/examples/talk-llama/llama-memory-hybrid-iswa.cpp +0 -275
  744. data/ext/sources/examples/talk-llama/llama-memory-hybrid-iswa.h +0 -140
  745. data/ext/sources/examples/talk-llama/llama-memory-hybrid.cpp +0 -268
  746. data/ext/sources/examples/talk-llama/llama-memory-hybrid.h +0 -139
  747. data/ext/sources/examples/talk-llama/llama-memory-recurrent.cpp +0 -1165
  748. data/ext/sources/examples/talk-llama/llama-memory-recurrent.h +0 -182
  749. data/ext/sources/examples/talk-llama/llama-memory.cpp +0 -59
  750. data/ext/sources/examples/talk-llama/llama-memory.h +0 -122
  751. data/ext/sources/examples/talk-llama/llama-mmap.cpp +0 -752
  752. data/ext/sources/examples/talk-llama/llama-mmap.h +0 -73
  753. data/ext/sources/examples/talk-llama/llama-model-loader.cpp +0 -1655
  754. data/ext/sources/examples/talk-llama/llama-model-loader.h +0 -206
  755. data/ext/sources/examples/talk-llama/llama-model-saver.cpp +0 -299
  756. data/ext/sources/examples/talk-llama/llama-model-saver.h +0 -40
  757. data/ext/sources/examples/talk-llama/llama-model.cpp +0 -9056
  758. data/ext/sources/examples/talk-llama/llama-model.h +0 -597
  759. data/ext/sources/examples/talk-llama/llama-quant.cpp +0 -1304
  760. data/ext/sources/examples/talk-llama/llama-quant.h +0 -1
  761. data/ext/sources/examples/talk-llama/llama-sampler.cpp +0 -3885
  762. data/ext/sources/examples/talk-llama/llama-sampler.h +0 -42
  763. data/ext/sources/examples/talk-llama/llama-vocab.cpp +0 -3970
  764. data/ext/sources/examples/talk-llama/llama-vocab.h +0 -187
  765. data/ext/sources/examples/talk-llama/llama.cpp +0 -1194
  766. data/ext/sources/examples/talk-llama/llama.h +0 -1573
  767. data/ext/sources/examples/talk-llama/models/afmoe.cpp +0 -190
  768. data/ext/sources/examples/talk-llama/models/apertus.cpp +0 -125
  769. data/ext/sources/examples/talk-llama/models/arcee.cpp +0 -135
  770. data/ext/sources/examples/talk-llama/models/arctic.cpp +0 -137
  771. data/ext/sources/examples/talk-llama/models/arwkv7.cpp +0 -86
  772. data/ext/sources/examples/talk-llama/models/baichuan.cpp +0 -123
  773. data/ext/sources/examples/talk-llama/models/bailingmoe.cpp +0 -143
  774. data/ext/sources/examples/talk-llama/models/bailingmoe2.cpp +0 -133
  775. data/ext/sources/examples/talk-llama/models/bert.cpp +0 -184
  776. data/ext/sources/examples/talk-llama/models/bitnet.cpp +0 -145
  777. data/ext/sources/examples/talk-llama/models/bloom.cpp +0 -101
  778. data/ext/sources/examples/talk-llama/models/chameleon.cpp +0 -178
  779. data/ext/sources/examples/talk-llama/models/chatglm.cpp +0 -132
  780. data/ext/sources/examples/talk-llama/models/codeshell.cpp +0 -111
  781. data/ext/sources/examples/talk-llama/models/cogvlm.cpp +0 -102
  782. data/ext/sources/examples/talk-llama/models/cohere2-iswa.cpp +0 -134
  783. data/ext/sources/examples/talk-llama/models/command-r.cpp +0 -122
  784. data/ext/sources/examples/talk-llama/models/dbrx.cpp +0 -122
  785. data/ext/sources/examples/talk-llama/models/deci.cpp +0 -135
  786. data/ext/sources/examples/talk-llama/models/deepseek.cpp +0 -142
  787. data/ext/sources/examples/talk-llama/models/deepseek2.cpp +0 -262
  788. data/ext/sources/examples/talk-llama/models/delta-net-base.cpp +0 -445
  789. data/ext/sources/examples/talk-llama/models/dots1.cpp +0 -132
  790. data/ext/sources/examples/talk-llama/models/dream.cpp +0 -105
  791. data/ext/sources/examples/talk-llama/models/ernie4-5-moe.cpp +0 -148
  792. data/ext/sources/examples/talk-llama/models/ernie4-5.cpp +0 -110
  793. data/ext/sources/examples/talk-llama/models/eurobert.cpp +0 -97
  794. data/ext/sources/examples/talk-llama/models/exaone-moe.cpp +0 -145
  795. data/ext/sources/examples/talk-llama/models/exaone.cpp +0 -114
  796. data/ext/sources/examples/talk-llama/models/exaone4.cpp +0 -123
  797. data/ext/sources/examples/talk-llama/models/falcon-h1.cpp +0 -111
  798. data/ext/sources/examples/talk-llama/models/falcon.cpp +0 -120
  799. data/ext/sources/examples/talk-llama/models/gemma-embedding.cpp +0 -116
  800. data/ext/sources/examples/talk-llama/models/gemma.cpp +0 -112
  801. data/ext/sources/examples/talk-llama/models/gemma2-iswa.cpp +0 -128
  802. data/ext/sources/examples/talk-llama/models/gemma3.cpp +0 -155
  803. data/ext/sources/examples/talk-llama/models/gemma3n-iswa.cpp +0 -384
  804. data/ext/sources/examples/talk-llama/models/glm4-moe.cpp +0 -170
  805. data/ext/sources/examples/talk-llama/models/glm4.cpp +0 -157
  806. data/ext/sources/examples/talk-llama/models/gpt2.cpp +0 -105
  807. data/ext/sources/examples/talk-llama/models/gptneox.cpp +0 -144
  808. data/ext/sources/examples/talk-llama/models/granite-hybrid.cpp +0 -195
  809. data/ext/sources/examples/talk-llama/models/granite.cpp +0 -210
  810. data/ext/sources/examples/talk-llama/models/grok.cpp +0 -159
  811. data/ext/sources/examples/talk-llama/models/grovemoe.cpp +0 -139
  812. data/ext/sources/examples/talk-llama/models/hunyuan-dense.cpp +0 -132
  813. data/ext/sources/examples/talk-llama/models/hunyuan-moe.cpp +0 -153
  814. data/ext/sources/examples/talk-llama/models/internlm2.cpp +0 -120
  815. data/ext/sources/examples/talk-llama/models/jais.cpp +0 -86
  816. data/ext/sources/examples/talk-llama/models/jais2.cpp +0 -123
  817. data/ext/sources/examples/talk-llama/models/jamba.cpp +0 -106
  818. data/ext/sources/examples/talk-llama/models/kimi-linear.cpp +0 -381
  819. data/ext/sources/examples/talk-llama/models/lfm2.cpp +0 -196
  820. data/ext/sources/examples/talk-llama/models/llada-moe.cpp +0 -122
  821. data/ext/sources/examples/talk-llama/models/llada.cpp +0 -99
  822. data/ext/sources/examples/talk-llama/models/llama-iswa.cpp +0 -178
  823. data/ext/sources/examples/talk-llama/models/llama.cpp +0 -175
  824. data/ext/sources/examples/talk-llama/models/maincoder.cpp +0 -117
  825. data/ext/sources/examples/talk-llama/models/mamba-base.cpp +0 -289
  826. data/ext/sources/examples/talk-llama/models/mamba.cpp +0 -54
  827. data/ext/sources/examples/talk-llama/models/mimo2-iswa.cpp +0 -129
  828. data/ext/sources/examples/talk-llama/models/minicpm3.cpp +0 -200
  829. data/ext/sources/examples/talk-llama/models/minimax-m2.cpp +0 -123
  830. data/ext/sources/examples/talk-llama/models/mistral3.cpp +0 -160
  831. data/ext/sources/examples/talk-llama/models/models.h +0 -704
  832. data/ext/sources/examples/talk-llama/models/modern-bert.cpp +0 -109
  833. data/ext/sources/examples/talk-llama/models/mpt.cpp +0 -126
  834. data/ext/sources/examples/talk-llama/models/nemotron-h.cpp +0 -162
  835. data/ext/sources/examples/talk-llama/models/nemotron.cpp +0 -122
  836. data/ext/sources/examples/talk-llama/models/neo-bert.cpp +0 -104
  837. data/ext/sources/examples/talk-llama/models/olmo.cpp +0 -121
  838. data/ext/sources/examples/talk-llama/models/olmo2.cpp +0 -150
  839. data/ext/sources/examples/talk-llama/models/olmoe.cpp +0 -124
  840. data/ext/sources/examples/talk-llama/models/openai-moe-iswa.cpp +0 -127
  841. data/ext/sources/examples/talk-llama/models/openelm.cpp +0 -124
  842. data/ext/sources/examples/talk-llama/models/orion.cpp +0 -123
  843. data/ext/sources/examples/talk-llama/models/paddleocr.cpp +0 -122
  844. data/ext/sources/examples/talk-llama/models/pangu-embedded.cpp +0 -121
  845. data/ext/sources/examples/talk-llama/models/phi2.cpp +0 -121
  846. data/ext/sources/examples/talk-llama/models/phi3.cpp +0 -152
  847. data/ext/sources/examples/talk-llama/models/plamo.cpp +0 -110
  848. data/ext/sources/examples/talk-llama/models/plamo2.cpp +0 -320
  849. data/ext/sources/examples/talk-llama/models/plamo3.cpp +0 -128
  850. data/ext/sources/examples/talk-llama/models/plm.cpp +0 -169
  851. data/ext/sources/examples/talk-llama/models/qwen.cpp +0 -108
  852. data/ext/sources/examples/talk-llama/models/qwen2.cpp +0 -126
  853. data/ext/sources/examples/talk-llama/models/qwen2moe.cpp +0 -151
  854. data/ext/sources/examples/talk-llama/models/qwen2vl.cpp +0 -117
  855. data/ext/sources/examples/talk-llama/models/qwen3.cpp +0 -120
  856. data/ext/sources/examples/talk-llama/models/qwen35.cpp +0 -381
  857. data/ext/sources/examples/talk-llama/models/qwen35moe.cpp +0 -422
  858. data/ext/sources/examples/talk-llama/models/qwen3moe.cpp +0 -131
  859. data/ext/sources/examples/talk-llama/models/qwen3next.cpp +0 -525
  860. data/ext/sources/examples/talk-llama/models/qwen3vl-moe.cpp +0 -140
  861. data/ext/sources/examples/talk-llama/models/qwen3vl.cpp +0 -132
  862. data/ext/sources/examples/talk-llama/models/refact.cpp +0 -94
  863. data/ext/sources/examples/talk-llama/models/rnd1.cpp +0 -126
  864. data/ext/sources/examples/talk-llama/models/rwkv6-base.cpp +0 -164
  865. data/ext/sources/examples/talk-llama/models/rwkv6.cpp +0 -94
  866. data/ext/sources/examples/talk-llama/models/rwkv6qwen2.cpp +0 -86
  867. data/ext/sources/examples/talk-llama/models/rwkv7-base.cpp +0 -137
  868. data/ext/sources/examples/talk-llama/models/rwkv7.cpp +0 -90
  869. data/ext/sources/examples/talk-llama/models/seed-oss.cpp +0 -124
  870. data/ext/sources/examples/talk-llama/models/smallthinker.cpp +0 -126
  871. data/ext/sources/examples/talk-llama/models/smollm3.cpp +0 -128
  872. data/ext/sources/examples/talk-llama/models/stablelm.cpp +0 -146
  873. data/ext/sources/examples/talk-llama/models/starcoder.cpp +0 -100
  874. data/ext/sources/examples/talk-llama/models/starcoder2.cpp +0 -121
  875. data/ext/sources/examples/talk-llama/models/step35-iswa.cpp +0 -165
  876. data/ext/sources/examples/talk-llama/models/t5-dec.cpp +0 -166
  877. data/ext/sources/examples/talk-llama/models/t5-enc.cpp +0 -96
  878. data/ext/sources/examples/talk-llama/models/wavtokenizer-dec.cpp +0 -149
  879. data/ext/sources/examples/talk-llama/models/xverse.cpp +0 -108
  880. data/ext/sources/examples/talk-llama/prompts/talk-alpaca.txt +0 -23
  881. data/ext/sources/examples/talk-llama/speak +0 -40
  882. data/ext/sources/examples/talk-llama/speak.bat +0 -1
  883. data/ext/sources/examples/talk-llama/speak.ps1 +0 -14
  884. data/ext/sources/examples/talk-llama/talk-llama.cpp +0 -813
  885. data/ext/sources/examples/talk-llama/unicode-data.cpp +0 -7034
  886. data/ext/sources/examples/talk-llama/unicode-data.h +0 -20
  887. data/ext/sources/examples/talk-llama/unicode.cpp +0 -1103
  888. data/ext/sources/examples/talk-llama/unicode.h +0 -111
  889. data/ext/sources/examples/wchess/CMakeLists.txt +0 -10
  890. data/ext/sources/examples/wchess/libwchess/CMakeLists.txt +0 -19
  891. data/ext/sources/examples/wchess/libwchess/Chessboard.cpp +0 -803
  892. data/ext/sources/examples/wchess/libwchess/Chessboard.h +0 -33
  893. data/ext/sources/examples/wchess/libwchess/WChess.cpp +0 -193
  894. data/ext/sources/examples/wchess/libwchess/WChess.h +0 -63
  895. data/ext/sources/examples/wchess/libwchess/test-chessboard.cpp +0 -117
  896. data/ext/sources/examples/wchess/wchess.cmd/CMakeLists.txt +0 -8
  897. data/ext/sources/examples/wchess/wchess.cmd/wchess.cmd.cpp +0 -253
  898. data/ext/sources/examples/whisper.wasm/CMakeLists.txt +0 -50
  899. data/ext/sources/examples/whisper.wasm/emscripten.cpp +0 -118
  900. data/ext/sources/examples/whisper.wasm/index-tmpl.html +0 -659
  901. data/ext/sources/ggml/src/ggml-cuda/template-instances/generate_cu_files.py +0 -99
  902. data/ext/sources/ggml/src/ggml-hexagon/htp/htp-msg.h +0 -155
  903. data/ext/sources/ggml/src/ggml-hexagon/op-desc.h +0 -153
  904. data/ext/sources/ggml/src/ggml-opencl/kernels/embed_kernel.py +0 -26
  905. data/ext/sources/ggml/src/ggml-openvino/openvino/pass/eliminate_zp.cpp +0 -123
  906. data/ext/sources/ggml/src/ggml-openvino/openvino/pass/eliminate_zp.h +0 -17
  907. data/ext/sources/ggml/src/ggml-virtgpu/regenerate_remoting.py +0 -333
  908. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/abs.comp +0 -21
  909. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/ceil.comp +0 -22
  910. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/clamp.comp +0 -17
  911. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/cos.comp +0 -17
  912. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/elu.comp +0 -27
  913. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/exp.comp +0 -21
  914. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/floor.comp +0 -22
  915. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/gelu.comp +0 -25
  916. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/gelu_erf.comp +0 -39
  917. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/gelu_quick.comp +0 -23
  918. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/hardsigmoid.comp +0 -22
  919. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/hardswish.comp +0 -22
  920. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/leaky_relu.comp +0 -22
  921. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/neg.comp +0 -20
  922. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/relu.comp +0 -21
  923. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/round.comp +0 -29
  924. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/rte.glsl +0 -5
  925. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/sgn.comp +0 -21
  926. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/sigmoid.comp +0 -20
  927. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/silu.comp +0 -22
  928. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/sin.comp +0 -17
  929. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/softplus.comp +0 -23
  930. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/sqrt.comp +0 -17
  931. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/square.comp +0 -17
  932. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/step.comp +0 -22
  933. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/tanh.comp +0 -20
  934. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/trunc.comp +0 -22
  935. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/xielu.comp +0 -35
  936. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/embed_wgsl.py +0 -182
  937. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/glu.tmpl.wgsl +0 -323
  938. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat.wgsl +0 -718
  939. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/rms_norm.wgsl +0 -123
  940. data/ext/sources/tests/CMakeLists.txt +0 -112
  941. data/ext/sources/tests/earnings21/eval.mk +0 -58
  942. data/ext/sources/tests/earnings21/eval.py +0 -68
  943. data/ext/sources/tests/earnings21/normalizers/__init__.py +0 -2
  944. data/ext/sources/tests/earnings21/normalizers/basic.py +0 -80
  945. data/ext/sources/tests/earnings21/normalizers/english.json +0 -1741
  946. data/ext/sources/tests/earnings21/normalizers/english.py +0 -550
  947. data/ext/sources/tests/earnings21/requirements.txt +0 -6
  948. data/ext/sources/tests/en-0-ref.txt +0 -1
  949. data/ext/sources/tests/en-1-ref.txt +0 -1
  950. data/ext/sources/tests/en-2-ref.txt +0 -1
  951. data/ext/sources/tests/es-0-ref.txt +0 -1
  952. data/ext/sources/tests/librispeech/eval.mk +0 -39
  953. data/ext/sources/tests/librispeech/eval.py +0 -47
  954. data/ext/sources/tests/librispeech/normalizers/__init__.py +0 -2
  955. data/ext/sources/tests/librispeech/normalizers/basic.py +0 -80
  956. data/ext/sources/tests/librispeech/normalizers/english.json +0 -1741
  957. data/ext/sources/tests/librispeech/normalizers/english.py +0 -550
  958. data/ext/sources/tests/librispeech/requirements.txt +0 -6
  959. data/ext/sources/tests/run-tests.sh +0 -130
  960. data/ext/sources/tests/test-c.c +0 -3
  961. data/ext/sources/tests/test-vad-full.cpp +0 -56
  962. data/ext/sources/tests/test-vad.cpp +0 -83
  963. data/ext/sources/tests/test-whisper.js +0 -58
  964. data/lib/whisper/context.rb +0 -15
  965. data/lib/whisper/segment.rb +0 -58
@@ -1,1304 +0,0 @@
1
- #include "llama.h"
2
- #include "llama-impl.h"
3
- #include "llama-model.h"
4
- #include "llama-model-loader.h"
5
-
6
- #include <cmath>
7
- #include <cstring>
8
- #include <string>
9
- #include <cinttypes>
10
- #include <fstream>
11
- #include <mutex>
12
- #include <regex>
13
- #include <thread>
14
- #include <unordered_map>
15
-
16
- // result of parsing --tensor-type option
17
- // (changes to this struct must be reflected in tools/quantize/quantize.cpp)
18
- struct tensor_type_option {
19
- std::string name;
20
- ggml_type type = GGML_TYPE_COUNT;
21
- };
22
-
23
- // tensor categorization - used to avoid repeated string matching in quantization logic.
24
- // this is different from LLM_TN - we want broad categories, not specific tensor names per arch.
25
- enum class tensor_category {
26
- TOKEN_EMBD,
27
- ATTENTION_Q,
28
- ATTENTION_V,
29
- ATTENTION_K,
30
- ATTENTION_QKV,
31
- ATTENTION_KV_B,
32
- ATTENTION_OUTPUT,
33
- FFN_UP,
34
- FFN_GATE,
35
- FFN_DOWN,
36
- OUTPUT,
37
- OTHER
38
- };
39
-
40
- static void zeros(std::ofstream & file, size_t n) {
41
- char zero = 0;
42
- for (size_t i = 0; i < n; ++i) {
43
- file.write(&zero, 1);
44
- }
45
- }
46
-
47
- static std::string remap_layer(const std::string & orig_name, const std::vector<int> & prune, std::map<int, std::string> & mapped, int & next_id) {
48
- if (prune.empty()) {
49
- return orig_name;
50
- }
51
-
52
- static const std::regex pattern(R"(blk\.(\d+)\.)");
53
- if (std::smatch match; std::regex_search(orig_name, match, pattern)) {
54
- const int blk = std::stoi(match[1]);
55
- std::string new_name = orig_name;
56
-
57
- if (mapped.count(blk)) {
58
- // Already mapped, do nothing
59
- } else if (std::find(prune.begin(), prune.end(), blk) != prune.end()) {
60
- mapped[blk] = "";
61
- } else if (blk < prune.front()) {
62
- mapped[blk] = std::to_string(blk);
63
- next_id = blk + 1;
64
- } else {
65
- mapped[blk] = std::to_string(next_id);
66
- ++next_id;
67
- }
68
-
69
- return mapped[blk].empty() ? mapped[blk] : new_name.replace(match.position(1), match.length(1), mapped[blk]);
70
- }
71
-
72
- return orig_name;
73
- }
74
-
75
- static std::string remap_imatrix(const std::string & orig_name, const std::map<int, std::string> & mapped) {
76
- if (mapped.empty()) {
77
- return orig_name;
78
- }
79
-
80
- static const std::regex pattern(R"(blk\.(\d+)\.)");
81
- if (std::smatch match; std::regex_search(orig_name, match, pattern)) {
82
- const std::string blk(match[1]);
83
- std::string new_name = orig_name;
84
-
85
- for (const auto & p : mapped) {
86
- if (p.second == blk) {
87
- LLAMA_LOG_DEBUG("(blk.%d imatrix) ", p.first);
88
- return new_name.replace(match.position(1), match.length(1), std::to_string(p.first));
89
- }
90
- }
91
- GGML_ABORT("\n%s: imatrix mapping error for %s\n", __func__, orig_name.c_str());
92
- }
93
-
94
- return orig_name;
95
- }
96
-
97
- //
98
- // helper functions for tensor name matching
99
- //
100
-
101
- static bool tensor_name_match_token_embd(const char * tensor_name) {
102
- return std::strcmp(tensor_name, "token_embd.weight") == 0 ||
103
- std::strcmp(tensor_name, "per_layer_token_embd.weight") == 0;
104
- }
105
-
106
- static bool tensor_name_match_output_weight(const char * tensor_name) {
107
- return std::strcmp(tensor_name, "output.weight") == 0;
108
- }
109
-
110
- //
111
- // tensor categorization for quantization
112
- //
113
- // (this is different from LLM_TN - we want broad categories, not specific tensor names per arch)
114
- //
115
-
116
- static tensor_category tensor_get_category(const std::string & tensor_name) {
117
- if (tensor_name_match_output_weight(tensor_name.c_str())) {
118
- return tensor_category::OUTPUT;
119
- }
120
- if (tensor_name_match_token_embd(tensor_name.c_str())) {
121
- return tensor_category::TOKEN_EMBD;
122
- }
123
- if (tensor_name.find("attn_qkv.weight") != std::string::npos) {
124
- return tensor_category::ATTENTION_QKV;
125
- }
126
- if (tensor_name.find("attn_kv_b.weight") != std::string::npos) {
127
- return tensor_category::ATTENTION_KV_B;
128
- }
129
- if (tensor_name.find("attn_v.weight") != std::string::npos) {
130
- return tensor_category::ATTENTION_V;
131
- }
132
- if (tensor_name.find("attn_k.weight") != std::string::npos) {
133
- return tensor_category::ATTENTION_K;
134
- }
135
- if (tensor_name.find("attn_q.weight") != std::string::npos) {
136
- return tensor_category::ATTENTION_Q;
137
- }
138
- if (tensor_name.find("attn_output.weight") != std::string::npos) {
139
- return tensor_category::ATTENTION_OUTPUT;
140
- }
141
- if (tensor_name.find("ffn_up") != std::string::npos) {
142
- return tensor_category::FFN_UP;
143
- }
144
- if (tensor_name.find("ffn_gate") != std::string::npos) {
145
- return tensor_category::FFN_GATE;
146
- }
147
- if (tensor_name.find("ffn_down") != std::string::npos) {
148
- return tensor_category::FFN_DOWN;
149
- }
150
- return tensor_category::OTHER;
151
- }
152
-
153
- // check if category is for attention-v-like tensors (more sensitive to quantization)
154
- static bool category_is_attn_v(tensor_category cat) {
155
- return cat == tensor_category::ATTENTION_V ||
156
- cat == tensor_category::ATTENTION_QKV ||
157
- cat == tensor_category::ATTENTION_KV_B;
158
- }
159
-
160
- //
161
- // quantization state
162
- //
163
-
164
- struct quantize_state_impl {
165
- const llama_model & model;
166
- const llama_model_quantize_params * params;
167
-
168
- int n_attention_wv = 0;
169
- int n_ffn_down = 0;
170
- int n_ffn_gate = 0;
171
- int n_ffn_up = 0;
172
- int i_attention_wv = 0;
173
- int i_ffn_down = 0;
174
- int i_ffn_gate = 0;
175
- int i_ffn_up = 0;
176
-
177
- int n_fallback = 0;
178
-
179
- bool has_imatrix = false;
180
-
181
- // used to figure out if a model has tied embeddings (tok_embd shares weights with output)
182
- bool has_tied_embeddings = true; // assume tied until we see output.weight
183
-
184
- // tensor type override patterns (compiled once, used twice)
185
- std::vector<std::pair<std::regex, ggml_type>> tensor_type_patterns;
186
-
187
- quantize_state_impl(const llama_model & model, const llama_model_quantize_params * params):
188
- model(model), params(params)
189
- {
190
- // compile regex patterns once - they are expensive
191
- if (params->tensor_types) {
192
- const auto & tensor_types = *static_cast<const std::vector<tensor_type_option> *>(params->tensor_types);
193
- for (const auto & [tname, qtype] : tensor_types) {
194
- tensor_type_patterns.emplace_back(std::regex(tname), qtype);
195
- }
196
- }
197
- }
198
- };
199
-
200
- // per-tensor metadata, computed in the preliminary loop and used in the main loop
201
- struct tensor_metadata {
202
- ggml_type target_type;
203
- tensor_category category;
204
- std::string remapped_imatrix_name;
205
- bool allows_quantization;
206
- bool requires_imatrix;
207
- };
208
-
209
- //
210
- // dequantization
211
- //
212
-
213
- static void llama_tensor_dequantize_impl(
214
- ggml_tensor * tensor, std::vector<no_init<float>> & output, std::vector<std::thread> & workers,
215
- const size_t nelements, const int nthread
216
- ) {
217
- if (output.size() < nelements) {
218
- output.resize(nelements);
219
- }
220
- float * f32_output = (float *) output.data();
221
-
222
- const ggml_type_traits * qtype = ggml_get_type_traits(tensor->type);
223
- if (ggml_is_quantized(tensor->type)) {
224
- if (qtype->to_float == NULL) {
225
- throw std::runtime_error(format("type %s unsupported for integer quantization: no dequantization available", ggml_type_name(tensor->type)));
226
- }
227
- } else if (tensor->type != GGML_TYPE_F16 &&
228
- tensor->type != GGML_TYPE_BF16) {
229
- throw std::runtime_error(format("cannot dequantize/convert tensor type %s", ggml_type_name(tensor->type)));
230
- }
231
-
232
- if (nthread < 2) {
233
- if (tensor->type == GGML_TYPE_F16) {
234
- ggml_fp16_to_fp32_row((ggml_fp16_t *)tensor->data, f32_output, nelements);
235
- } else if (tensor->type == GGML_TYPE_BF16) {
236
- ggml_bf16_to_fp32_row((ggml_bf16_t *)tensor->data, f32_output, nelements);
237
- } else if (ggml_is_quantized(tensor->type)) {
238
- qtype->to_float(tensor->data, f32_output, nelements);
239
- } else {
240
- GGML_ABORT("fatal error"); // unreachable
241
- }
242
- return;
243
- }
244
-
245
- size_t block_size;
246
- if (tensor->type == GGML_TYPE_F16 ||
247
- tensor->type == GGML_TYPE_BF16) {
248
- block_size = 1;
249
- } else {
250
- block_size = (size_t)ggml_blck_size(tensor->type);
251
- }
252
-
253
- size_t block_size_bytes = ggml_type_size(tensor->type);
254
-
255
- GGML_ASSERT(nelements % block_size == 0);
256
- size_t nblocks = nelements / block_size;
257
- size_t blocks_per_thread = nblocks / nthread;
258
- size_t spare_blocks = nblocks - (blocks_per_thread * nthread); // if blocks aren't divisible by thread count
259
-
260
- size_t in_buff_offs = 0;
261
- size_t out_buff_offs = 0;
262
-
263
- for (int tnum = 0; tnum < nthread; tnum++) {
264
- size_t thr_blocks = blocks_per_thread + (tnum == nthread - 1 ? spare_blocks : 0); // num blocks for this thread
265
- size_t thr_elems = thr_blocks * block_size; // number of elements for this thread
266
- size_t thr_block_bytes = thr_blocks * block_size_bytes; // number of input bytes for this thread
267
-
268
- auto compute = [qtype] (ggml_type typ, uint8_t * inbuf, float * outbuf, int nels) {
269
- if (typ == GGML_TYPE_F16) {
270
- ggml_fp16_to_fp32_row((ggml_fp16_t *)inbuf, outbuf, nels);
271
- } else if (typ == GGML_TYPE_BF16) {
272
- ggml_bf16_to_fp32_row((ggml_bf16_t *)inbuf, outbuf, nels);
273
- } else {
274
- qtype->to_float(inbuf, outbuf, nels);
275
- }
276
- };
277
- workers.emplace_back(compute, tensor->type, (uint8_t *) tensor->data + in_buff_offs, f32_output + out_buff_offs, thr_elems);
278
- in_buff_offs += thr_block_bytes;
279
- out_buff_offs += thr_elems;
280
- }
281
- for (auto & w : workers) { w.join(); }
282
- workers.clear();
283
- }
284
-
285
- //
286
- // do we allow this tensor to be quantized?
287
- //
288
-
289
- static bool tensor_allows_quantization(const llama_model_quantize_params * params, llm_arch arch, const ggml_tensor * tensor) {
290
- // trivial checks first -- no string ops needed
291
- if (params->only_copy) return false;
292
-
293
- // quantize only 2D and 3D tensors (experts)
294
- if (ggml_n_dims(tensor) < 2) return false;
295
-
296
- const std::string name = ggml_get_name(tensor);
297
-
298
- // This used to be a regex, but <regex> has an extreme cost to compile times.
299
- bool quantize = name.rfind("weight") == name.size() - 6; // ends with 'weight'?
300
-
301
- // do not quantize norm tensors
302
- quantize &= name.find("_norm.weight") == std::string::npos;
303
-
304
- quantize &= params->quantize_output_tensor || name != "output.weight";
305
-
306
- // do not quantize expert gating tensors
307
- // NOTE: can't use LLM_TN here because the layer number is not known
308
- quantize &= name.find("ffn_gate_inp.weight") == std::string::npos;
309
-
310
- // these are very small (e.g. 4x4)
311
- quantize &= name.find("altup") == std::string::npos;
312
- quantize &= name.find("laurel") == std::string::npos;
313
-
314
- // these are not too big so keep them as it is
315
- quantize &= name.find("per_layer_model_proj") == std::string::npos;
316
-
317
- // do not quantize positional embeddings and token types (BERT)
318
- quantize &= name != LLM_TN(arch)(LLM_TENSOR_POS_EMBD, "weight");
319
- quantize &= name != LLM_TN(arch)(LLM_TENSOR_TOKEN_TYPES, "weight");
320
-
321
- // do not quantize Mamba/Kimi's small conv1d weights
322
- // NOTE: can't use LLM_TN here because the layer number is not known
323
- quantize &= name.find("ssm_conv1d") == std::string::npos;
324
- quantize &= name.find("shortconv.conv.weight") == std::string::npos;
325
-
326
- // do not quantize RWKV's small yet 2D weights
327
- quantize &= name.find("time_mix_first.weight") == std::string::npos;
328
- quantize &= name.find("time_mix_w0.weight") == std::string::npos;
329
- quantize &= name.find("time_mix_w1.weight") == std::string::npos;
330
- quantize &= name.find("time_mix_w2.weight") == std::string::npos;
331
- quantize &= name.find("time_mix_v0.weight") == std::string::npos;
332
- quantize &= name.find("time_mix_v1.weight") == std::string::npos;
333
- quantize &= name.find("time_mix_v2.weight") == std::string::npos;
334
- quantize &= name.find("time_mix_a0.weight") == std::string::npos;
335
- quantize &= name.find("time_mix_a1.weight") == std::string::npos;
336
- quantize &= name.find("time_mix_a2.weight") == std::string::npos;
337
- quantize &= name.find("time_mix_g1.weight") == std::string::npos;
338
- quantize &= name.find("time_mix_g2.weight") == std::string::npos;
339
- quantize &= name.find("time_mix_decay_w1.weight") == std::string::npos;
340
- quantize &= name.find("time_mix_decay_w2.weight") == std::string::npos;
341
- quantize &= name.find("time_mix_lerp_fused.weight") == std::string::npos;
342
-
343
- // do not quantize relative position bias (T5)
344
- quantize &= name.find("attn_rel_b.weight") == std::string::npos;
345
-
346
- // do not quantize specific multimodal tensors
347
- quantize &= name.find(".position_embd.") == std::string::npos;
348
-
349
- return quantize;
350
- }
351
-
352
- //
353
- // tensor type selection
354
- //
355
-
356
- // incompatible tensor shapes are handled here - fallback to a compatible type
357
- static ggml_type tensor_type_fallback(quantize_state_impl & qs, const ggml_tensor * t, const ggml_type target_type) {
358
- ggml_type return_type = target_type;
359
-
360
- const int64_t ncols = t->ne[0];
361
- const int64_t qk_k = ggml_blck_size(target_type);
362
-
363
- if (ncols % qk_k != 0) { // this tensor's shape is incompatible with this quant
364
- LLAMA_LOG_WARN("warning: %-36s - ncols %6" PRId64 " not divisible by %3" PRId64 " (required for type %7s) ",
365
- t->name, ncols, qk_k, ggml_type_name(target_type));
366
- ++qs.n_fallback;
367
-
368
- switch (target_type) {
369
- // types on the left: block size 256
370
- case GGML_TYPE_IQ1_S:
371
- case GGML_TYPE_IQ1_M:
372
- case GGML_TYPE_IQ2_XXS:
373
- case GGML_TYPE_IQ2_XS:
374
- case GGML_TYPE_IQ2_S:
375
- case GGML_TYPE_IQ3_XXS:
376
- case GGML_TYPE_IQ3_S: // types on the right: block size 32
377
- case GGML_TYPE_IQ4_XS: return_type = GGML_TYPE_IQ4_NL; break;
378
- case GGML_TYPE_Q2_K:
379
- case GGML_TYPE_Q3_K:
380
- case GGML_TYPE_TQ1_0:
381
- case GGML_TYPE_TQ2_0: return_type = GGML_TYPE_Q4_0; break;
382
- case GGML_TYPE_Q4_K: return_type = GGML_TYPE_Q5_0; break;
383
- case GGML_TYPE_Q5_K: return_type = GGML_TYPE_Q5_1; break;
384
- case GGML_TYPE_Q6_K: return_type = GGML_TYPE_Q8_0; break;
385
- default:
386
- throw std::runtime_error(format("no tensor type fallback is defined for type %s",
387
- ggml_type_name(target_type)));
388
- }
389
- if (ncols % ggml_blck_size(return_type) != 0) {
390
- //
391
- // the fallback return type is still not compatible for this tensor!
392
- //
393
- // most likely, this tensor's first dimension is not divisible by 32.
394
- // this is very rare. we can either abort the quantization, or
395
- // fallback to F16 / F32.
396
- //
397
- LLAMA_LOG_WARN("(WARNING: must use F16 due to unusual shape) ");
398
- return_type = GGML_TYPE_F16;
399
- }
400
- LLAMA_LOG_WARN("-> falling back to %7s\n", ggml_type_name(return_type));
401
- }
402
- return return_type;
403
- }
404
-
405
- // internal standard logic for selecting the target tensor type based on tensor category, ftype, and model arch
406
- static ggml_type llama_tensor_get_type_impl(quantize_state_impl & qs, ggml_type new_type, const ggml_tensor * tensor, llama_ftype ftype, tensor_category category) {
407
- const std::string name = ggml_get_name(tensor);
408
-
409
- // TODO: avoid hardcoded tensor names - use the TN_* constants
410
- const llm_arch arch = qs.model.arch;
411
-
412
- auto use_more_bits = [](int i_layer, int n_layers) -> bool {
413
- return i_layer < n_layers/8 || i_layer >= 7*n_layers/8 || (i_layer - n_layers/8)%3 == 2;
414
- };
415
- const int n_expert = std::max(1, (int)qs.model.hparams.n_expert);
416
- auto layer_info = [n_expert] (int i_layer, int n_layer, const char * name) {
417
- if (n_expert > 1) {
418
- // Believe it or not, "experts" in the FFN of Mixtral-8x7B are not consecutive, but occasionally randomly
419
- // sprinkled in the model. Hence, simply dividing i_ffn_down by n_expert does not work
420
- // for getting the current layer as I initially thought, and we need to resort to parsing the
421
- // tensor name.
422
- if (sscanf(name, "blk.%d.", &i_layer) != 1) {
423
- throw std::runtime_error(format("Failed to determine layer for tensor %s", name));
424
- }
425
- if (i_layer < 0 || i_layer >= n_layer) {
426
- throw std::runtime_error(format("Bad layer %d for tensor %s. Must be in [0, %d)", i_layer, name, n_layer));
427
- }
428
- }
429
- return std::make_pair(i_layer, n_layer);
430
- };
431
-
432
- // for arches that share the same tensor between the token embeddings and the output, we quantize the token embeddings
433
- // with the quantization of the output tensor
434
- if (category == tensor_category::OUTPUT || (qs.has_tied_embeddings && category == tensor_category::TOKEN_EMBD)) {
435
- if (qs.params->output_tensor_type < GGML_TYPE_COUNT) {
436
- new_type = qs.params->output_tensor_type;
437
- } else {
438
- const int64_t nx = tensor->ne[0];
439
- const int64_t qk_k = ggml_blck_size(new_type);
440
-
441
- if (ftype == LLAMA_FTYPE_MOSTLY_MXFP4_MOE) {
442
- new_type = GGML_TYPE_Q8_0;
443
- }
444
- else if (arch == LLM_ARCH_FALCON || nx % qk_k != 0) {
445
- new_type = GGML_TYPE_Q8_0;
446
- }
447
- else if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_XXS || ftype == LLAMA_FTYPE_MOSTLY_IQ2_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS ||
448
- ftype == LLAMA_FTYPE_MOSTLY_IQ1_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M ||
449
- ftype == LLAMA_FTYPE_MOSTLY_IQ1_M) {
450
- new_type = GGML_TYPE_Q5_K;
451
- }
452
- else if (new_type != GGML_TYPE_Q8_0) {
453
- new_type = GGML_TYPE_Q6_K;
454
- }
455
- }
456
- } else if (ftype == LLAMA_FTYPE_MOSTLY_MXFP4_MOE) {
457
- // MoE tensors -> MXFP4
458
- // other tensors -> Q8_0
459
- if (tensor->ne[2] > 1) {
460
- new_type = GGML_TYPE_MXFP4;
461
- } else {
462
- new_type = GGML_TYPE_Q8_0;
463
- }
464
- } else if (category == tensor_category::TOKEN_EMBD) {
465
- if (qs.params->token_embedding_type < GGML_TYPE_COUNT) {
466
- new_type = qs.params->token_embedding_type;
467
- } else {
468
- if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_XXS || ftype == LLAMA_FTYPE_MOSTLY_IQ2_XS ||
469
- ftype == LLAMA_FTYPE_MOSTLY_IQ1_S || ftype == LLAMA_FTYPE_MOSTLY_IQ1_M) {
470
- new_type = GGML_TYPE_Q2_K;
471
- }
472
- else if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M) {
473
- new_type = GGML_TYPE_IQ3_S;
474
- }
475
- else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) {
476
- new_type = GGML_TYPE_IQ3_S;
477
- }
478
- else if (ftype == LLAMA_FTYPE_MOSTLY_TQ1_0 || ftype == LLAMA_FTYPE_MOSTLY_TQ2_0) {
479
- new_type = GGML_TYPE_Q4_K;
480
- }
481
- }
482
- } else if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_XXS || ftype == LLAMA_FTYPE_MOSTLY_IQ2_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ1_S ||
483
- ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M || ftype == LLAMA_FTYPE_MOSTLY_IQ1_M) {
484
- if (category_is_attn_v(category)) {
485
- if (qs.model.hparams.n_gqa() >= 4 || qs.model.hparams.n_expert >= 4) new_type = GGML_TYPE_Q4_K;
486
- else new_type = ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M ? GGML_TYPE_IQ3_S : GGML_TYPE_Q2_K;
487
- ++qs.i_attention_wv;
488
- }
489
- else if (qs.model.hparams.n_expert == 8 && category == tensor_category::ATTENTION_K) {
490
- new_type = GGML_TYPE_Q4_K;
491
- }
492
- else if (category == tensor_category::FFN_DOWN) {
493
- if (qs.i_ffn_down < qs.n_ffn_down/8) {
494
- new_type = ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M ? GGML_TYPE_IQ3_S : GGML_TYPE_Q2_K;
495
- }
496
- ++qs.i_ffn_down;
497
- }
498
- else if (category == tensor_category::ATTENTION_OUTPUT) {
499
- if (qs.model.hparams.n_expert == 8) {
500
- new_type = GGML_TYPE_Q5_K;
501
- } else {
502
- if (ftype == LLAMA_FTYPE_MOSTLY_IQ1_S || ftype == LLAMA_FTYPE_MOSTLY_IQ1_M) new_type = GGML_TYPE_IQ2_XXS;
503
- else if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M) new_type = GGML_TYPE_IQ3_S;
504
- }
505
- }
506
- } else if (category_is_attn_v(category)) {
507
- if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K) {
508
- new_type = qs.model.hparams.n_gqa() >= 4 ? GGML_TYPE_Q4_K : GGML_TYPE_Q3_K;
509
- }
510
- else if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K_S && qs.model.hparams.n_gqa() >= 4) {
511
- new_type = GGML_TYPE_Q4_K;
512
- }
513
- else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) {
514
- new_type = qs.model.hparams.n_gqa() >= 4 ? GGML_TYPE_Q4_K : !qs.has_imatrix ? GGML_TYPE_IQ3_S : GGML_TYPE_IQ3_XXS;
515
- }
516
- else if ((ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ3_S) && qs.model.hparams.n_gqa() >= 4) {
517
- new_type = GGML_TYPE_Q4_K;
518
- }
519
- else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_M) {
520
- new_type = GGML_TYPE_Q4_K;
521
- }
522
- else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M) {
523
- new_type = qs.i_attention_wv < 2 ? GGML_TYPE_Q5_K : GGML_TYPE_Q4_K;
524
- }
525
- else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) new_type = GGML_TYPE_Q5_K;
526
- else if ((ftype == LLAMA_FTYPE_MOSTLY_IQ4_NL || ftype == LLAMA_FTYPE_MOSTLY_IQ4_XS) && qs.model.hparams.n_gqa() >= 4) {
527
- new_type = GGML_TYPE_Q5_K;
528
- }
529
- else if ((ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M || ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M) &&
530
- use_more_bits(qs.i_attention_wv, qs.n_attention_wv)) new_type = GGML_TYPE_Q6_K;
531
- else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_S && qs.i_attention_wv < 4) new_type = GGML_TYPE_Q5_K;
532
- if (qs.model.type == LLM_TYPE_70B) {
533
- // In the 70B model we have 8 heads sharing the same attn_v weights. As a result, the attn_v.weight tensor is
534
- // 8x smaller compared to attn_q.weight. Hence, we can get a nice boost in quantization accuracy with
535
- // nearly negligible increase in model size by quantizing this tensor with more bits:
536
- if (new_type == GGML_TYPE_Q3_K || new_type == GGML_TYPE_Q4_K) new_type = GGML_TYPE_Q5_K;
537
- }
538
- if (qs.model.hparams.n_expert == 8) {
539
- // for the 8-expert model, bumping this to Q8_0 trades just ~128MB
540
- // TODO: explore better strategies
541
- new_type = GGML_TYPE_Q8_0;
542
- }
543
- ++qs.i_attention_wv;
544
- } else if (category == tensor_category::ATTENTION_K) {
545
- if (qs.model.hparams.n_expert == 8) {
546
- // for the 8-expert model, bumping this to Q8_0 trades just ~128MB
547
- // TODO: explore better strategies
548
- new_type = GGML_TYPE_Q8_0;
549
- }
550
- else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS) {
551
- new_type = GGML_TYPE_IQ3_XXS;
552
- }
553
- else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) {
554
- new_type = GGML_TYPE_IQ2_S;
555
- }
556
- } else if (category == tensor_category::ATTENTION_Q) {
557
- if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS) {
558
- new_type = GGML_TYPE_IQ3_XXS;
559
- }
560
- else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) {
561
- new_type = GGML_TYPE_IQ2_S;
562
- }
563
- } else if (category == tensor_category::FFN_DOWN) {
564
- auto info = layer_info(qs.i_ffn_down, qs.n_ffn_down, name.c_str());
565
- int i_layer = info.first, n_layer = info.second;
566
- if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K) new_type = GGML_TYPE_Q3_K;
567
- else if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K_S) {
568
- if (i_layer < n_layer/8) new_type = GGML_TYPE_Q4_K;
569
- }
570
- else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS && !qs.has_imatrix) {
571
- new_type = i_layer < n_layer/8 ? GGML_TYPE_Q4_K : GGML_TYPE_Q3_K;
572
- }
573
- else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M) {
574
- new_type = i_layer < n_layer/16 ? GGML_TYPE_Q5_K
575
- : arch != LLM_ARCH_FALCON || use_more_bits(i_layer, n_layer) ? GGML_TYPE_Q4_K
576
- : GGML_TYPE_Q3_K;
577
- }
578
- else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_M && (i_layer < n_layer/8 ||
579
- (qs.model.hparams.n_expert == 8 && use_more_bits(i_layer, n_layer)))) {
580
- new_type = GGML_TYPE_Q4_K;
581
- }
582
- else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) {
583
- new_type = arch == LLM_ARCH_FALCON ? GGML_TYPE_Q4_K : GGML_TYPE_Q5_K;
584
- }
585
- else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M) {
586
- if (arch == LLM_ARCH_FALCON) {
587
- new_type = i_layer < n_layer/16 ? GGML_TYPE_Q6_K :
588
- use_more_bits(i_layer, n_layer) ? GGML_TYPE_Q5_K : GGML_TYPE_Q4_K;
589
- } else {
590
- if (use_more_bits(i_layer, n_layer)) new_type = GGML_TYPE_Q6_K;
591
- }
592
- }
593
- else if (i_layer < n_layer/8 && (ftype == LLAMA_FTYPE_MOSTLY_IQ4_NL || ftype == LLAMA_FTYPE_MOSTLY_IQ4_XS) && !qs.has_imatrix) {
594
- new_type = GGML_TYPE_Q5_K;
595
- }
596
- else if (ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M && use_more_bits(i_layer, n_layer)) new_type = GGML_TYPE_Q6_K;
597
- else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_S && arch != LLM_ARCH_FALCON && i_layer < n_layer/8) {
598
- new_type = GGML_TYPE_Q5_K;
599
- }
600
- else if ((ftype == LLAMA_FTYPE_MOSTLY_Q4_0 || ftype == LLAMA_FTYPE_MOSTLY_Q5_0)
601
- && qs.has_imatrix && i_layer < n_layer/8) {
602
- // Guard against craziness in the first few ffn_down layers that can happen even with imatrix for Q4_0/Q5_0.
603
- // We only do it when an imatrix is provided because a) we want to make sure that one can always get the
604
- // same quantization as before imatrix stuff, and b) Q4_1/Q5_1 do go crazy on ffn_down without an imatrix.
605
- new_type = ftype == LLAMA_FTYPE_MOSTLY_Q4_0 ? GGML_TYPE_Q4_1 : GGML_TYPE_Q5_1;
606
- }
607
- ++qs.i_ffn_down;
608
- } else if (category == tensor_category::ATTENTION_OUTPUT) {
609
- if (arch != LLM_ARCH_FALCON) {
610
- if (qs.model.hparams.n_expert == 8) {
611
- if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K || ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS ||
612
- ftype == LLAMA_FTYPE_MOSTLY_Q3_K_S || ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M || ftype == LLAMA_FTYPE_MOSTLY_IQ4_NL ||
613
- ftype == LLAMA_FTYPE_MOSTLY_Q4_K_S || ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M || ftype == LLAMA_FTYPE_MOSTLY_IQ3_S ||
614
- ftype == LLAMA_FTYPE_MOSTLY_IQ3_M || ftype == LLAMA_FTYPE_MOSTLY_IQ4_XS) {
615
- new_type = GGML_TYPE_Q5_K;
616
- }
617
- } else {
618
- if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K ) new_type = GGML_TYPE_Q3_K;
619
- else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) new_type = GGML_TYPE_IQ3_S;
620
- else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M ) new_type = GGML_TYPE_Q4_K;
621
- else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L ) new_type = GGML_TYPE_Q5_K;
622
- else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_M ) new_type = GGML_TYPE_Q4_K;
623
- }
624
- } else {
625
- if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) new_type = GGML_TYPE_Q4_K;
626
- }
627
- }
628
- else if (category == tensor_category::ATTENTION_QKV) {
629
- if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M || ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L || ftype == LLAMA_FTYPE_MOSTLY_IQ3_M) {
630
- new_type = GGML_TYPE_Q4_K;
631
- }
632
- else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M) new_type = GGML_TYPE_Q5_K;
633
- else if (ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M) new_type = GGML_TYPE_Q6_K;
634
- }
635
- else if (category == tensor_category::FFN_GATE) {
636
- auto info = layer_info(qs.i_ffn_gate, qs.n_ffn_gate, name.c_str());
637
- int i_layer = info.first, n_layer = info.second;
638
- if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS && (i_layer >= n_layer/8 && i_layer < 7*n_layer/8)) {
639
- new_type = GGML_TYPE_IQ3_XXS;
640
- }
641
- ++qs.i_ffn_gate;
642
- }
643
- else if (category == tensor_category::FFN_UP) {
644
- auto info = layer_info(qs.i_ffn_up, qs.n_ffn_up, name.c_str());
645
- int i_layer = info.first, n_layer = info.second;
646
- if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS && (i_layer >= n_layer/8 && i_layer < 7*n_layer/8)) {
647
- new_type = GGML_TYPE_IQ3_XXS;
648
- }
649
- ++qs.i_ffn_up;
650
- }
651
-
652
- return new_type;
653
- }
654
-
655
- // outer wrapper: determine the ggml_type that this tensor should be quantized to
656
- static ggml_type llama_tensor_get_type(quantize_state_impl & qs, const llama_model_quantize_params * params, const ggml_tensor * tensor, ggml_type default_type, const tensor_metadata & tm) {
657
- if (!tensor_allows_quantization(params, qs.model.arch, tensor)) {
658
- return tensor->type;
659
- }
660
- if (params->token_embedding_type < GGML_TYPE_COUNT && tm.category == tensor_category::TOKEN_EMBD) {
661
- return params->token_embedding_type;
662
- }
663
- if (params->output_tensor_type < GGML_TYPE_COUNT && tm.category == tensor_category::OUTPUT) {
664
- return params->output_tensor_type;
665
- }
666
-
667
- ggml_type new_type = default_type;
668
-
669
- // get more optimal quantization type based on the tensor shape, layer, etc.
670
- if (!params->pure && ggml_is_quantized(default_type)) {
671
- // if the user provided tensor types - use those
672
- bool manual = false;
673
- if (!qs.tensor_type_patterns.empty()) {
674
- const std::string tensor_name(tensor->name);
675
- for (const auto & [pattern, qtype] : qs.tensor_type_patterns) {
676
- if (std::regex_search(tensor_name, pattern)) {
677
- if (qtype != new_type) {
678
- LLAMA_LOG_WARN("%s: %-36s - applying manual override: %s -> %s\n",
679
- __func__, tensor_name.c_str(), ggml_type_name(new_type), ggml_type_name(qtype));
680
- new_type = qtype;
681
- manual = true;
682
- break;
683
- }
684
- }
685
- }
686
- }
687
-
688
- // if not manual - use the standard logic for choosing the quantization type based on the selected mixture
689
- if (!manual) {
690
- new_type = llama_tensor_get_type_impl(qs, new_type, tensor, params->ftype, tm.category);
691
- }
692
-
693
- // incompatible tensor shapes are handled here - fallback to a compatible type
694
- new_type = tensor_type_fallback(qs, tensor, new_type);
695
- }
696
-
697
- return new_type;
698
- }
699
-
700
- //
701
- // quantization implementation
702
- //
703
-
704
- static size_t llama_tensor_quantize_impl(enum ggml_type new_type, const float * f32_data, void * new_data, const int64_t chunk_size, int64_t nrows, int64_t n_per_row, const float * imatrix, std::vector<std::thread> & workers, const int nthread) {
705
- if (nthread < 2) {
706
- // single-thread
707
- size_t new_size = ggml_quantize_chunk(new_type, f32_data, new_data, 0, nrows, n_per_row, imatrix);
708
- if (!ggml_validate_row_data(new_type, new_data, new_size)) {
709
- throw std::runtime_error("quantized data validation failed");
710
- }
711
- return new_size;
712
- }
713
-
714
- std::mutex mutex;
715
- int64_t counter = 0;
716
- size_t new_size = 0;
717
- bool valid = true;
718
- auto compute = [&mutex, &counter, &new_size, &valid, new_type, f32_data, new_data, chunk_size,
719
- nrows, n_per_row, imatrix]() {
720
- const int64_t nrows_per_chunk = chunk_size / n_per_row;
721
- size_t local_size = 0;
722
- while (true) {
723
- std::unique_lock<std::mutex> lock(mutex);
724
- int64_t first_row = counter; counter += nrows_per_chunk;
725
- if (first_row >= nrows) {
726
- if (local_size > 0) {
727
- new_size += local_size;
728
- }
729
- break;
730
- }
731
- lock.unlock();
732
- const int64_t this_nrow = std::min(nrows - first_row, nrows_per_chunk);
733
- size_t this_size = ggml_quantize_chunk(new_type, f32_data, new_data, first_row * n_per_row, this_nrow, n_per_row, imatrix);
734
- local_size += this_size;
735
-
736
- // validate the quantized data
737
- const size_t row_size = ggml_row_size(new_type, n_per_row);
738
- void * this_data = (char *) new_data + first_row * row_size;
739
- if (!ggml_validate_row_data(new_type, this_data, this_size)) {
740
- std::unique_lock<std::mutex> lock(mutex);
741
- valid = false;
742
- break;
743
- }
744
- }
745
- };
746
- for (int it = 0; it < nthread - 1; ++it) {
747
- workers.emplace_back(compute);
748
- }
749
- compute();
750
- for (auto & w : workers) { w.join(); }
751
- workers.clear();
752
- if (!valid) {
753
- throw std::runtime_error("quantized data validation failed");
754
- }
755
- return new_size;
756
- }
757
-
758
- //
759
- // imatrix requirement check
760
- //
761
-
762
- static bool tensor_requires_imatrix(const char * tensor_name, const ggml_type dst_type, const llama_ftype ftype) {
763
- if (tensor_name_match_token_embd(tensor_name) || tensor_name_match_output_weight(tensor_name)) {
764
- return false;
765
- }
766
- switch (dst_type) {
767
- case GGML_TYPE_IQ3_XXS:
768
- case GGML_TYPE_IQ2_XXS:
769
- case GGML_TYPE_IQ2_XS:
770
- case GGML_TYPE_IQ2_S:
771
- case GGML_TYPE_IQ1_M:
772
- case GGML_TYPE_IQ1_S:
773
- return true;
774
- case GGML_TYPE_Q2_K:
775
- // as a general rule, the k-type quantizations don't require imatrix data.
776
- // the only exception is Q2_K tensors that are part of a Q2_K_S file.
777
- return ftype == LLAMA_FTYPE_MOSTLY_Q2_K_S;
778
- default:
779
- return false;
780
- }
781
- }
782
-
783
- //
784
- // given a file type, get the default tensor type
785
- //
786
-
787
- static ggml_type llama_ftype_get_default_type(llama_ftype ftype) {
788
- switch (ftype) {
789
- case LLAMA_FTYPE_MOSTLY_Q4_0: return GGML_TYPE_Q4_0;
790
- case LLAMA_FTYPE_MOSTLY_Q4_1: return GGML_TYPE_Q4_1;
791
- case LLAMA_FTYPE_MOSTLY_Q5_0: return GGML_TYPE_Q5_0;
792
- case LLAMA_FTYPE_MOSTLY_Q5_1: return GGML_TYPE_Q5_1;
793
- case LLAMA_FTYPE_MOSTLY_Q8_0: return GGML_TYPE_Q8_0;
794
- case LLAMA_FTYPE_MOSTLY_F16: return GGML_TYPE_F16;
795
- case LLAMA_FTYPE_MOSTLY_BF16: return GGML_TYPE_BF16;
796
- case LLAMA_FTYPE_ALL_F32: return GGML_TYPE_F32;
797
-
798
- case LLAMA_FTYPE_MOSTLY_MXFP4_MOE: return GGML_TYPE_MXFP4;
799
-
800
- // K-quants
801
- case LLAMA_FTYPE_MOSTLY_Q2_K_S:
802
- case LLAMA_FTYPE_MOSTLY_Q2_K: return GGML_TYPE_Q2_K;
803
- case LLAMA_FTYPE_MOSTLY_IQ3_XS: return GGML_TYPE_IQ3_S;
804
- case LLAMA_FTYPE_MOSTLY_Q3_K_S:
805
- case LLAMA_FTYPE_MOSTLY_Q3_K_M:
806
- case LLAMA_FTYPE_MOSTLY_Q3_K_L: return GGML_TYPE_Q3_K;
807
- case LLAMA_FTYPE_MOSTLY_Q4_K_S:
808
- case LLAMA_FTYPE_MOSTLY_Q4_K_M: return GGML_TYPE_Q4_K;
809
- case LLAMA_FTYPE_MOSTLY_Q5_K_S:
810
- case LLAMA_FTYPE_MOSTLY_Q5_K_M: return GGML_TYPE_Q5_K;
811
- case LLAMA_FTYPE_MOSTLY_Q6_K: return GGML_TYPE_Q6_K;
812
- case LLAMA_FTYPE_MOSTLY_TQ1_0: return GGML_TYPE_TQ1_0;
813
- case LLAMA_FTYPE_MOSTLY_TQ2_0: return GGML_TYPE_TQ2_0;
814
- case LLAMA_FTYPE_MOSTLY_IQ2_XXS: return GGML_TYPE_IQ2_XXS;
815
- case LLAMA_FTYPE_MOSTLY_IQ2_XS: return GGML_TYPE_IQ2_XS;
816
- case LLAMA_FTYPE_MOSTLY_IQ2_S: return GGML_TYPE_IQ2_XS;
817
- case LLAMA_FTYPE_MOSTLY_IQ2_M: return GGML_TYPE_IQ2_S;
818
- case LLAMA_FTYPE_MOSTLY_IQ3_XXS: return GGML_TYPE_IQ3_XXS;
819
- case LLAMA_FTYPE_MOSTLY_IQ1_S: return GGML_TYPE_IQ1_S;
820
- case LLAMA_FTYPE_MOSTLY_IQ1_M: return GGML_TYPE_IQ1_M;
821
- case LLAMA_FTYPE_MOSTLY_IQ4_NL: return GGML_TYPE_IQ4_NL;
822
- case LLAMA_FTYPE_MOSTLY_IQ4_XS: return GGML_TYPE_IQ4_XS;
823
- case LLAMA_FTYPE_MOSTLY_IQ3_S:
824
- case LLAMA_FTYPE_MOSTLY_IQ3_M: return GGML_TYPE_IQ3_S;
825
-
826
- default: throw std::runtime_error(format("invalid output file type %d\n", ftype));
827
- }
828
- }
829
-
830
- //
831
- // main quantization driver
832
- //
833
-
834
- static void llama_model_quantize_impl(const std::string & fname_inp, const std::string & fname_out, const llama_model_quantize_params * params) {
835
- ggml_type default_type;
836
- llama_ftype ftype = params->ftype;
837
-
838
- int nthread = params->nthread;
839
-
840
- if (nthread <= 0) {
841
- nthread = std::thread::hardware_concurrency();
842
- }
843
-
844
- default_type = llama_ftype_get_default_type(ftype);
845
-
846
- // mmap consistently increases speed on Linux, and also increases speed on Windows with
847
- // hot cache. It may cause a slowdown on macOS, possibly related to free memory.
848
- #if defined(__linux__) || defined(_WIN32)
849
- constexpr bool use_mmap = true;
850
- #else
851
- constexpr bool use_mmap = false;
852
- #endif
853
-
854
- llama_model_kv_override * kv_overrides = nullptr;
855
- if (params->kv_overrides) {
856
- auto * v = (std::vector<llama_model_kv_override>*)params->kv_overrides;
857
- kv_overrides = v->data();
858
- }
859
-
860
- std::vector<std::string> splits = {};
861
- llama_model_loader ml(/*metadata*/ nullptr, /*set_tensor_data*/ nullptr, /*set_tensor_data_ud*/ nullptr,
862
- fname_inp, splits, use_mmap, /*use_direct_io*/ false, /*check_tensors*/ true, /*no_alloc*/ false, kv_overrides, nullptr);
863
- ml.init_mappings(false); // no prefetching
864
-
865
- llama_model model(llama_model_default_params());
866
-
867
- model.load_arch (ml);
868
- model.load_hparams(ml);
869
- model.load_stats (ml);
870
-
871
- quantize_state_impl qs(model, params);
872
-
873
- if (params->only_copy) {
874
- ftype = ml.ftype;
875
- }
876
- const std::unordered_map<std::string, std::vector<float>> * imatrix_data = nullptr;
877
- if (params->imatrix) {
878
- imatrix_data = static_cast<const std::unordered_map<std::string, std::vector<float>>*>(params->imatrix);
879
- if (imatrix_data) {
880
- LLAMA_LOG_INFO("\n%s: have importance matrix data with %d entries\n",
881
- __func__, (int)imatrix_data->size());
882
- qs.has_imatrix = true;
883
- // check imatrix for nans or infs
884
- for (const auto & kv : *imatrix_data) {
885
- for (float f : kv.second) {
886
- if (!std::isfinite(f)) {
887
- throw std::runtime_error(format("imatrix contains non-finite value %f\n", f));
888
- }
889
- }
890
- }
891
- }
892
- }
893
-
894
- const size_t align = GGUF_DEFAULT_ALIGNMENT;
895
- gguf_context_ptr ctx_out { gguf_init_empty() };
896
-
897
- std::vector<int> prune_list = {};
898
- if (params->prune_layers) {
899
- prune_list = *static_cast<const std::vector<int> *>(params->prune_layers);
900
- }
901
-
902
- // copy the KV pairs from the input file
903
- gguf_set_kv (ctx_out.get(), ml.metadata);
904
- gguf_set_val_u32(ctx_out.get(), "general.quantization_version", GGML_QNT_VERSION); // TODO: use LLM_KV
905
- gguf_set_val_u32(ctx_out.get(), "general.file_type", ftype); // TODO: use LLM_KV
906
-
907
- // Remove split metadata
908
- gguf_remove_key(ctx_out.get(), ml.llm_kv(LLM_KV_SPLIT_NO).c_str());
909
- gguf_remove_key(ctx_out.get(), ml.llm_kv(LLM_KV_SPLIT_COUNT).c_str());
910
- gguf_remove_key(ctx_out.get(), ml.llm_kv(LLM_KV_SPLIT_TENSORS_COUNT).c_str());
911
-
912
- if (params->kv_overrides) {
913
- const std::vector<llama_model_kv_override> & overrides = *(const std::vector<llama_model_kv_override> *)params->kv_overrides;
914
- for (const auto & o : overrides) {
915
- if (o.key[0] == 0) break;
916
- if (o.tag == LLAMA_KV_OVERRIDE_TYPE_FLOAT) {
917
- gguf_set_val_f32(ctx_out.get(), o.key, o.val_f64);
918
- } else if (o.tag == LLAMA_KV_OVERRIDE_TYPE_INT) {
919
- // Setting type to UINT32. See https://github.com/ggml-org/llama.cpp/pull/14182 for context
920
- gguf_set_val_u32(ctx_out.get(), o.key, (uint32_t)std::abs(o.val_i64));
921
- } else if (o.tag == LLAMA_KV_OVERRIDE_TYPE_BOOL) {
922
- gguf_set_val_bool(ctx_out.get(), o.key, o.val_bool);
923
- } else if (o.tag == LLAMA_KV_OVERRIDE_TYPE_STR) {
924
- gguf_set_val_str(ctx_out.get(), o.key, o.val_str);
925
- } else {
926
- LLAMA_LOG_WARN("%s: unknown KV override type for key %s\n", __func__, o.key);
927
- }
928
- }
929
- }
930
-
931
- std::map<int, std::string> mapped;
932
- int blk_id = 0;
933
-
934
- // make a list of weights
935
- std::vector<const llama_model_loader::llama_tensor_weight *> tensors;
936
- tensors.reserve(ml.weights_map.size());
937
- for (const auto & it : ml.weights_map) {
938
- const std::string remapped_name(remap_layer(it.first, prune_list, mapped, blk_id));
939
- if (remapped_name.empty()) {
940
- LLAMA_LOG_DEBUG("%s: pruning tensor %s\n", __func__, it.first.c_str());
941
- continue;
942
- }
943
-
944
- if (remapped_name != it.first) {
945
- ggml_set_name(it.second.tensor, remapped_name.c_str());
946
- LLAMA_LOG_DEBUG("%s: tensor %s remapped to %s\n", __func__, it.first.c_str(), ggml_get_name(it.second.tensor));
947
- }
948
- tensors.push_back(&it.second);
949
- }
950
- if (!prune_list.empty()) {
951
- gguf_set_val_u32(ctx_out.get(), ml.llm_kv(LLM_KV_BLOCK_COUNT).c_str(), blk_id);
952
- }
953
-
954
- // keep_split requires that the weights are sorted by split index
955
- if (params->keep_split) {
956
- std::sort(tensors.begin(), tensors.end(), [](const llama_model_loader::llama_tensor_weight * a, const llama_model_loader::llama_tensor_weight * b) {
957
- if (a->idx == b->idx) {
958
- return a->offs < b->offs;
959
- }
960
- return a->idx < b->idx;
961
- });
962
- }
963
-
964
- int idx = 0;
965
- uint16_t n_split = 1;
966
-
967
- // Assume split index is continuous
968
- if (params->keep_split) {
969
- for (const auto * it : tensors) {
970
- n_split = std::max(uint16_t(it->idx + 1), n_split);
971
- }
972
- }
973
- std::vector<gguf_context_ptr> ctx_outs(n_split);
974
- ctx_outs[0] = std::move(ctx_out);
975
-
976
- // compute tensor metadata once and cache it
977
- std::vector<tensor_metadata> metadata(tensors.size());
978
-
979
- // initialize quantization state before preliminary loop (counters for use_more_bits)
980
- {
981
- for (size_t i = 0; i < tensors.size(); ++i) {
982
- const auto cat = tensor_get_category(tensors[i]->tensor->name);
983
- if (category_is_attn_v(cat)) {
984
- ++qs.n_attention_wv;
985
- }
986
- if (cat == tensor_category::OUTPUT) {
987
- qs.has_tied_embeddings = false;
988
- }
989
- metadata[i].category = cat; // save and re-use the category while we're at it
990
- }
991
- // these also need to be set to n_layer by default
992
- qs.n_ffn_down = qs.n_ffn_gate = qs.n_ffn_up = (int)qs.model.hparams.n_layer;
993
- }
994
-
995
- // flag for --dry-run
996
- bool will_require_imatrix = false;
997
-
998
- //
999
- // preliminary iteration over all weights
1000
- //
1001
-
1002
- for (size_t i = 0; i < tensors.size(); ++i) {
1003
- const auto * it = tensors[i];
1004
- const struct ggml_tensor * tensor = it->tensor;
1005
- const std::string name = ggml_get_name(tensor);
1006
-
1007
- uint16_t i_split = params->keep_split ? it->idx : 0;
1008
- if (!ctx_outs[i_split]) {
1009
- ctx_outs[i_split].reset(gguf_init_empty());
1010
- }
1011
- gguf_add_tensor(ctx_outs[i_split].get(), tensor);
1012
-
1013
- metadata[i].allows_quantization = tensor_allows_quantization(params, model.arch, tensor);
1014
-
1015
- if (metadata[i].allows_quantization) {
1016
- metadata[i].target_type = llama_tensor_get_type(qs, params, tensor, default_type, metadata[i]);
1017
- } else {
1018
- metadata[i].target_type = tensor->type;
1019
- }
1020
-
1021
- metadata[i].requires_imatrix = tensor_requires_imatrix(tensor->name, metadata[i].target_type, ftype);
1022
-
1023
- if (params->imatrix) {
1024
- metadata[i].remapped_imatrix_name = remap_imatrix(tensor->name, mapped);
1025
- } else if (metadata[i].allows_quantization && metadata[i].requires_imatrix) {
1026
- if (params->dry_run) {
1027
- will_require_imatrix = true;
1028
- } else {
1029
- LLAMA_LOG_ERROR("\n============================================================================\n"
1030
- " ERROR: this quantization requires an importance matrix!\n"
1031
- " - offending tensor: %s\n"
1032
- " - target type: %s\n"
1033
- "============================================================================\n\n",
1034
- name.c_str(), ggml_type_name(metadata[i].target_type));
1035
- throw std::runtime_error("this quantization requires an imatrix!");
1036
- }
1037
- }
1038
- }
1039
-
1040
- // Set split info if needed
1041
- if (n_split > 1) {
1042
- for (size_t i = 0; i < ctx_outs.size(); ++i) {
1043
- gguf_set_val_u16(ctx_outs[i].get(), ml.llm_kv(LLM_KV_SPLIT_NO).c_str(), i);
1044
- gguf_set_val_u16(ctx_outs[i].get(), ml.llm_kv(LLM_KV_SPLIT_COUNT).c_str(), n_split);
1045
- gguf_set_val_i32(ctx_outs[i].get(), ml.llm_kv(LLM_KV_SPLIT_TENSORS_COUNT).c_str(), (int32_t)tensors.size());
1046
- }
1047
- }
1048
-
1049
- size_t total_size_org = 0;
1050
- size_t total_size_new = 0;
1051
-
1052
- std::vector<std::thread> workers;
1053
- workers.reserve(nthread);
1054
-
1055
- std::vector<no_init<uint8_t>> read_data;
1056
- std::vector<no_init<uint8_t>> work;
1057
- std::vector<no_init<float>> f32_conv_buf;
1058
-
1059
- int cur_split = -1;
1060
- std::ofstream fout;
1061
- auto close_ofstream = [&]() {
1062
- // Write metadata and close file handler
1063
- if (fout.is_open()) {
1064
- fout.seekp(0);
1065
- std::vector<uint8_t> data(gguf_get_meta_size(ctx_outs[cur_split].get()));
1066
- gguf_get_meta_data(ctx_outs[cur_split].get(), data.data());
1067
- fout.write((const char *) data.data(), data.size());
1068
- fout.close();
1069
- }
1070
- };
1071
- auto new_ofstream = [&](int index) {
1072
- cur_split = index;
1073
- GGML_ASSERT(ctx_outs[cur_split] && "Find uninitialized gguf_context");
1074
- std::string fname = fname_out;
1075
- if (params->keep_split) {
1076
- std::vector<char> split_path(llama_path_max(), 0);
1077
- llama_split_path(split_path.data(), split_path.size(), fname_out.c_str(), cur_split, n_split);
1078
- fname = std::string(split_path.data());
1079
- }
1080
-
1081
- fout = std::ofstream(fname, std::ios::binary);
1082
- fout.exceptions(std::ofstream::failbit); // fail fast on write errors
1083
- const size_t meta_size = gguf_get_meta_size(ctx_outs[cur_split].get());
1084
- // placeholder for the meta data
1085
- ::zeros(fout, meta_size);
1086
- };
1087
-
1088
- // no output file for --dry-run
1089
- if (!params->dry_run) {
1090
- new_ofstream(0);
1091
- }
1092
-
1093
- //
1094
- // main loop: iterate over all weights
1095
- //
1096
-
1097
- for (size_t i = 0; i < tensors.size(); ++i) {
1098
- const auto & weight = *tensors[i];
1099
- const auto & tm = metadata[i];
1100
- ggml_tensor * tensor = weight.tensor;
1101
-
1102
- if (!params->dry_run && (weight.idx != cur_split && params->keep_split)) {
1103
- close_ofstream();
1104
- new_ofstream(weight.idx);
1105
- }
1106
-
1107
- const std::string name = ggml_get_name(tensor);
1108
- const size_t tensor_size = ggml_nbytes(tensor);
1109
-
1110
- if (!params->dry_run) {
1111
- if (!ml.use_mmap) {
1112
- if (read_data.size() < tensor_size) {
1113
- read_data.resize(tensor_size);
1114
- }
1115
- tensor->data = read_data.data();
1116
- }
1117
- ml.load_data_for(tensor);
1118
- }
1119
-
1120
- LLAMA_LOG_INFO("[%4d/%4d] %-36s - [%s], type = %6s, ",
1121
- ++idx, ml.n_tensors,
1122
- ggml_get_name(tensor),
1123
- llama_format_tensor_shape(tensor).c_str(),
1124
- ggml_type_name(tensor->type));
1125
-
1126
- const ggml_type cur_type = tensor->type;
1127
- const ggml_type new_type = tm.target_type;
1128
-
1129
- // If we've decided to quantize to the same type the tensor is already
1130
- // in then there's nothing to do.
1131
- bool quantize = cur_type != new_type;
1132
-
1133
- void * new_data;
1134
- size_t new_size;
1135
-
1136
- if (params->dry_run) {
1137
- // the --dry-run option calculates the final quantization size without quantizing
1138
- if (quantize) {
1139
- new_size = ggml_nrows(tensor) * ggml_row_size(new_type, tensor->ne[0]);
1140
- LLAMA_LOG_INFO("size = %8.2f MiB -> %8.2f MiB (%s)\n",
1141
- tensor_size/1024.0/1024.0,
1142
- new_size/1024.0/1024.0,
1143
- ggml_type_name(new_type));
1144
- if (!will_require_imatrix && tm.requires_imatrix) {
1145
- will_require_imatrix = true;
1146
- }
1147
- } else {
1148
- new_size = tensor_size;
1149
- LLAMA_LOG_INFO("size = %8.3f MiB\n", new_size/1024.0/1024.0);
1150
- }
1151
- total_size_org += tensor_size;
1152
- total_size_new += new_size;
1153
- continue;
1154
- } else {
1155
- // no --dry-run, perform quantization
1156
- if (!quantize) {
1157
- new_data = tensor->data;
1158
- new_size = tensor_size;
1159
- LLAMA_LOG_INFO("size = %8.3f MiB\n", tensor_size/1024.0/1024.0);
1160
- } else {
1161
- const int64_t nelements = ggml_nelements(tensor);
1162
-
1163
- const float * imatrix = nullptr;
1164
- if (imatrix_data) {
1165
- auto it = imatrix_data->find(tm.remapped_imatrix_name);
1166
- if (it == imatrix_data->end()) {
1167
- LLAMA_LOG_INFO("\n====== %s: did not find weights for %s\n", __func__, tensor->name);
1168
- } else {
1169
- if (it->second.size() == (size_t)tensor->ne[0]*tensor->ne[2]) {
1170
- imatrix = it->second.data();
1171
- } else {
1172
- LLAMA_LOG_INFO("\n====== %s: imatrix size %d is different from tensor size %d for %s\n", __func__,
1173
- int(it->second.size()), int(tensor->ne[0]*tensor->ne[2]), tensor->name);
1174
-
1175
- // this can happen when quantizing an old mixtral model with split tensors with a new incompatible imatrix
1176
- // this is a significant error and it may be good idea to abort the process if this happens,
1177
- // since many people will miss the error and not realize that most of the model is being quantized without an imatrix
1178
- // tok_embd should be ignored in this case, since it always causes this warning
1179
- if (!tensor_name_match_token_embd(tensor->name)) {
1180
- throw std::runtime_error(format("imatrix size %d is different from tensor size %d for %s",
1181
- int(it->second.size()), int(tensor->ne[0]*tensor->ne[2]), tensor->name));
1182
- }
1183
- }
1184
- }
1185
- }
1186
- if (!imatrix && tm.requires_imatrix) {
1187
- LLAMA_LOG_ERROR("\n\n============================================================\n");
1188
- LLAMA_LOG_ERROR("Missing importance matrix for tensor %s in a very low-bit quantization\n", tensor->name);
1189
- LLAMA_LOG_ERROR("The result will be garbage, so bailing out\n");
1190
- LLAMA_LOG_ERROR("============================================================\n\n");
1191
- throw std::runtime_error(format("Missing importance matrix for tensor %s in a very low-bit quantization", tensor->name));
1192
- }
1193
-
1194
- float * f32_data;
1195
-
1196
- if (tensor->type == GGML_TYPE_F32) {
1197
- f32_data = (float *) tensor->data;
1198
- } else if (ggml_is_quantized(tensor->type) && !params->allow_requantize) {
1199
- throw std::runtime_error(format("requantizing from type %s is disabled", ggml_type_name(tensor->type)));
1200
- } else {
1201
- llama_tensor_dequantize_impl(tensor, f32_conv_buf, workers, nelements, nthread);
1202
- f32_data = (float *) f32_conv_buf.data();
1203
- }
1204
-
1205
- LLAMA_LOG_INFO("converting to %s .. ", ggml_type_name(new_type));
1206
- fflush(stdout);
1207
-
1208
- if (work.size() < (size_t)nelements * 4) {
1209
- work.resize(nelements * 4); // upper bound on size
1210
- }
1211
- new_data = work.data();
1212
-
1213
- const int64_t n_per_row = tensor->ne[0];
1214
- const int64_t nrows = tensor->ne[1];
1215
-
1216
- static const int64_t min_chunk_size = 32 * 512;
1217
- const int64_t chunk_size = (n_per_row >= min_chunk_size ? n_per_row : n_per_row * ((min_chunk_size + n_per_row - 1)/n_per_row));
1218
-
1219
- const int64_t nelements_matrix = tensor->ne[0] * tensor->ne[1];
1220
- const int64_t nchunk = (nelements_matrix + chunk_size - 1)/chunk_size;
1221
- const int64_t nthread_use = nthread > 1 ? std::max((int64_t)1, std::min((int64_t)nthread, nchunk)) : 1;
1222
-
1223
- // quantize each expert separately since they have different importance matrices
1224
- new_size = 0;
1225
- for (int64_t i03 = 0; i03 < tensor->ne[2]; ++i03) {
1226
- const float * f32_data_03 = f32_data + i03 * nelements_matrix;
1227
- void * new_data_03 = (char *)new_data + ggml_row_size(new_type, n_per_row) * i03 * nrows;
1228
- const float * imatrix_03 = imatrix ? imatrix + i03 * n_per_row : nullptr;
1229
-
1230
- new_size += llama_tensor_quantize_impl(new_type, f32_data_03, new_data_03, chunk_size, nrows, n_per_row, imatrix_03, workers, nthread_use);
1231
- }
1232
- LLAMA_LOG_INFO("size = %8.2f MiB -> %8.2f MiB\n", tensor_size/1024.0/1024.0, new_size/1024.0/1024.0);
1233
- }
1234
- total_size_org += tensor_size;
1235
- total_size_new += new_size;
1236
-
1237
- // update the gguf meta data as we go
1238
- gguf_set_tensor_type(ctx_outs[cur_split].get(), name.c_str(), new_type);
1239
- GGML_ASSERT(gguf_get_tensor_size(ctx_outs[cur_split].get(), gguf_find_tensor(ctx_outs[cur_split].get(), name.c_str())) == new_size);
1240
- gguf_set_tensor_data(ctx_outs[cur_split].get(), name.c_str(), new_data);
1241
-
1242
- // write tensor data + padding
1243
- fout.write((const char *) new_data, new_size);
1244
- zeros(fout, GGML_PAD(new_size, align) - new_size);
1245
- } // no --dry-run
1246
- } // main loop
1247
-
1248
- if (!params->dry_run) {
1249
- close_ofstream();
1250
- }
1251
-
1252
- LLAMA_LOG_INFO("%s: model size = %8.2f MiB (%.2f BPW)\n", __func__, total_size_org/1024.0/1024.0, total_size_org*8.0/ml.n_elements);
1253
- LLAMA_LOG_INFO("%s: quant size = %8.2f MiB (%.2f BPW)\n", __func__, total_size_new/1024.0/1024.0, total_size_new*8.0/ml.n_elements);
1254
-
1255
- if (!params->imatrix && params->dry_run && will_require_imatrix) {
1256
- LLAMA_LOG_WARN("%s: WARNING: dry run completed successfully, but actually completing this quantization will require an imatrix!\n",
1257
- __func__
1258
- );
1259
- }
1260
-
1261
- if (qs.n_fallback > 0) {
1262
- LLAMA_LOG_WARN("%s: WARNING: %d of %d tensor(s) required fallback quantization\n",
1263
- __func__, qs.n_fallback, ml.n_tensors);
1264
- }
1265
- }
1266
-
1267
- //
1268
- // interface implementation
1269
- //
1270
-
1271
- llama_model_quantize_params llama_model_quantize_default_params() {
1272
- llama_model_quantize_params result = {
1273
- /*.nthread =*/ 0,
1274
- /*.ftype =*/ LLAMA_FTYPE_MOSTLY_Q5_1,
1275
- /*.output_tensor_type =*/ GGML_TYPE_COUNT,
1276
- /*.token_embedding_type =*/ GGML_TYPE_COUNT,
1277
- /*.allow_requantize =*/ false,
1278
- /*.quantize_output_tensor =*/ true,
1279
- /*.only_copy =*/ false,
1280
- /*.pure =*/ false,
1281
- /*.keep_split =*/ false,
1282
- /*.dry_run =*/ false,
1283
- /*.imatrix =*/ nullptr,
1284
- /*.kv_overrides =*/ nullptr,
1285
- /*.tensor_type =*/ nullptr,
1286
- /*.prune_layers =*/ nullptr
1287
- };
1288
-
1289
- return result;
1290
- }
1291
-
1292
- uint32_t llama_model_quantize(
1293
- const char * fname_inp,
1294
- const char * fname_out,
1295
- const llama_model_quantize_params * params) {
1296
- try {
1297
- llama_model_quantize_impl(fname_inp, fname_out, params);
1298
- } catch (const std::exception & err) {
1299
- LLAMA_LOG_ERROR("%s: failed to quantize: %s\n", __func__, err.what());
1300
- return 1;
1301
- }
1302
-
1303
- return 0;
1304
- }