whispercpp 1.3.5 → 1.3.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (1017) hide show
  1. checksums.yaml +4 -4
  2. data/.document +3 -0
  3. data/.rdoc_options +2 -0
  4. data/LICENSE +1 -1
  5. data/README.md +133 -3
  6. data/Rakefile +18 -3
  7. data/ext/dependencies.rb +10 -4
  8. data/ext/dependencies_for_windows.rb +17 -0
  9. data/ext/extconf.rb +20 -7
  10. data/ext/options.rb +54 -14
  11. data/ext/options_for_windows.rb +51 -0
  12. data/ext/ruby_whisper.c +56 -46
  13. data/ext/ruby_whisper.h +165 -2
  14. data/ext/ruby_whisper_context.c +297 -126
  15. data/ext/ruby_whisper_context_params.c +163 -0
  16. data/ext/ruby_whisper_log_queue.c +180 -0
  17. data/ext/ruby_whisper_log_settable.h +47 -0
  18. data/ext/ruby_whisper_model.c +0 -1
  19. data/ext/ruby_whisper_parakeet.c +49 -0
  20. data/ext/ruby_whisper_parakeet_context.c +304 -0
  21. data/ext/ruby_whisper_parakeet_context_params.c +117 -0
  22. data/ext/ruby_whisper_parakeet_model.c +84 -0
  23. data/ext/ruby_whisper_parakeet_params.c +548 -0
  24. data/ext/ruby_whisper_parakeet_segment.c +157 -0
  25. data/ext/ruby_whisper_parakeet_token.c +188 -0
  26. data/ext/ruby_whisper_parakeet_transcribe.cpp +58 -0
  27. data/ext/ruby_whisper_params.c +256 -66
  28. data/ext/ruby_whisper_segment.c +6 -7
  29. data/ext/ruby_whisper_token.c +29 -9
  30. data/ext/ruby_whisper_transcribe.cpp +46 -16
  31. data/ext/ruby_whisper_vad_context.c +48 -1
  32. data/ext/ruby_whisper_vad_context_detect.cpp +6 -5
  33. data/ext/ruby_whisper_vad_params.c +0 -1
  34. data/ext/ruby_whisper_vad_segment.c +0 -1
  35. data/ext/ruby_whisper_vad_segments.c +0 -1
  36. data/ext/sources/CMakeLists.txt +41 -3
  37. data/ext/sources/CMakePresets.json +95 -0
  38. data/ext/sources/cmake/parakeet-config.cmake.in +30 -0
  39. data/ext/sources/cmake/parakeet.pc.in +10 -0
  40. data/ext/sources/cmake/whisper-config.cmake.in +5 -40
  41. data/ext/sources/cmake/whisper.pc.in +1 -1
  42. data/ext/sources/examples/CMakeLists.txt +4 -2
  43. data/ext/sources/examples/bench/bench.cpp +24 -19
  44. data/ext/sources/examples/cli/cli.cpp +51 -9
  45. data/ext/sources/examples/common-ggml.cpp +4 -0
  46. data/ext/sources/examples/common-whisper.cpp +139 -67
  47. data/ext/sources/examples/common-whisper.h +11 -0
  48. data/ext/sources/examples/ffmpeg-transcode.cpp +211 -341
  49. data/ext/sources/examples/miniaudio.h +4507 -2131
  50. data/ext/sources/examples/parakeet-cli/CMakeLists.txt +8 -0
  51. data/ext/sources/examples/parakeet-cli/parakeet-cli.cpp +243 -0
  52. data/ext/sources/examples/parakeet-quantize/CMakeLists.txt +7 -0
  53. data/ext/sources/examples/parakeet-quantize/parakeet-quantize.cpp +230 -0
  54. data/ext/sources/examples/server/server.cpp +213 -163
  55. data/ext/sources/ggml/CMakeLists.txt +29 -15
  56. data/ext/sources/ggml/cmake/FindNCCL.cmake +36 -0
  57. data/ext/sources/ggml/cmake/ggml-config.cmake.in +12 -2
  58. data/ext/sources/ggml/include/ggml-alloc.h +1 -0
  59. data/ext/sources/ggml/include/ggml-backend.h +73 -11
  60. data/ext/sources/ggml/include/ggml-cann.h +1 -1
  61. data/ext/sources/ggml/include/ggml-cpu.h +5 -0
  62. data/ext/sources/ggml/include/ggml-cuda.h +3 -0
  63. data/ext/sources/ggml/include/ggml-openvino.h +37 -0
  64. data/ext/sources/ggml/include/ggml-opt.h +1 -1
  65. data/ext/sources/ggml/include/ggml-rpc.h +8 -3
  66. data/ext/sources/ggml/include/ggml-virtgpu.h +14 -0
  67. data/ext/sources/ggml/include/ggml.h +155 -16
  68. data/ext/sources/ggml/include/gguf.h +10 -2
  69. data/ext/sources/ggml/src/CMakeLists.txt +25 -5
  70. data/ext/sources/ggml/src/ggml-alloc.c +9 -10
  71. data/ext/sources/ggml/src/ggml-backend-dl.cpp +48 -0
  72. data/ext/sources/ggml/src/ggml-backend-dl.h +45 -0
  73. data/ext/sources/ggml/src/ggml-backend-impl.h +22 -2
  74. data/ext/sources/ggml/src/ggml-backend-meta.cpp +2263 -0
  75. data/ext/sources/ggml/src/ggml-backend-reg.cpp +40 -86
  76. data/ext/sources/ggml/src/ggml-backend.cpp +114 -10
  77. data/ext/sources/ggml/src/ggml-blas/CMakeLists.txt +1 -1
  78. data/ext/sources/ggml/src/ggml-blas/ggml-blas.cpp +10 -2
  79. data/ext/sources/ggml/src/ggml-cann/acl_tensor.cpp +1 -1
  80. data/ext/sources/ggml/src/ggml-cann/acl_tensor.h +1 -1
  81. data/ext/sources/ggml/src/ggml-cann/aclnn_ops.cpp +1016 -442
  82. data/ext/sources/ggml/src/ggml-cann/aclnn_ops.h +111 -85
  83. data/ext/sources/ggml/src/ggml-cann/common.h +23 -14
  84. data/ext/sources/ggml/src/ggml-cann/ggml-cann.cpp +255 -92
  85. data/ext/sources/ggml/src/ggml-common.h +22 -0
  86. data/ext/sources/ggml/src/ggml-cpu/CMakeLists.txt +68 -34
  87. data/ext/sources/ggml/src/ggml-cpu/amx/amx.cpp +44 -19
  88. data/ext/sources/ggml/src/ggml-cpu/amx/common.h +34 -10
  89. data/ext/sources/ggml/src/ggml-cpu/amx/mmq.cpp +101 -101
  90. data/ext/sources/ggml/src/ggml-cpu/arch/arm/quants.c +194 -1
  91. data/ext/sources/ggml/src/ggml-cpu/arch/arm/repack.cpp +2874 -613
  92. data/ext/sources/ggml/src/ggml-cpu/arch/loongarch/quants.c +151 -1
  93. data/ext/sources/ggml/src/ggml-cpu/arch/powerpc/quants.c +0 -1
  94. data/ext/sources/ggml/src/ggml-cpu/arch/riscv/quants.c +5480 -840
  95. data/ext/sources/ggml/src/ggml-cpu/arch/riscv/repack.cpp +1361 -0
  96. data/ext/sources/ggml/src/ggml-cpu/arch/s390/quants.c +8 -11
  97. data/ext/sources/ggml/src/ggml-cpu/arch/wasm/quants.c +72 -1
  98. data/ext/sources/ggml/src/ggml-cpu/arch/x86/quants.c +186 -36
  99. data/ext/sources/ggml/src/ggml-cpu/arch/x86/repack.cpp +119 -19
  100. data/ext/sources/ggml/src/ggml-cpu/arch-fallback.h +112 -26
  101. data/ext/sources/ggml/src/ggml-cpu/binary-ops.cpp +2 -6
  102. data/ext/sources/ggml/src/ggml-cpu/cmake/FindSMTIME.cmake +32 -0
  103. data/ext/sources/ggml/src/ggml-cpu/common.h +8 -0
  104. data/ext/sources/ggml/src/ggml-cpu/ggml-cpu-impl.h +13 -0
  105. data/ext/sources/ggml/src/ggml-cpu/ggml-cpu.c +153 -16
  106. data/ext/sources/ggml/src/ggml-cpu/ggml-cpu.cpp +17 -0
  107. data/ext/sources/ggml/src/ggml-cpu/kleidiai/kernels.cpp +21 -20
  108. data/ext/sources/ggml/src/ggml-cpu/kleidiai/kleidiai.cpp +976 -251
  109. data/ext/sources/ggml/src/ggml-cpu/llamafile/sgemm.cpp +671 -266
  110. data/ext/sources/ggml/src/ggml-cpu/ops.cpp +1277 -263
  111. data/ext/sources/ggml/src/ggml-cpu/ops.h +4 -0
  112. data/ext/sources/ggml/src/ggml-cpu/quants.c +95 -0
  113. data/ext/sources/ggml/src/ggml-cpu/quants.h +6 -0
  114. data/ext/sources/ggml/src/ggml-cpu/repack.cpp +2893 -679
  115. data/ext/sources/ggml/src/ggml-cpu/repack.h +119 -8
  116. data/ext/sources/ggml/src/ggml-cpu/simd-gemm.h +226 -0
  117. data/ext/sources/ggml/src/ggml-cpu/simd-mappings.h +114 -19
  118. data/ext/sources/ggml/src/ggml-cpu/spacemit/ime.cpp +1402 -687
  119. data/ext/sources/ggml/src/ggml-cpu/spacemit/ime.h +8 -0
  120. data/ext/sources/ggml/src/ggml-cpu/spacemit/ime1_kernels.cpp +597 -2766
  121. data/ext/sources/ggml/src/ggml-cpu/spacemit/ime2_kernels.cpp +5768 -0
  122. data/ext/sources/ggml/src/ggml-cpu/spacemit/ime_env.cpp +320 -0
  123. data/ext/sources/ggml/src/ggml-cpu/spacemit/ime_env.h +55 -0
  124. data/ext/sources/ggml/src/ggml-cpu/spacemit/ime_kernels.h +182 -19
  125. data/ext/sources/ggml/src/ggml-cpu/spacemit/repack.cpp +1795 -0
  126. data/ext/sources/ggml/src/ggml-cpu/spacemit/repack.h +14 -0
  127. data/ext/sources/ggml/src/ggml-cpu/spacemit/rvv_kernels.cpp +3178 -0
  128. data/ext/sources/ggml/src/ggml-cpu/spacemit/rvv_kernels.h +95 -0
  129. data/ext/sources/ggml/src/ggml-cpu/spacemit/spine_barrier.h +34 -0
  130. data/ext/sources/ggml/src/ggml-cpu/spacemit/spine_mem_pool.cpp +760 -0
  131. data/ext/sources/ggml/src/ggml-cpu/spacemit/spine_mem_pool.h +32 -0
  132. data/ext/sources/ggml/src/ggml-cpu/spacemit/spine_tcm.h +409 -0
  133. data/ext/sources/ggml/src/ggml-cpu/unary-ops.cpp +1 -1
  134. data/ext/sources/ggml/src/ggml-cpu/vec.cpp +54 -53
  135. data/ext/sources/ggml/src/ggml-cpu/vec.h +225 -240
  136. data/ext/sources/ggml/src/ggml-cuda/CMakeLists.txt +18 -8
  137. data/ext/sources/ggml/src/ggml-cuda/allreduce.cu +971 -0
  138. data/ext/sources/ggml/src/ggml-cuda/allreduce.cuh +29 -0
  139. data/ext/sources/ggml/src/ggml-cuda/argsort.cu +73 -28
  140. data/ext/sources/ggml/src/ggml-cuda/binbcast.cu +69 -41
  141. data/ext/sources/ggml/src/ggml-cuda/binbcast.cuh +1 -0
  142. data/ext/sources/ggml/src/ggml-cuda/common.cuh +359 -29
  143. data/ext/sources/ggml/src/ggml-cuda/concat.cu +120 -114
  144. data/ext/sources/ggml/src/ggml-cuda/conv2d-transpose.cu +45 -21
  145. data/ext/sources/ggml/src/ggml-cuda/conv2d-transpose.cuh +1 -0
  146. data/ext/sources/ggml/src/ggml-cuda/convert.cu +94 -27
  147. data/ext/sources/ggml/src/ggml-cuda/convert.cuh +10 -0
  148. data/ext/sources/ggml/src/ggml-cuda/cpy.cu +20 -9
  149. data/ext/sources/ggml/src/ggml-cuda/dequantize.cuh +22 -0
  150. data/ext/sources/ggml/src/ggml-cuda/fattn-common.cuh +333 -85
  151. data/ext/sources/ggml/src/ggml-cuda/fattn-mma-f16.cuh +632 -190
  152. data/ext/sources/ggml/src/ggml-cuda/fattn-tile.cu +12 -0
  153. data/ext/sources/ggml/src/ggml-cuda/fattn-tile.cuh +162 -49
  154. data/ext/sources/ggml/src/ggml-cuda/fattn-vec.cuh +43 -18
  155. data/ext/sources/ggml/src/ggml-cuda/fattn-wmma-f16.cu +44 -14
  156. data/ext/sources/ggml/src/ggml-cuda/fattn-wmma-f16.cuh +1 -1
  157. data/ext/sources/ggml/src/ggml-cuda/fattn.cu +241 -23
  158. data/ext/sources/ggml/src/ggml-cuda/fattn.cuh +2 -0
  159. data/ext/sources/ggml/src/ggml-cuda/fwht.cu +101 -0
  160. data/ext/sources/ggml/src/ggml-cuda/fwht.cuh +4 -0
  161. data/ext/sources/ggml/src/ggml-cuda/gated_delta_net.cu +312 -0
  162. data/ext/sources/ggml/src/ggml-cuda/gated_delta_net.cuh +4 -0
  163. data/ext/sources/ggml/src/ggml-cuda/getrows.cu +34 -12
  164. data/ext/sources/ggml/src/ggml-cuda/ggml-cuda.cu +1454 -599
  165. data/ext/sources/ggml/src/ggml-cuda/im2col.cu +32 -29
  166. data/ext/sources/ggml/src/ggml-cuda/mean.cu +13 -10
  167. data/ext/sources/ggml/src/ggml-cuda/mma.cuh +397 -183
  168. data/ext/sources/ggml/src/ggml-cuda/mmf.cu +30 -10
  169. data/ext/sources/ggml/src/ggml-cuda/mmf.cuh +161 -88
  170. data/ext/sources/ggml/src/ggml-cuda/mmq.cu +18 -12
  171. data/ext/sources/ggml/src/ggml-cuda/mmq.cuh +522 -431
  172. data/ext/sources/ggml/src/ggml-cuda/mmvf.cu +139 -72
  173. data/ext/sources/ggml/src/ggml-cuda/mmvf.cuh +2 -0
  174. data/ext/sources/ggml/src/ggml-cuda/mmvq.cu +608 -88
  175. data/ext/sources/ggml/src/ggml-cuda/mmvq.cuh +6 -0
  176. data/ext/sources/ggml/src/ggml-cuda/norm.cu +47 -79
  177. data/ext/sources/ggml/src/ggml-cuda/out-prod.cu +23 -7
  178. data/ext/sources/ggml/src/ggml-cuda/pad.cu +13 -10
  179. data/ext/sources/ggml/src/ggml-cuda/quantize.cu +134 -27
  180. data/ext/sources/ggml/src/ggml-cuda/quantize.cuh +1 -1
  181. data/ext/sources/ggml/src/ggml-cuda/reduce_rows.cuh +7 -17
  182. data/ext/sources/ggml/src/ggml-cuda/rope.cu +244 -137
  183. data/ext/sources/ggml/src/ggml-cuda/scale.cu +4 -1
  184. data/ext/sources/ggml/src/ggml-cuda/set-rows.cu +14 -6
  185. data/ext/sources/ggml/src/ggml-cuda/snake.cu +72 -0
  186. data/ext/sources/ggml/src/ggml-cuda/snake.cuh +8 -0
  187. data/ext/sources/ggml/src/ggml-cuda/softcap.cu +4 -1
  188. data/ext/sources/ggml/src/ggml-cuda/softmax.cu +8 -83
  189. data/ext/sources/ggml/src/ggml-cuda/solve_tri.cu +1 -1
  190. data/ext/sources/ggml/src/ggml-cuda/ssm-conv.cu +96 -40
  191. data/ext/sources/ggml/src/ggml-cuda/ssm-conv.cuh +1 -1
  192. data/ext/sources/ggml/src/ggml-cuda/ssm-scan.cu +40 -18
  193. data/ext/sources/ggml/src/ggml-cuda/sumrows.cu +8 -4
  194. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_1-ncols2_16.cu +1 -0
  195. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_1-ncols2_32.cu +6 -0
  196. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_1-ncols2_8.cu +2 -0
  197. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_16-ncols2_4.cu +2 -0
  198. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_2-ncols2_16.cu +1 -0
  199. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_2-ncols2_32.cu +6 -0
  200. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_2-ncols2_4.cu +2 -0
  201. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_2-ncols2_8.cu +2 -0
  202. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_4-ncols2_16.cu +1 -0
  203. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_4-ncols2_4.cu +2 -0
  204. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_4-ncols2_8.cu +2 -0
  205. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_8-ncols2_4.cu +2 -0
  206. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-mma-f16-instance-ncols1_8-ncols2_8.cu +2 -0
  207. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-tile-instance-dkq192-dv128.cu +5 -0
  208. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-tile-instance-dkq320-dv256.cu +5 -0
  209. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-tile-instance-dkq512-dv512.cu +5 -0
  210. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-bf16-bf16.cu +7 -0
  211. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-bf16-f16.cu +7 -0
  212. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-bf16-q4_0.cu +7 -0
  213. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-bf16-q4_1.cu +7 -0
  214. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-bf16-q5_0.cu +7 -0
  215. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-bf16-q5_1.cu +7 -0
  216. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-bf16-q8_0.cu +7 -0
  217. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-f16-bf16.cu +7 -0
  218. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q4_0-bf16.cu +7 -0
  219. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q4_1-bf16.cu +7 -0
  220. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q5_0-bf16.cu +7 -0
  221. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q5_1-bf16.cu +7 -0
  222. data/ext/sources/ggml/src/ggml-cuda/template-instances/fattn-vec-instance-q8_0-bf16.cu +7 -0
  223. data/ext/sources/ggml/src/ggml-cuda/template-instances/mmq-instance-nvfp4.cu +5 -0
  224. data/ext/sources/ggml/src/ggml-cuda/template-instances/mmq-instance-q1_0.cu +5 -0
  225. data/ext/sources/ggml/src/ggml-cuda/top-k.cu +5 -5
  226. data/ext/sources/ggml/src/ggml-cuda/topk-moe.cu +202 -135
  227. data/ext/sources/ggml/src/ggml-cuda/topk-moe.cuh +20 -14
  228. data/ext/sources/ggml/src/ggml-cuda/unary.cu +86 -2
  229. data/ext/sources/ggml/src/ggml-cuda/unary.cuh +4 -0
  230. data/ext/sources/ggml/src/ggml-cuda/vecdotq.cuh +111 -17
  231. data/ext/sources/ggml/src/ggml-cuda/vendors/cuda.h +7 -2
  232. data/ext/sources/ggml/src/ggml-cuda/vendors/hip.h +30 -2
  233. data/ext/sources/ggml/src/ggml-cuda/vendors/musa.h +3 -0
  234. data/ext/sources/ggml/src/ggml-hexagon/CMakeLists.txt +84 -46
  235. data/ext/sources/ggml/src/ggml-hexagon/ggml-hexagon.cpp +1612 -753
  236. data/ext/sources/ggml/src/ggml-hexagon/htp/CMakeLists.txt +51 -11
  237. data/ext/sources/ggml/src/ggml-hexagon/htp/act-ops.c +361 -261
  238. data/ext/sources/ggml/src/ggml-hexagon/htp/argsort-ops.c +294 -0
  239. data/ext/sources/ggml/src/ggml-hexagon/htp/binary-ops.c +753 -241
  240. data/ext/sources/ggml/src/ggml-hexagon/htp/cmake-toolchain.cmake +5 -5
  241. data/ext/sources/ggml/src/ggml-hexagon/htp/concat-ops.c +277 -0
  242. data/ext/sources/ggml/src/ggml-hexagon/htp/cpy-ops.c +295 -0
  243. data/ext/sources/ggml/src/ggml-hexagon/htp/cumsum-ops.c +270 -0
  244. data/ext/sources/ggml/src/ggml-hexagon/htp/diag-ops.c +216 -0
  245. data/ext/sources/ggml/src/ggml-hexagon/htp/fill-ops.c +123 -0
  246. data/ext/sources/ggml/src/ggml-hexagon/htp/flash-attn-ops.c +471 -296
  247. data/ext/sources/ggml/src/ggml-hexagon/htp/gated-delta-net-ops.c +1148 -0
  248. data/ext/sources/ggml/src/ggml-hexagon/htp/get-rows-ops.c +159 -53
  249. data/ext/sources/ggml/src/ggml-hexagon/htp/{htp-dma.c → hex-dma.c} +3 -3
  250. data/ext/sources/ggml/src/ggml-hexagon/htp/hex-dma.h +372 -0
  251. data/ext/sources/ggml/src/ggml-hexagon/htp/hex-dump.h +86 -0
  252. data/ext/sources/ggml/src/ggml-hexagon/htp/hex-fastdiv.h +37 -0
  253. data/ext/sources/ggml/src/ggml-hexagon/htp/hex-utils.h +137 -0
  254. data/ext/sources/ggml/src/ggml-hexagon/htp/hmx-flash-attn-ops.c +1878 -0
  255. data/ext/sources/ggml/src/ggml-hexagon/htp/hmx-matmul-ops.c +2066 -0
  256. data/ext/sources/ggml/src/ggml-hexagon/htp/hmx-ops.c +6 -0
  257. data/ext/sources/ggml/src/ggml-hexagon/htp/hmx-ops.h +88 -0
  258. data/ext/sources/ggml/src/ggml-hexagon/htp/hmx-profile.h +34 -0
  259. data/ext/sources/ggml/src/ggml-hexagon/htp/hmx-queue.c +158 -0
  260. data/ext/sources/ggml/src/ggml-hexagon/htp/hmx-queue.h +134 -0
  261. data/ext/sources/ggml/src/ggml-hexagon/htp/hmx-utils.h +200 -0
  262. data/ext/sources/ggml/src/ggml-hexagon/htp/htp-ctx.h +97 -14
  263. data/ext/sources/ggml/src/ggml-hexagon/htp/htp-ops.h +163 -67
  264. data/ext/sources/ggml/src/ggml-hexagon/htp/htp_iface.idl +9 -3
  265. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-arith.h +443 -0
  266. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-base.h +308 -0
  267. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-copy.h +262 -0
  268. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-div.h +291 -0
  269. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-dump.h +129 -0
  270. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-exp.h +216 -0
  271. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-flash-attn.h +47 -0
  272. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-floor.h +100 -0
  273. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-inverse.h +210 -0
  274. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-log.h +65 -0
  275. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-pow.h +42 -0
  276. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-reduce.h +296 -0
  277. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-repl.h +74 -0
  278. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-scale.h +133 -0
  279. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-sigmoid.h +142 -0
  280. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-sin-cos.h +90 -0
  281. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-sqrt.h +126 -0
  282. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-types.h +36 -0
  283. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-utils.h +18 -1348
  284. data/ext/sources/ggml/src/ggml-hexagon/htp/main.c +547 -635
  285. data/ext/sources/ggml/src/ggml-hexagon/htp/matmul-ops.c +3556 -1101
  286. data/ext/sources/ggml/src/ggml-hexagon/htp/pad-ops.c +547 -0
  287. data/ext/sources/ggml/src/ggml-hexagon/htp/repeat-ops.c +148 -0
  288. data/ext/sources/ggml/src/ggml-hexagon/htp/rope-ops.c +475 -269
  289. data/ext/sources/ggml/src/ggml-hexagon/htp/set-rows-ops.c +94 -72
  290. data/ext/sources/ggml/src/ggml-hexagon/htp/softmax-ops.c +222 -217
  291. data/ext/sources/ggml/src/ggml-hexagon/htp/solve-tri-ops.c +267 -0
  292. data/ext/sources/ggml/src/ggml-hexagon/htp/ssm-conv.c +432 -0
  293. data/ext/sources/ggml/src/ggml-hexagon/htp/sum-rows-ops.c +128 -0
  294. data/ext/sources/ggml/src/ggml-hexagon/htp/unary-ops.c +886 -117
  295. data/ext/sources/ggml/src/ggml-hexagon/htp/vtcm-utils.h +16 -0
  296. data/ext/sources/ggml/src/ggml-hexagon/htp/worker-pool.c +1 -5
  297. data/ext/sources/ggml/src/ggml-hexagon/htp-drv.cpp +418 -0
  298. data/ext/sources/ggml/src/ggml-hexagon/htp-drv.h +121 -0
  299. data/ext/sources/ggml/src/ggml-hexagon/htp-opnode.h +272 -0
  300. data/ext/sources/ggml/src/ggml-hexagon/libdl.h +79 -0
  301. data/ext/sources/ggml/src/ggml-hexagon/libggml-htp.inf +40 -0
  302. data/ext/sources/ggml/src/ggml-hip/CMakeLists.txt +28 -9
  303. data/ext/sources/ggml/src/ggml-impl.h +68 -1
  304. data/ext/sources/ggml/src/ggml-metal/CMakeLists.txt +10 -10
  305. data/ext/sources/ggml/src/ggml-metal/ggml-metal-common.cpp +13 -2
  306. data/ext/sources/ggml/src/ggml-metal/ggml-metal-context.h +8 -0
  307. data/ext/sources/ggml/src/ggml-metal/ggml-metal-context.m +147 -17
  308. data/ext/sources/ggml/src/ggml-metal/ggml-metal-device.cpp +409 -83
  309. data/ext/sources/ggml/src/ggml-metal/ggml-metal-device.h +54 -5
  310. data/ext/sources/ggml/src/ggml-metal/ggml-metal-device.m +254 -52
  311. data/ext/sources/ggml/src/ggml-metal/ggml-metal-impl.h +254 -23
  312. data/ext/sources/ggml/src/ggml-metal/ggml-metal-ops.cpp +756 -285
  313. data/ext/sources/ggml/src/ggml-metal/ggml-metal-ops.h +7 -4
  314. data/ext/sources/ggml/src/ggml-metal/ggml-metal.cpp +359 -133
  315. data/ext/sources/ggml/src/ggml-metal/ggml-metal.metal +1867 -1123
  316. data/ext/sources/ggml/src/ggml-musa/CMakeLists.txt +5 -6
  317. data/ext/sources/ggml/src/ggml-opencl/CMakeLists.txt +71 -4
  318. data/ext/sources/ggml/src/ggml-opencl/ggml-opencl.cpp +14127 -5314
  319. data/ext/sources/ggml/src/ggml-opencl/kernels/concat.cl +97 -88
  320. data/ext/sources/ggml/src/ggml-opencl/kernels/cpy.cl +104 -0
  321. data/ext/sources/ggml/src/ggml-opencl/kernels/cumsum.cl +139 -0
  322. data/ext/sources/ggml/src/ggml-opencl/kernels/cvt.cl +1978 -67
  323. data/ext/sources/ggml/src/ggml-opencl/kernels/diag.cl +27 -0
  324. data/ext/sources/ggml/src/ggml-opencl/kernels/exp.cl +125 -0
  325. data/ext/sources/ggml/src/ggml-opencl/kernels/expm1.cl +87 -56
  326. data/ext/sources/ggml/src/ggml-opencl/kernels/gated_delta_net.cl +249 -0
  327. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_mxfp4_f32_ns.cl +306 -0
  328. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_q4_0_f32_ns.cl +256 -0
  329. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_q4_1_f32_ns.cl +258 -0
  330. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_q4_k_f32_ns.cl +283 -0
  331. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_q5_0_f32_ns.cl +260 -0
  332. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_q5_1_f32_ns.cl +262 -0
  333. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_q5_k_f32_ns.cl +288 -0
  334. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_moe_q6_k_f32_ns.cl +267 -0
  335. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_iq4_nl_f32.cl +150 -0
  336. data/ext/sources/ggml/src/ggml-opencl/kernels/{mul_mat_Ab_Bi_8x4.cl → gemm_noshuffle_q4_0_f32.cl} +1 -1
  337. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q4_1_f32.cl +132 -0
  338. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q4_k_f32.cl +172 -0
  339. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q5_0_f32.cl +131 -0
  340. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q5_1_f32.cl +134 -0
  341. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q5_k_f32.cl +176 -0
  342. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q6_k_f32.cl +140 -0
  343. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_noshuffle_q8_0_f32.cl +129 -0
  344. data/ext/sources/ggml/src/ggml-opencl/kernels/gemm_xmem_f16_f32_os8.cl +233 -0
  345. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_mxfp4_f32_ns.cl +165 -0
  346. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_q4_0_f32_ns.cl +120 -0
  347. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_q4_1_f32_ns.cl +123 -0
  348. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_q4_k_f32_ns.cl +155 -0
  349. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_q5_0_f32_ns.cl +123 -0
  350. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_q5_1_f32_ns.cl +125 -0
  351. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_q5_k_f32_ns.cl +160 -0
  352. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_moe_q6_k_f32_ns.cl +141 -0
  353. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_iq4_nl_f32.cl +302 -0
  354. data/ext/sources/ggml/src/ggml-opencl/kernels/{gemv_noshuffle_general.cl → gemv_noshuffle_q4_0_f32.cl} +5 -5
  355. data/ext/sources/ggml/src/ggml-opencl/kernels/{gemv_noshuffle.cl → gemv_noshuffle_q4_0_f32_spec.cl} +5 -5
  356. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_1_f32.cl +283 -0
  357. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q4_k_f32.cl +318 -0
  358. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q5_0_f32.cl +291 -0
  359. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q5_1_f32.cl +294 -0
  360. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q5_k_f32.cl +326 -0
  361. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q6_k_f32.cl +293 -0
  362. data/ext/sources/ggml/src/ggml-opencl/kernels/gemv_noshuffle_q8_0_f32.cl +195 -0
  363. data/ext/sources/ggml/src/ggml-opencl/kernels/get_rows.cl +15 -9
  364. data/ext/sources/ggml/src/ggml-opencl/kernels/l2_norm.cl +71 -0
  365. data/ext/sources/ggml/src/ggml-opencl/kernels/mean.cl +114 -13
  366. data/ext/sources/ggml/src/ggml-opencl/kernels/moe_reorder_b.cl +30 -0
  367. data/ext/sources/ggml/src/ggml-opencl/kernels/moe_sort_by_expert.cl +82 -0
  368. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mm_iq4_nl_f32_l4_lm.cl +171 -0
  369. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mm_q4_0_f32_l4_lm.cl +163 -0
  370. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mm_q4_1_f32_l4_lm.cl +165 -0
  371. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mm_q4_k_f32_l4_lm.cl +179 -0
  372. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mm_q5_0_f32_l4_lm.cl +173 -0
  373. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mm_q5_1_f32_l4_lm.cl +175 -0
  374. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mm_q5_k_f32_l4_lm.cl +192 -0
  375. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mm_q6_k_f32_l4_lm.cl +158 -0
  376. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_iq4_nl_f32.cl +164 -0
  377. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_iq4_nl_f32_flat.cl +202 -0
  378. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q4_1_f32.cl +219 -0
  379. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q4_1_f32_flat.cl +229 -0
  380. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q4_k_f32.cl +180 -0
  381. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q4_k_f32_flat.cl +196 -0
  382. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q5_0_f32.cl +241 -0
  383. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q5_0_f32_flat.cl +243 -0
  384. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q5_1_f32.cl +243 -0
  385. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q5_1_f32_flat.cl +247 -0
  386. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q5_k_f32.cl +187 -0
  387. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q5_k_f32_flat.cl +203 -0
  388. data/ext/sources/ggml/src/ggml-opencl/kernels/{mul_mv_q6_k.cl → mul_mv_q6_k_f32.cl} +4 -0
  389. data/ext/sources/ggml/src/ggml-opencl/kernels/mul_mv_q6_k_f32_flat.cl +178 -0
  390. data/ext/sources/ggml/src/ggml-opencl/kernels/neg.cl +125 -0
  391. data/ext/sources/ggml/src/ggml-opencl/kernels/repeat.cl +31 -32
  392. data/ext/sources/ggml/src/ggml-opencl/kernels/scale.cl +14 -4
  393. data/ext/sources/ggml/src/ggml-opencl/kernels/softplus.cl +88 -60
  394. data/ext/sources/ggml/src/ggml-opencl/kernels/solve_tri.cl +51 -0
  395. data/ext/sources/ggml/src/ggml-opencl/kernels/sum_rows.cl +114 -13
  396. data/ext/sources/ggml/src/ggml-opencl/kernels/tanh.cl +94 -48
  397. data/ext/sources/ggml/src/ggml-opencl/kernels/transpose.cl +26 -0
  398. data/ext/sources/ggml/src/ggml-opencl/kernels/tri.cl +32 -0
  399. data/ext/sources/ggml/src/ggml-openvino/.clang-format +154 -0
  400. data/ext/sources/ggml/src/ggml-openvino/CMakeLists.txt +22 -0
  401. data/ext/sources/ggml/src/ggml-openvino/ggml-decoder.cpp +985 -0
  402. data/ext/sources/ggml/src/ggml-openvino/ggml-decoder.h +294 -0
  403. data/ext/sources/ggml/src/ggml-openvino/ggml-openvino-extra.cpp +380 -0
  404. data/ext/sources/ggml/src/ggml-openvino/ggml-openvino-extra.h +182 -0
  405. data/ext/sources/ggml/src/ggml-openvino/ggml-openvino.cpp +1132 -0
  406. data/ext/sources/ggml/src/ggml-openvino/ggml-quants.cpp +956 -0
  407. data/ext/sources/ggml/src/ggml-openvino/ggml-quants.h +153 -0
  408. data/ext/sources/ggml/src/ggml-openvino/openvino/decoder.h +74 -0
  409. data/ext/sources/ggml/src/ggml-openvino/openvino/frontend.cpp +27 -0
  410. data/ext/sources/ggml/src/ggml-openvino/openvino/frontend.h +23 -0
  411. data/ext/sources/ggml/src/ggml-openvino/openvino/input_model.cpp +17 -0
  412. data/ext/sources/ggml/src/ggml-openvino/openvino/input_model.h +29 -0
  413. data/ext/sources/ggml/src/ggml-openvino/openvino/node_context.h +112 -0
  414. data/ext/sources/ggml/src/ggml-openvino/openvino/op/cont.cpp +48 -0
  415. data/ext/sources/ggml/src/ggml-openvino/openvino/op/cpy.cpp +21 -0
  416. data/ext/sources/ggml/src/ggml-openvino/openvino/op/flash_attn_ext.cpp +90 -0
  417. data/ext/sources/ggml/src/ggml-openvino/openvino/op/get_rows.cpp +69 -0
  418. data/ext/sources/ggml/src/ggml-openvino/openvino/op/glu_geglu.cpp +61 -0
  419. data/ext/sources/ggml/src/ggml-openvino/openvino/op/glu_swiglu.cpp +62 -0
  420. data/ext/sources/ggml/src/ggml-openvino/openvino/op/mulmat.cpp +90 -0
  421. data/ext/sources/ggml/src/ggml-openvino/openvino/op/permute.cpp +102 -0
  422. data/ext/sources/ggml/src/ggml-openvino/openvino/op/reshape.cpp +83 -0
  423. data/ext/sources/ggml/src/ggml-openvino/openvino/op/rms_norm.cpp +46 -0
  424. data/ext/sources/ggml/src/ggml-openvino/openvino/op/rope.cpp +149 -0
  425. data/ext/sources/ggml/src/ggml-openvino/openvino/op/scale.cpp +41 -0
  426. data/ext/sources/ggml/src/ggml-openvino/openvino/op/set_rows.cpp +76 -0
  427. data/ext/sources/ggml/src/ggml-openvino/openvino/op/softmax.cpp +89 -0
  428. data/ext/sources/ggml/src/ggml-openvino/openvino/op/transpose.cpp +23 -0
  429. data/ext/sources/ggml/src/ggml-openvino/openvino/op/unary_gelu.cpp +25 -0
  430. data/ext/sources/ggml/src/ggml-openvino/openvino/op/unary_silu.cpp +27 -0
  431. data/ext/sources/ggml/src/ggml-openvino/openvino/op/view.cpp +53 -0
  432. data/ext/sources/ggml/src/ggml-openvino/openvino/op_table.cpp +47 -0
  433. data/ext/sources/ggml/src/ggml-openvino/openvino/op_table.h +40 -0
  434. data/ext/sources/ggml/src/ggml-openvino/openvino/pass/fuse_to_sdpa.cpp +60 -0
  435. data/ext/sources/ggml/src/ggml-openvino/openvino/pass/fuse_to_sdpa.h +17 -0
  436. data/ext/sources/ggml/src/ggml-openvino/openvino/pass/mark_decompression_convert_constant_folding.h +29 -0
  437. data/ext/sources/ggml/src/ggml-openvino/openvino/pass/squeeze_matmul.cpp +58 -0
  438. data/ext/sources/ggml/src/ggml-openvino/openvino/pass/squeeze_matmul.h +17 -0
  439. data/ext/sources/ggml/src/ggml-openvino/openvino/rt_info/weightless_caching_attributes.hpp +41 -0
  440. data/ext/sources/ggml/src/ggml-openvino/openvino/translate_session.cpp +317 -0
  441. data/ext/sources/ggml/src/ggml-openvino/openvino/translate_session.h +28 -0
  442. data/ext/sources/ggml/src/ggml-openvino/openvino/utils.cpp +257 -0
  443. data/ext/sources/ggml/src/ggml-openvino/openvino/utils.h +86 -0
  444. data/ext/sources/ggml/src/ggml-openvino/utils.cpp +880 -0
  445. data/ext/sources/ggml/src/ggml-openvino/utils.h +143 -0
  446. data/ext/sources/ggml/src/ggml-opt.cpp +1 -0
  447. data/ext/sources/ggml/src/ggml-quants.c +385 -119
  448. data/ext/sources/ggml/src/ggml-quants.h +6 -0
  449. data/ext/sources/ggml/src/ggml-rpc/CMakeLists.txt +24 -0
  450. data/ext/sources/ggml/src/ggml-rpc/ggml-rpc.cpp +167 -311
  451. data/ext/sources/ggml/src/ggml-rpc/transport.cpp +683 -0
  452. data/ext/sources/ggml/src/ggml-rpc/transport.h +34 -0
  453. data/ext/sources/ggml/src/ggml-sycl/CMakeLists.txt +64 -91
  454. data/ext/sources/ggml/src/ggml-sycl/add-id.cpp +5 -1
  455. data/ext/sources/ggml/src/ggml-sycl/backend.hpp +4 -1
  456. data/ext/sources/ggml/src/ggml-sycl/binbcast.cpp +21 -20
  457. data/ext/sources/ggml/src/ggml-sycl/common.cpp +74 -2
  458. data/ext/sources/ggml/src/ggml-sycl/common.hpp +356 -11
  459. data/ext/sources/ggml/src/ggml-sycl/convert.cpp +184 -14
  460. data/ext/sources/ggml/src/ggml-sycl/convert.hpp +31 -1
  461. data/ext/sources/ggml/src/ggml-sycl/count-equal.cpp +1 -1
  462. data/ext/sources/ggml/src/ggml-sycl/cumsum.cpp +148 -0
  463. data/ext/sources/ggml/src/ggml-sycl/cumsum.hpp +5 -0
  464. data/ext/sources/ggml/src/ggml-sycl/dequantize.hpp +663 -0
  465. data/ext/sources/ggml/src/ggml-sycl/diag.cpp +67 -0
  466. data/ext/sources/ggml/src/ggml-sycl/diag.hpp +5 -0
  467. data/ext/sources/ggml/src/ggml-sycl/dmmv.cpp +586 -6
  468. data/ext/sources/ggml/src/ggml-sycl/dpct/helper.hpp +791 -47
  469. data/ext/sources/ggml/src/ggml-sycl/element_wise.cpp +77 -156
  470. data/ext/sources/ggml/src/ggml-sycl/element_wise.hpp +2 -2
  471. data/ext/sources/ggml/src/ggml-sycl/fattn-buffers.cpp +56 -0
  472. data/ext/sources/ggml/src/ggml-sycl/fattn-buffers.hpp +63 -0
  473. data/ext/sources/ggml/src/ggml-sycl/fattn-common.hpp +1181 -0
  474. data/ext/sources/ggml/src/ggml-sycl/fattn-tile.cpp +59 -0
  475. data/ext/sources/ggml/src/ggml-sycl/fattn-tile.hpp +1246 -0
  476. data/ext/sources/ggml/src/ggml-sycl/fattn-vec.hpp +674 -0
  477. data/ext/sources/ggml/src/ggml-sycl/fattn.cpp +227 -0
  478. data/ext/sources/ggml/src/ggml-sycl/fattn.hpp +22 -0
  479. data/ext/sources/ggml/src/ggml-sycl/fill.cpp +55 -0
  480. data/ext/sources/ggml/src/ggml-sycl/fill.hpp +5 -0
  481. data/ext/sources/ggml/src/ggml-sycl/gated_delta_net.cpp +347 -0
  482. data/ext/sources/ggml/src/ggml-sycl/gated_delta_net.hpp +9 -0
  483. data/ext/sources/ggml/src/ggml-sycl/gemm.hpp +3 -0
  484. data/ext/sources/ggml/src/ggml-sycl/getrows.cpp +79 -3
  485. data/ext/sources/ggml/src/ggml-sycl/ggml-sycl.cpp +1134 -236
  486. data/ext/sources/ggml/src/ggml-sycl/im2col.cpp +353 -89
  487. data/ext/sources/ggml/src/ggml-sycl/im2col.hpp +5 -3
  488. data/ext/sources/ggml/src/ggml-sycl/mmvq.cpp +1344 -26
  489. data/ext/sources/ggml/src/ggml-sycl/mmvq.hpp +16 -0
  490. data/ext/sources/ggml/src/ggml-sycl/norm.cpp +65 -66
  491. data/ext/sources/ggml/src/ggml-sycl/outprod.cpp +3 -3
  492. data/ext/sources/ggml/src/ggml-sycl/pad.cpp +27 -27
  493. data/ext/sources/ggml/src/ggml-sycl/presets.hpp +3 -0
  494. data/ext/sources/ggml/src/ggml-sycl/quants.hpp +72 -1
  495. data/ext/sources/ggml/src/ggml-sycl/rope.cpp +450 -287
  496. data/ext/sources/ggml/src/ggml-sycl/rope.hpp +6 -0
  497. data/ext/sources/ggml/src/ggml-sycl/set_rows.cpp +7 -1
  498. data/ext/sources/ggml/src/ggml-sycl/softmax.cpp +6 -6
  499. data/ext/sources/ggml/src/ggml-sycl/solve_tri.cpp +172 -0
  500. data/ext/sources/ggml/src/ggml-sycl/solve_tri.hpp +8 -0
  501. data/ext/sources/ggml/src/ggml-sycl/ssm_conv.cpp +6 -1
  502. data/ext/sources/ggml/src/ggml-sycl/ssm_scan.cpp +156 -0
  503. data/ext/sources/ggml/src/ggml-sycl/ssm_scan.hpp +5 -0
  504. data/ext/sources/ggml/src/ggml-sycl/sycl_hw.cpp +62 -10
  505. data/ext/sources/ggml/src/ggml-sycl/sycl_hw.hpp +18 -6
  506. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-tile-instance-dkq112-dv112.cpp +5 -0
  507. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-tile-instance-dkq128-dv128.cpp +5 -0
  508. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-tile-instance-dkq256-dv256.cpp +5 -0
  509. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-tile-instance-dkq40-dv40.cpp +5 -0
  510. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-tile-instance-dkq512-dv512.cpp +6 -0
  511. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-tile-instance-dkq576-dv512.cpp +5 -0
  512. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-tile-instance-dkq64-dv64.cpp +5 -0
  513. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-tile-instance-dkq72-dv72.cpp +5 -0
  514. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-tile-instance-dkq80-dv80.cpp +5 -0
  515. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-tile-instance-dkq96-dv96.cpp +5 -0
  516. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-f16-f16.cpp +8 -0
  517. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-f16-q4_0.cpp +8 -0
  518. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-f16-q4_1.cpp +8 -0
  519. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-f16-q5_0.cpp +8 -0
  520. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-f16-q5_1.cpp +8 -0
  521. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-f16-q8_0.cpp +8 -0
  522. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_0-f16.cpp +8 -0
  523. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_0-q4_0.cpp +8 -0
  524. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_0-q4_1.cpp +8 -0
  525. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_0-q5_0.cpp +8 -0
  526. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_0-q5_1.cpp +8 -0
  527. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_0-q8_0.cpp +8 -0
  528. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_1-f16.cpp +8 -0
  529. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_1-q4_0.cpp +8 -0
  530. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_1-q4_1.cpp +8 -0
  531. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_1-q5_0.cpp +8 -0
  532. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_1-q5_1.cpp +8 -0
  533. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q4_1-q8_0.cpp +8 -0
  534. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_0-f16.cpp +8 -0
  535. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_0-q4_0.cpp +8 -0
  536. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_0-q4_1.cpp +8 -0
  537. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_0-q5_0.cpp +8 -0
  538. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_0-q5_1.cpp +8 -0
  539. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_0-q8_0.cpp +8 -0
  540. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_1-f16.cpp +8 -0
  541. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_1-q4_0.cpp +8 -0
  542. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_1-q4_1.cpp +8 -0
  543. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_1-q5_0.cpp +8 -0
  544. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_1-q5_1.cpp +8 -0
  545. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q5_1-q8_0.cpp +8 -0
  546. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q8_0-f16.cpp +8 -0
  547. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q8_0-q4_0.cpp +8 -0
  548. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q8_0-q4_1.cpp +8 -0
  549. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q8_0-q5_0.cpp +8 -0
  550. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q8_0-q5_1.cpp +8 -0
  551. data/ext/sources/ggml/src/ggml-sycl/template-instances/fattn-vec-instance-q8_0-q8_0.cpp +8 -0
  552. data/ext/sources/ggml/src/ggml-sycl/type.hpp +112 -0
  553. data/ext/sources/ggml/src/ggml-sycl/upscale.cpp +410 -0
  554. data/ext/sources/ggml/src/ggml-sycl/upscale.hpp +9 -0
  555. data/ext/sources/ggml/src/ggml-sycl/vecdotq.hpp +228 -53
  556. data/ext/sources/ggml/src/ggml-sycl/wkv.cpp +1 -1
  557. data/ext/sources/ggml/src/ggml-virtgpu/CMakeLists.txt +70 -0
  558. data/ext/sources/ggml/src/ggml-virtgpu/apir_cs_ggml-rpc-front.cpp +87 -0
  559. data/ext/sources/ggml/src/ggml-virtgpu/backend/CMakeLists.txt +21 -0
  560. data/ext/sources/ggml/src/ggml-virtgpu/backend/apir_cs_ggml-rpc-back.cpp +115 -0
  561. data/ext/sources/ggml/src/ggml-virtgpu/backend/backend-convert.h +13 -0
  562. data/ext/sources/ggml/src/ggml-virtgpu/backend/backend-dispatched-backend.cpp +102 -0
  563. data/ext/sources/ggml/src/ggml-virtgpu/backend/backend-dispatched-buffer-type.cpp +105 -0
  564. data/ext/sources/ggml/src/ggml-virtgpu/backend/backend-dispatched-buffer.cpp +179 -0
  565. data/ext/sources/ggml/src/ggml-virtgpu/backend/backend-dispatched-device.cpp +148 -0
  566. data/ext/sources/ggml/src/ggml-virtgpu/backend/backend-dispatched.cpp +51 -0
  567. data/ext/sources/ggml/src/ggml-virtgpu/backend/backend-dispatched.gen.h +73 -0
  568. data/ext/sources/ggml/src/ggml-virtgpu/backend/backend-dispatched.h +27 -0
  569. data/ext/sources/ggml/src/ggml-virtgpu/backend/backend-virgl-apir.h +32 -0
  570. data/ext/sources/ggml/src/ggml-virtgpu/backend/backend.cpp +144 -0
  571. data/ext/sources/ggml/src/ggml-virtgpu/backend/shared/api_remoting.h +95 -0
  572. data/ext/sources/ggml/src/ggml-virtgpu/backend/shared/apir_backend.gen.h +94 -0
  573. data/ext/sources/ggml/src/ggml-virtgpu/backend/shared/apir_backend.h +50 -0
  574. data/ext/sources/ggml/src/ggml-virtgpu/backend/shared/apir_cs.h +378 -0
  575. data/ext/sources/ggml/src/ggml-virtgpu/backend/shared/apir_cs_ggml.h +232 -0
  576. data/ext/sources/ggml/src/ggml-virtgpu/backend/shared/apir_cs_rpc.h +58 -0
  577. data/ext/sources/ggml/src/ggml-virtgpu/ggml-backend-buffer-type.cpp +81 -0
  578. data/ext/sources/ggml/src/ggml-virtgpu/ggml-backend-buffer.cpp +123 -0
  579. data/ext/sources/ggml/src/ggml-virtgpu/ggml-backend-device.cpp +160 -0
  580. data/ext/sources/ggml/src/ggml-virtgpu/ggml-backend-reg.cpp +213 -0
  581. data/ext/sources/ggml/src/ggml-virtgpu/ggml-backend.cpp +71 -0
  582. data/ext/sources/ggml/src/ggml-virtgpu/ggml-remoting.h +71 -0
  583. data/ext/sources/ggml/src/ggml-virtgpu/ggmlremoting_functions.yaml +166 -0
  584. data/ext/sources/ggml/src/ggml-virtgpu/include/apir_hw.h +9 -0
  585. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu-apir.h +15 -0
  586. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu-forward-backend.cpp +58 -0
  587. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu-forward-buffer-type.cpp +110 -0
  588. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu-forward-buffer.cpp +173 -0
  589. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu-forward-device.cpp +192 -0
  590. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu-forward-impl.h +36 -0
  591. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu-forward.gen.h +53 -0
  592. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu-shm.cpp +99 -0
  593. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu-shm.h +23 -0
  594. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu-utils.cpp +179 -0
  595. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu-utils.h +86 -0
  596. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu.cpp +545 -0
  597. data/ext/sources/ggml/src/ggml-virtgpu/virtgpu.h +115 -0
  598. data/ext/sources/ggml/src/ggml-vulkan/CMakeLists.txt +12 -1
  599. data/ext/sources/ggml/src/ggml-vulkan/ggml-vulkan.cpp +3250 -940
  600. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/CMakeLists.txt +4 -0
  601. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/acc.comp +16 -8
  602. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/contig_copy.comp +6 -2
  603. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/conv2d_mm.comp +146 -13
  604. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/copy.comp +3 -1
  605. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/copy_from_quant.comp +1 -1
  606. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/copy_to_quant.comp +25 -1
  607. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs.glsl +88 -0
  608. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/dequant_funcs_cm2.glsl +643 -1
  609. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/dequant_nvfp4.comp +32 -0
  610. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/dequant_q1_0.comp +29 -0
  611. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/diag.comp +0 -1
  612. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/dot_product_funcs.glsl +27 -0
  613. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/elu.comp +27 -0
  614. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/exp.comp +0 -1
  615. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/feature-tests/coopmat2_decode_vector.comp +7 -0
  616. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn.comp +533 -180
  617. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_base.glsl +113 -68
  618. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_cm1.comp +412 -222
  619. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_cm2.comp +222 -83
  620. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_dequant.glsl +131 -0
  621. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_mask_opt.comp +162 -0
  622. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_mmq_funcs.glsl +203 -0
  623. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/flash_attn_split_k_reduce.comp +9 -8
  624. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/fwht.comp +115 -0
  625. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/gated_delta_net.comp +189 -0
  626. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/generic_binary_head.glsl +0 -1
  627. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/glu_head.glsl +10 -1
  628. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/glu_main.glsl +16 -6
  629. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/im2col.comp +76 -54
  630. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/im2col_3d.comp +0 -1
  631. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/l2_norm.comp +12 -9
  632. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/log.comp +0 -1
  633. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec.comp +122 -27
  634. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_base.glsl +20 -17
  635. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_iface.glsl +6 -6
  636. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_q2_k.comp +1 -1
  637. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_q4_k.comp +1 -1
  638. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vec_q5_k.comp +1 -1
  639. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vecq.comp +1 -0
  640. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mat_vecq_funcs.glsl +88 -55
  641. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm.comp +22 -20
  642. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_cm2.comp +51 -14
  643. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_funcs.glsl +159 -125
  644. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mm_id_funcs.glsl +3 -1
  645. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq.comp +5 -3
  646. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_funcs.glsl +8 -8
  647. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/mul_mmq_shmem_types.glsl +24 -9
  648. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/multi_add.comp +0 -1
  649. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/rms_norm.comp +2 -3
  650. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/rope_funcs.glsl +39 -63
  651. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/rope_head.glsl +0 -1
  652. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/rope_multi.comp +7 -4
  653. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/rope_neox.comp +7 -4
  654. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/rope_norm.comp +7 -4
  655. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/rope_params.glsl +13 -7
  656. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/rope_vision.comp +7 -4
  657. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/sgn.comp +21 -0
  658. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/snake.comp +49 -0
  659. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/ssm_conv.comp +27 -11
  660. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/tri.comp +0 -1
  661. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/types.glsl +79 -2
  662. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/vulkan-shaders-gen.cpp +193 -149
  663. data/ext/sources/ggml/src/ggml-webgpu/CMakeLists.txt +5 -2
  664. data/ext/sources/ggml/src/ggml-webgpu/ggml-webgpu-shader-lib.hpp +3221 -97
  665. data/ext/sources/ggml/src/ggml-webgpu/ggml-webgpu.cpp +3493 -1997
  666. data/ext/sources/ggml/src/ggml-webgpu/pre_wgsl.hpp +37 -7
  667. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/add_id.wgsl +64 -0
  668. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/argmax.wgsl +72 -0
  669. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/argsort.wgsl +106 -0
  670. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/argsort_merge.wgsl +134 -0
  671. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/binary.wgsl +142 -0
  672. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/common_decls.tmpl +115 -141
  673. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/concat.wgsl +93 -0
  674. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/conv2d.wgsl +165 -0
  675. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/{cpy.tmpl.wgsl → cpy.wgsl} +25 -44
  676. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/cumsum.wgsl +66 -0
  677. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn.wgsl +198 -230
  678. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_quant_staging.tmpl +124 -0
  679. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_tile.wgsl +397 -0
  680. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_vec_blk.wgsl +101 -0
  681. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_vec_reduce.wgsl +84 -0
  682. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/flash_attn_vec_split.wgsl +619 -0
  683. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/gated_delta_net.wgsl +149 -0
  684. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/{get_rows.tmpl.wgsl → get_rows.wgsl} +234 -335
  685. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/glu.wgsl +155 -0
  686. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/im2col.wgsl +101 -0
  687. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_decls.tmpl +871 -42
  688. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_id.wgsl +195 -0
  689. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_id_gather.wgsl +52 -0
  690. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_id_vec.wgsl +154 -0
  691. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_reg_tile.wgsl +149 -0
  692. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/{mul_mat_subgroup_matrix.tmpl.wgsl → mul_mat_subgroup_matrix.wgsl} +36 -138
  693. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec.wgsl +151 -0
  694. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec_acc.tmpl +1432 -0
  695. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec_q_acc.tmpl +303 -0
  696. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/pad.wgsl +86 -0
  697. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/quant_inner_loops.tmpl +21 -0
  698. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/quantize_q8.wgsl +173 -0
  699. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/repeat.wgsl +67 -0
  700. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/rms_norm_mul.wgsl +152 -0
  701. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/{rope.tmpl.wgsl → rope.wgsl} +71 -142
  702. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/row_norm.wgsl +153 -0
  703. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/{scale.tmpl.wgsl → scale.wgsl} +15 -40
  704. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/set.wgsl +109 -0
  705. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/set_rows.wgsl +39 -12
  706. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/set_rows_quant.wgsl +224 -0
  707. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/{soft_max.tmpl.wgsl → soft_max.wgsl} +106 -206
  708. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/solve_tri.wgsl +121 -0
  709. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/ssm_conv.wgsl +65 -0
  710. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/ssm_scan.wgsl +193 -0
  711. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/sum_rows.wgsl +55 -0
  712. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/unary.wgsl +213 -0
  713. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/upscale.wgsl +240 -0
  714. data/ext/sources/ggml/src/ggml-zdnn/ggml-zdnn.cpp +24 -15
  715. data/ext/sources/ggml/src/ggml-zendnn/CMakeLists.txt +31 -32
  716. data/ext/sources/ggml/src/ggml-zendnn/ggml-zendnn.cpp +253 -16
  717. data/ext/sources/ggml/src/ggml.c +268 -52
  718. data/ext/sources/ggml/src/gguf.cpp +377 -47
  719. data/ext/sources/include/parakeet.h +342 -0
  720. data/ext/sources/include/whisper.h +10 -0
  721. data/ext/sources/media/matmul.png +0 -0
  722. data/ext/sources/src/CMakeLists.txt +23 -0
  723. data/ext/sources/src/parakeet-arch.h +188 -0
  724. data/ext/sources/src/parakeet.cpp +3838 -0
  725. data/ext/sources/src/whisper.cpp +62 -40
  726. data/extsources.rb +26 -10
  727. data/lib/whisper/log_settable.rb +36 -0
  728. data/lib/whisper/model/uri.rb +13 -1
  729. data/lib/whisper/output.rb +74 -0
  730. data/sig/whisper.rbs +445 -55
  731. data/test/helper.rb +2 -0
  732. data/test/jfk_reader/jfk_reader.c +50 -7
  733. data/test/test_callback.rb +1 -0
  734. data/test/test_context_params.rb +82 -0
  735. data/test/test_package.rb +6 -5
  736. data/test/test_parakeet.rb +28 -0
  737. data/test/test_parakeet_callback.rb +107 -0
  738. data/test/test_parakeet_context.rb +116 -0
  739. data/test/test_parakeet_context_params.rb +24 -0
  740. data/test/test_parakeet_model.rb +21 -0
  741. data/test/test_parakeet_params.rb +78 -0
  742. data/test/test_parakeet_segment.rb +42 -0
  743. data/test/test_parakeet_token.rb +73 -0
  744. data/test/test_params.rb +2 -0
  745. data/test/test_token.rb +11 -0
  746. data/test/test_vad_context.rb +58 -8
  747. data/test/test_vad_segment.rb +1 -1
  748. data/test/test_whisper.rb +44 -6
  749. data/whispercpp.gemspec +2 -2
  750. metadata +426 -280
  751. data/ext/sources/bindings/javascript/CMakeLists.txt +0 -41
  752. data/ext/sources/bindings/javascript/emscripten.cpp +0 -93
  753. data/ext/sources/bindings/javascript/libwhisper.worker.js +0 -1
  754. data/ext/sources/bindings/javascript/package.json +0 -26
  755. data/ext/sources/bindings/javascript/whisper.js +0 -19
  756. data/ext/sources/examples/addon.node/CMakeLists.txt +0 -31
  757. data/ext/sources/examples/addon.node/__test__/whisper.spec.js +0 -133
  758. data/ext/sources/examples/addon.node/addon.cpp +0 -557
  759. data/ext/sources/examples/addon.node/index.js +0 -59
  760. data/ext/sources/examples/addon.node/package.json +0 -16
  761. data/ext/sources/examples/addon.node/vad-example.js +0 -132
  762. data/ext/sources/examples/bench.wasm/CMakeLists.txt +0 -49
  763. data/ext/sources/examples/bench.wasm/emscripten.cpp +0 -87
  764. data/ext/sources/examples/bench.wasm/index-tmpl.html +0 -285
  765. data/ext/sources/examples/coi-serviceworker.js +0 -146
  766. data/ext/sources/examples/command/CMakeLists.txt +0 -10
  767. data/ext/sources/examples/command/command.cpp +0 -802
  768. data/ext/sources/examples/command/commands.txt +0 -9
  769. data/ext/sources/examples/command.wasm/CMakeLists.txt +0 -50
  770. data/ext/sources/examples/command.wasm/emscripten.cpp +0 -327
  771. data/ext/sources/examples/command.wasm/index-tmpl.html +0 -415
  772. data/ext/sources/examples/generate-karaoke.sh +0 -57
  773. data/ext/sources/examples/helpers.js +0 -191
  774. data/ext/sources/examples/livestream.sh +0 -112
  775. data/ext/sources/examples/lsp/CMakeLists.txt +0 -10
  776. data/ext/sources/examples/lsp/lsp.cpp +0 -471
  777. data/ext/sources/examples/lsp/whisper.vim +0 -362
  778. data/ext/sources/examples/python/test_whisper_processor.py +0 -7
  779. data/ext/sources/examples/python/whisper_processor.py +0 -54
  780. data/ext/sources/examples/server/bench.js +0 -29
  781. data/ext/sources/examples/server.py +0 -120
  782. data/ext/sources/examples/stream/CMakeLists.txt +0 -10
  783. data/ext/sources/examples/stream/stream.cpp +0 -437
  784. data/ext/sources/examples/stream.wasm/CMakeLists.txt +0 -49
  785. data/ext/sources/examples/stream.wasm/emscripten.cpp +0 -216
  786. data/ext/sources/examples/stream.wasm/index-tmpl.html +0 -491
  787. data/ext/sources/examples/sycl/CMakeLists.txt +0 -9
  788. data/ext/sources/examples/sycl/build.sh +0 -22
  789. data/ext/sources/examples/sycl/ls-sycl-device.cpp +0 -11
  790. data/ext/sources/examples/sycl/run-whisper.sh +0 -17
  791. data/ext/sources/examples/talk-llama/CMakeLists.txt +0 -47
  792. data/ext/sources/examples/talk-llama/eleven-labs.py +0 -80
  793. data/ext/sources/examples/talk-llama/llama-adapter.cpp +0 -494
  794. data/ext/sources/examples/talk-llama/llama-adapter.h +0 -88
  795. data/ext/sources/examples/talk-llama/llama-arch.cpp +0 -2559
  796. data/ext/sources/examples/talk-llama/llama-arch.h +0 -586
  797. data/ext/sources/examples/talk-llama/llama-batch.cpp +0 -917
  798. data/ext/sources/examples/talk-llama/llama-batch.h +0 -173
  799. data/ext/sources/examples/talk-llama/llama-chat.cpp +0 -876
  800. data/ext/sources/examples/talk-llama/llama-chat.h +0 -70
  801. data/ext/sources/examples/talk-llama/llama-context.cpp +0 -3645
  802. data/ext/sources/examples/talk-llama/llama-context.h +0 -360
  803. data/ext/sources/examples/talk-llama/llama-cparams.cpp +0 -5
  804. data/ext/sources/examples/talk-llama/llama-cparams.h +0 -42
  805. data/ext/sources/examples/talk-llama/llama-grammar.cpp +0 -1464
  806. data/ext/sources/examples/talk-llama/llama-grammar.h +0 -194
  807. data/ext/sources/examples/talk-llama/llama-graph.cpp +0 -2282
  808. data/ext/sources/examples/talk-llama/llama-graph.h +0 -910
  809. data/ext/sources/examples/talk-llama/llama-hparams.cpp +0 -241
  810. data/ext/sources/examples/talk-llama/llama-hparams.h +0 -284
  811. data/ext/sources/examples/talk-llama/llama-impl.cpp +0 -171
  812. data/ext/sources/examples/talk-llama/llama-impl.h +0 -63
  813. data/ext/sources/examples/talk-llama/llama-io.cpp +0 -15
  814. data/ext/sources/examples/talk-llama/llama-io.h +0 -35
  815. data/ext/sources/examples/talk-llama/llama-kv-cache-iswa.cpp +0 -328
  816. data/ext/sources/examples/talk-llama/llama-kv-cache-iswa.h +0 -137
  817. data/ext/sources/examples/talk-llama/llama-kv-cache.cpp +0 -2100
  818. data/ext/sources/examples/talk-llama/llama-kv-cache.h +0 -390
  819. data/ext/sources/examples/talk-llama/llama-kv-cells.h +0 -533
  820. data/ext/sources/examples/talk-llama/llama-memory-hybrid.cpp +0 -268
  821. data/ext/sources/examples/talk-llama/llama-memory-hybrid.h +0 -139
  822. data/ext/sources/examples/talk-llama/llama-memory-recurrent.cpp +0 -1167
  823. data/ext/sources/examples/talk-llama/llama-memory-recurrent.h +0 -182
  824. data/ext/sources/examples/talk-llama/llama-memory.cpp +0 -59
  825. data/ext/sources/examples/talk-llama/llama-memory.h +0 -122
  826. data/ext/sources/examples/talk-llama/llama-mmap.cpp +0 -735
  827. data/ext/sources/examples/talk-llama/llama-mmap.h +0 -73
  828. data/ext/sources/examples/talk-llama/llama-model-loader.cpp +0 -1247
  829. data/ext/sources/examples/talk-llama/llama-model-loader.h +0 -176
  830. data/ext/sources/examples/talk-llama/llama-model-saver.cpp +0 -285
  831. data/ext/sources/examples/talk-llama/llama-model-saver.h +0 -37
  832. data/ext/sources/examples/talk-llama/llama-model.cpp +0 -8338
  833. data/ext/sources/examples/talk-llama/llama-model.h +0 -544
  834. data/ext/sources/examples/talk-llama/llama-quant.cpp +0 -1072
  835. data/ext/sources/examples/talk-llama/llama-quant.h +0 -1
  836. data/ext/sources/examples/talk-llama/llama-sampling.cpp +0 -3771
  837. data/ext/sources/examples/talk-llama/llama-sampling.h +0 -44
  838. data/ext/sources/examples/talk-llama/llama-vocab.cpp +0 -3900
  839. data/ext/sources/examples/talk-llama/llama-vocab.h +0 -182
  840. data/ext/sources/examples/talk-llama/llama.cpp +0 -1140
  841. data/ext/sources/examples/talk-llama/llama.h +0 -1540
  842. data/ext/sources/examples/talk-llama/models/afmoe.cpp +0 -191
  843. data/ext/sources/examples/talk-llama/models/apertus.cpp +0 -125
  844. data/ext/sources/examples/talk-llama/models/arcee.cpp +0 -135
  845. data/ext/sources/examples/talk-llama/models/arctic.cpp +0 -138
  846. data/ext/sources/examples/talk-llama/models/arwkv7.cpp +0 -86
  847. data/ext/sources/examples/talk-llama/models/baichuan.cpp +0 -122
  848. data/ext/sources/examples/talk-llama/models/bailingmoe.cpp +0 -144
  849. data/ext/sources/examples/talk-llama/models/bailingmoe2.cpp +0 -135
  850. data/ext/sources/examples/talk-llama/models/bert.cpp +0 -178
  851. data/ext/sources/examples/talk-llama/models/bitnet.cpp +0 -160
  852. data/ext/sources/examples/talk-llama/models/bloom.cpp +0 -101
  853. data/ext/sources/examples/talk-llama/models/chameleon.cpp +0 -178
  854. data/ext/sources/examples/talk-llama/models/chatglm.cpp +0 -132
  855. data/ext/sources/examples/talk-llama/models/codeshell.cpp +0 -111
  856. data/ext/sources/examples/talk-llama/models/cogvlm.cpp +0 -102
  857. data/ext/sources/examples/talk-llama/models/cohere2-iswa.cpp +0 -134
  858. data/ext/sources/examples/talk-llama/models/command-r.cpp +0 -122
  859. data/ext/sources/examples/talk-llama/models/dbrx.cpp +0 -123
  860. data/ext/sources/examples/talk-llama/models/deci.cpp +0 -135
  861. data/ext/sources/examples/talk-llama/models/deepseek.cpp +0 -144
  862. data/ext/sources/examples/talk-llama/models/deepseek2.cpp +0 -259
  863. data/ext/sources/examples/talk-llama/models/dots1.cpp +0 -134
  864. data/ext/sources/examples/talk-llama/models/dream.cpp +0 -105
  865. data/ext/sources/examples/talk-llama/models/ernie4-5-moe.cpp +0 -150
  866. data/ext/sources/examples/talk-llama/models/ernie4-5.cpp +0 -110
  867. data/ext/sources/examples/talk-llama/models/exaone.cpp +0 -114
  868. data/ext/sources/examples/talk-llama/models/exaone4.cpp +0 -123
  869. data/ext/sources/examples/talk-llama/models/falcon-h1.cpp +0 -113
  870. data/ext/sources/examples/talk-llama/models/falcon.cpp +0 -120
  871. data/ext/sources/examples/talk-llama/models/gemma-embedding.cpp +0 -116
  872. data/ext/sources/examples/talk-llama/models/gemma.cpp +0 -112
  873. data/ext/sources/examples/talk-llama/models/gemma2-iswa.cpp +0 -128
  874. data/ext/sources/examples/talk-llama/models/gemma3.cpp +0 -155
  875. data/ext/sources/examples/talk-llama/models/gemma3n-iswa.cpp +0 -384
  876. data/ext/sources/examples/talk-llama/models/glm4-moe.cpp +0 -170
  877. data/ext/sources/examples/talk-llama/models/glm4.cpp +0 -150
  878. data/ext/sources/examples/talk-llama/models/gpt2.cpp +0 -105
  879. data/ext/sources/examples/talk-llama/models/gptneox.cpp +0 -144
  880. data/ext/sources/examples/talk-llama/models/granite-hybrid.cpp +0 -196
  881. data/ext/sources/examples/talk-llama/models/granite.cpp +0 -211
  882. data/ext/sources/examples/talk-llama/models/graph-context-mamba.cpp +0 -283
  883. data/ext/sources/examples/talk-llama/models/grok.cpp +0 -159
  884. data/ext/sources/examples/talk-llama/models/grovemoe.cpp +0 -141
  885. data/ext/sources/examples/talk-llama/models/hunyuan-dense.cpp +0 -132
  886. data/ext/sources/examples/talk-llama/models/hunyuan-moe.cpp +0 -154
  887. data/ext/sources/examples/talk-llama/models/internlm2.cpp +0 -120
  888. data/ext/sources/examples/talk-llama/models/jais.cpp +0 -86
  889. data/ext/sources/examples/talk-llama/models/jamba.cpp +0 -106
  890. data/ext/sources/examples/talk-llama/models/lfm2.cpp +0 -175
  891. data/ext/sources/examples/talk-llama/models/llada-moe.cpp +0 -122
  892. data/ext/sources/examples/talk-llama/models/llada.cpp +0 -99
  893. data/ext/sources/examples/talk-llama/models/llama-iswa.cpp +0 -178
  894. data/ext/sources/examples/talk-llama/models/llama.cpp +0 -168
  895. data/ext/sources/examples/talk-llama/models/maincoder.cpp +0 -117
  896. data/ext/sources/examples/talk-llama/models/mamba.cpp +0 -55
  897. data/ext/sources/examples/talk-llama/models/mimo2-iswa.cpp +0 -123
  898. data/ext/sources/examples/talk-llama/models/minicpm3.cpp +0 -199
  899. data/ext/sources/examples/talk-llama/models/minimax-m2.cpp +0 -124
  900. data/ext/sources/examples/talk-llama/models/mistral3.cpp +0 -160
  901. data/ext/sources/examples/talk-llama/models/models.h +0 -569
  902. data/ext/sources/examples/talk-llama/models/modern-bert.cpp +0 -116
  903. data/ext/sources/examples/talk-llama/models/mpt.cpp +0 -126
  904. data/ext/sources/examples/talk-llama/models/nemotron-h.cpp +0 -150
  905. data/ext/sources/examples/talk-llama/models/nemotron.cpp +0 -122
  906. data/ext/sources/examples/talk-llama/models/neo-bert.cpp +0 -104
  907. data/ext/sources/examples/talk-llama/models/olmo.cpp +0 -121
  908. data/ext/sources/examples/talk-llama/models/olmo2.cpp +0 -150
  909. data/ext/sources/examples/talk-llama/models/olmoe.cpp +0 -124
  910. data/ext/sources/examples/talk-llama/models/openai-moe-iswa.cpp +0 -127
  911. data/ext/sources/examples/talk-llama/models/openelm.cpp +0 -124
  912. data/ext/sources/examples/talk-llama/models/orion.cpp +0 -123
  913. data/ext/sources/examples/talk-llama/models/pangu-embedded.cpp +0 -121
  914. data/ext/sources/examples/talk-llama/models/phi2.cpp +0 -121
  915. data/ext/sources/examples/talk-llama/models/phi3.cpp +0 -152
  916. data/ext/sources/examples/talk-llama/models/plamo.cpp +0 -110
  917. data/ext/sources/examples/talk-llama/models/plamo2.cpp +0 -316
  918. data/ext/sources/examples/talk-llama/models/plamo3.cpp +0 -128
  919. data/ext/sources/examples/talk-llama/models/plm.cpp +0 -168
  920. data/ext/sources/examples/talk-llama/models/qwen.cpp +0 -108
  921. data/ext/sources/examples/talk-llama/models/qwen2.cpp +0 -126
  922. data/ext/sources/examples/talk-llama/models/qwen2moe.cpp +0 -151
  923. data/ext/sources/examples/talk-llama/models/qwen2vl.cpp +0 -117
  924. data/ext/sources/examples/talk-llama/models/qwen3.cpp +0 -117
  925. data/ext/sources/examples/talk-llama/models/qwen3moe.cpp +0 -124
  926. data/ext/sources/examples/talk-llama/models/qwen3next.cpp +0 -873
  927. data/ext/sources/examples/talk-llama/models/qwen3vl-moe.cpp +0 -149
  928. data/ext/sources/examples/talk-llama/models/qwen3vl.cpp +0 -141
  929. data/ext/sources/examples/talk-llama/models/refact.cpp +0 -94
  930. data/ext/sources/examples/talk-llama/models/rnd1.cpp +0 -126
  931. data/ext/sources/examples/talk-llama/models/rwkv6-base.cpp +0 -162
  932. data/ext/sources/examples/talk-llama/models/rwkv6.cpp +0 -94
  933. data/ext/sources/examples/talk-llama/models/rwkv6qwen2.cpp +0 -86
  934. data/ext/sources/examples/talk-llama/models/rwkv7-base.cpp +0 -135
  935. data/ext/sources/examples/talk-llama/models/rwkv7.cpp +0 -90
  936. data/ext/sources/examples/talk-llama/models/seed-oss.cpp +0 -124
  937. data/ext/sources/examples/talk-llama/models/smallthinker.cpp +0 -126
  938. data/ext/sources/examples/talk-llama/models/smollm3.cpp +0 -128
  939. data/ext/sources/examples/talk-llama/models/stablelm.cpp +0 -146
  940. data/ext/sources/examples/talk-llama/models/starcoder.cpp +0 -100
  941. data/ext/sources/examples/talk-llama/models/starcoder2.cpp +0 -121
  942. data/ext/sources/examples/talk-llama/models/t5-dec.cpp +0 -166
  943. data/ext/sources/examples/talk-llama/models/t5-enc.cpp +0 -96
  944. data/ext/sources/examples/talk-llama/models/wavtokenizer-dec.cpp +0 -149
  945. data/ext/sources/examples/talk-llama/models/xverse.cpp +0 -108
  946. data/ext/sources/examples/talk-llama/prompts/talk-alpaca.txt +0 -23
  947. data/ext/sources/examples/talk-llama/speak +0 -40
  948. data/ext/sources/examples/talk-llama/speak.bat +0 -1
  949. data/ext/sources/examples/talk-llama/speak.ps1 +0 -14
  950. data/ext/sources/examples/talk-llama/talk-llama.cpp +0 -813
  951. data/ext/sources/examples/talk-llama/unicode-data.cpp +0 -7034
  952. data/ext/sources/examples/talk-llama/unicode-data.h +0 -20
  953. data/ext/sources/examples/talk-llama/unicode.cpp +0 -1147
  954. data/ext/sources/examples/talk-llama/unicode.h +0 -111
  955. data/ext/sources/examples/wchess/CMakeLists.txt +0 -10
  956. data/ext/sources/examples/wchess/libwchess/CMakeLists.txt +0 -19
  957. data/ext/sources/examples/wchess/libwchess/Chessboard.cpp +0 -803
  958. data/ext/sources/examples/wchess/libwchess/Chessboard.h +0 -33
  959. data/ext/sources/examples/wchess/libwchess/WChess.cpp +0 -193
  960. data/ext/sources/examples/wchess/libwchess/WChess.h +0 -63
  961. data/ext/sources/examples/wchess/libwchess/test-chessboard.cpp +0 -117
  962. data/ext/sources/examples/wchess/wchess.cmd/CMakeLists.txt +0 -8
  963. data/ext/sources/examples/wchess/wchess.cmd/wchess.cmd.cpp +0 -253
  964. data/ext/sources/examples/whisper.wasm/CMakeLists.txt +0 -50
  965. data/ext/sources/examples/whisper.wasm/emscripten.cpp +0 -118
  966. data/ext/sources/examples/whisper.wasm/index-tmpl.html +0 -659
  967. data/ext/sources/ggml/cmake/BuildTypes.cmake +0 -54
  968. data/ext/sources/ggml/src/ggml-cpu/llamafile/sgemm-ppc.h +0 -333
  969. data/ext/sources/ggml/src/ggml-cuda/template-instances/generate_cu_files.py +0 -99
  970. data/ext/sources/ggml/src/ggml-hexagon/htp/htp-dma.h +0 -157
  971. data/ext/sources/ggml/src/ggml-hexagon/htp/htp-msg.h +0 -165
  972. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-exp.c +0 -94
  973. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-inverse.c +0 -72
  974. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-sigmoid.c +0 -49
  975. data/ext/sources/ggml/src/ggml-hexagon/htp/hvx-utils.c +0 -1020
  976. data/ext/sources/ggml/src/ggml-hexagon/htp/ops-utils.h +0 -149
  977. data/ext/sources/ggml/src/ggml-hexagon/htp-utils.c +0 -454
  978. data/ext/sources/ggml/src/ggml-hexagon/htp-utils.h +0 -221
  979. data/ext/sources/ggml/src/ggml-hexagon/op-desc.h +0 -153
  980. data/ext/sources/ggml/src/ggml-opencl/kernels/embed_kernel.py +0 -26
  981. data/ext/sources/ggml/src/ggml-vulkan/vulkan-shaders/rte.glsl +0 -5
  982. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/bin_op.tmpl.wgsl +0 -188
  983. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/binary_head.tmpl +0 -45
  984. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/embed_wgsl.py +0 -147
  985. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/glu.tmpl.wgsl +0 -323
  986. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat.tmpl.wgsl +0 -907
  987. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_reg_tile.tmpl.wgsl +0 -247
  988. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/mul_mat_vec.tmpl.wgsl +0 -267
  989. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/rms_norm.wgsl +0 -123
  990. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/set_rows.tmpl.wgsl +0 -112
  991. data/ext/sources/ggml/src/ggml-webgpu/wgsl-shaders/unary_op.wgsl +0 -483
  992. data/ext/sources/tests/CMakeLists.txt +0 -112
  993. data/ext/sources/tests/earnings21/eval.mk +0 -58
  994. data/ext/sources/tests/earnings21/eval.py +0 -68
  995. data/ext/sources/tests/earnings21/normalizers/__init__.py +0 -2
  996. data/ext/sources/tests/earnings21/normalizers/basic.py +0 -80
  997. data/ext/sources/tests/earnings21/normalizers/english.json +0 -1741
  998. data/ext/sources/tests/earnings21/normalizers/english.py +0 -550
  999. data/ext/sources/tests/earnings21/requirements.txt +0 -6
  1000. data/ext/sources/tests/en-0-ref.txt +0 -1
  1001. data/ext/sources/tests/en-1-ref.txt +0 -1
  1002. data/ext/sources/tests/en-2-ref.txt +0 -1
  1003. data/ext/sources/tests/es-0-ref.txt +0 -1
  1004. data/ext/sources/tests/librispeech/eval.mk +0 -39
  1005. data/ext/sources/tests/librispeech/eval.py +0 -47
  1006. data/ext/sources/tests/librispeech/normalizers/__init__.py +0 -2
  1007. data/ext/sources/tests/librispeech/normalizers/basic.py +0 -80
  1008. data/ext/sources/tests/librispeech/normalizers/english.json +0 -1741
  1009. data/ext/sources/tests/librispeech/normalizers/english.py +0 -550
  1010. data/ext/sources/tests/librispeech/requirements.txt +0 -6
  1011. data/ext/sources/tests/run-tests.sh +0 -130
  1012. data/ext/sources/tests/test-c.c +0 -3
  1013. data/ext/sources/tests/test-vad-full.cpp +0 -56
  1014. data/ext/sources/tests/test-vad.cpp +0 -83
  1015. data/ext/sources/tests/test-whisper.js +0 -58
  1016. data/lib/whisper/context.rb +0 -15
  1017. data/lib/whisper/segment.rb +0 -58
@@ -0,0 +1,1795 @@
1
+ #define GGML_COMMON_IMPL_CPP
2
+ #define GGML_COMMON_DECL_CPP
3
+
4
+ #include "repack.h"
5
+
6
+ #include "ggml-common.h"
7
+ #include "ggml-cpu.h"
8
+ #include "ggml-impl.h"
9
+ #include "ime_kernels.h"
10
+
11
+ #include <algorithm>
12
+ #include <cassert>
13
+ #include <cmath>
14
+ #include <cstring>
15
+
16
+ // clang-format off
17
+ #if defined(__riscv)
18
+
19
+ #if !defined(__riscv_v) || !defined(__riscv_v_intrinsic)
20
+ #error "riscv v extension or v_intrinsic not enabled"
21
+ #else
22
+ #include <riscv_vector.h>
23
+ #endif
24
+
25
+ #if !defined(__riscv_zfh)
26
+ #error "riscv zfh extension not enabled"
27
+ #endif
28
+
29
+ #else
30
+ #error "riscv not enabled in this build"
31
+ #endif
32
+
33
+ #if defined(__GNUC__)
34
+ #pragma GCC diagnostic ignored "-Wcast-qual"
35
+ #pragma GCC diagnostic ignored "-Wunused-parameter"
36
+ #endif
37
+
38
+ // clang-format on
39
+
40
+ template <int K> constexpr int QK_0() {
41
+ if constexpr (K == 4) {
42
+ return QK4_0;
43
+ }
44
+ if constexpr (K == 8) {
45
+ return QK8_0;
46
+ }
47
+ return -1;
48
+ }
49
+
50
+ template <int K, int N> struct block {
51
+ ggml_half d[N]; // deltas for N qK_0 blocks
52
+ uint8_t qs[(QK_0<K>() * N * K) / 8]; // quants for N qK_0 blocks
53
+ };
54
+
55
+ template <int K, int N> struct block_with_zp {
56
+ ggml_half d[N]; // deltas for N qK_1 blocks
57
+ uint8_t zp[N]; // zero points for N qK_1 blocks
58
+ uint8_t qs[(QK_0<K>() * N * K) / 8]; // quants for N qK_1 blocks
59
+ };
60
+
61
+ // control size
62
+ static_assert(sizeof(block<4, 16>) == 16 * sizeof(ggml_half) + QK4_0 * 8, "wrong block<4,16> size/padding");
63
+ static_assert(sizeof(block_with_zp<4, 16>) == 16 * sizeof(ggml_half) + QK4_0 * 8 + 16 * sizeof(uint8_t),
64
+ "wrong block_with_zp<4,16> size/padding");
65
+
66
+ static_assert(sizeof(block<8, 16>) == 16 * sizeof(ggml_half) + QK4_0 * 16, "wrong block<8,16> size/padding");
67
+
68
+ static_assert(sizeof(block<4, 32>) == 32 * sizeof(ggml_half) + QK4_0 * 16, "wrong block<4,32> size/padding");
69
+ static_assert(sizeof(block_with_zp<4, 32>) == 32 * sizeof(ggml_half) + QK4_0 * 16 + 32 * sizeof(uint8_t),
70
+ "wrong block_with_zp<4,32> size/padding");
71
+
72
+ using block_q4_0x16 = block<4, 16>;
73
+ using block_q4_1x16 = block_with_zp<4, 16>;
74
+ using block_q8_0x16 = block<8, 16>;
75
+
76
+ using block_q4_0x32 = block<4, 32>;
77
+ using block_q4_1x32 = block_with_zp<4, 32>;
78
+ using block_q8_0x32 = block<8, 32>;
79
+
80
+ struct block_q4_0x32x256 {
81
+ block_q4_0x32 blocks[8]; // [f16 * 32 | i4 * 32 * 32] * 8
82
+ };
83
+
84
+ struct block_q4_1x32x256 {
85
+ block_q4_0x32 blocks[8];
86
+ uint8_t zps[32 * 8];
87
+ };
88
+
89
+ static block_q4_0x16 make_block_q4_0x16(block_q4_0 * in, unsigned int blck_size_interleave) {
90
+ block_q4_0x16 out;
91
+ GGML_ASSERT(QK4_0 / blck_size_interleave == 2);
92
+
93
+ for (int i = 0; i < 16; i++) {
94
+ out.d[i] = in[i].d;
95
+ }
96
+
97
+ for (int i = 0; i < 16; i++) {
98
+ // [0, 15], in.d & 0x0F
99
+ for (int j = 0; j < QK4_0 / 4; j++) {
100
+ //src [b0 b16] ......... [b8 b24] ......... [b15 b31]
101
+ //dst [b0 b8] ......... [b7 b15]
102
+ out.qs[i * QK4_0 / 4 + j] = (in[i].qs[j] & 0x0F) | ((in[i].qs[j + QK4_0 / 4] & 0x0F) << 4);
103
+ }
104
+ }
105
+
106
+ for (int i = 0; i < 16; i++) {
107
+ // [16, 31], in.d & 0xF0
108
+ for (int j = 0; j < QK4_0 / 4; j++) {
109
+ //src [b0 b16] ......... [b8 b24] ......... [b15 b31]
110
+ //dst [b16 b24] ......... [b23 b31]
111
+ out.qs[4 * QK4_0 + i * QK4_0 / 4 + j] = ((in[i].qs[j] & 0xF0) >> 4) | (in[i].qs[j + QK4_0 / 4] & 0xF0);
112
+ }
113
+ }
114
+
115
+ return out;
116
+ }
117
+
118
+ static block_q4_1x16 make_block_q4_1x16(block_q4_1 * in, unsigned int blck_size_interleave) {
119
+ block_q4_1x16 out;
120
+ GGML_ASSERT(QK4_1 / blck_size_interleave == 2);
121
+
122
+ for (int i = 0; i < 16; i++) {
123
+ float d = GGML_FP16_TO_FP32(in[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d);
124
+ float m = GGML_FP16_TO_FP32(in[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.m);
125
+ float mid = -std::nearbyintf(m / d);
126
+ mid = std::min(15.0f, std::max(0.0f, mid));
127
+ out.d[i] = GGML_FP32_TO_FP16(d);
128
+ out.zp[i] = static_cast<uint8_t>(mid);
129
+ }
130
+
131
+ for (int i = 0; i < 16; i++) {
132
+ // [0, 15], in.d & 0x0F
133
+ for (int j = 0; j < QK4_1 / 4; j++) {
134
+ //src [b0 b16] ......... [b8 b24] ......... [b15 b31]
135
+ //dst [b0 b8] ......... [b7 b15]
136
+ out.qs[i * QK4_1 / 4 + j] = (in[i].qs[j] & 0x0F) | ((in[i].qs[j + QK4_1 / 4] & 0x0F) << 4);
137
+ }
138
+ }
139
+
140
+ for (int i = 0; i < 16; i++) {
141
+ // [16, 31], in.d & 0xF0
142
+ for (int j = 0; j < QK4_1 / 4; j++) {
143
+ //src [b0 b16] ......... [b8 b24] ......... [b15 b31]
144
+ //dst [b16 b24] ......... [b23 b31]
145
+ out.qs[4 * QK4_1 + i * QK4_1 / 4 + j] = ((in[i].qs[j] & 0xF0) >> 4) | (in[i].qs[j + QK4_1 / 4] & 0xF0);
146
+ }
147
+ }
148
+
149
+ return out;
150
+ }
151
+
152
+ static int repack_q4_0_to_q4_0_16_bl(ggml_tensor * t,
153
+ int interleave_block,
154
+ const void * GGML_RESTRICT data,
155
+ size_t data_size) {
156
+ GGML_ASSERT(t->type == GGML_TYPE_Q4_0);
157
+ GGML_ASSERT(interleave_block == 16);
158
+
159
+ constexpr int nrows_interleaved = 16;
160
+
161
+ block_q4_0x16 * dst = (block_q4_0x16 *) t->data;
162
+ const block_q4_0 * src = (const block_q4_0 *) data;
163
+ block_q4_0 dst_tmp[16];
164
+ int nrow = ggml_nrows(t);
165
+ int nblocks = t->ne[0] / QK4_0;
166
+
167
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q4_0));
168
+
169
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK4_0 != 0) {
170
+ return -1;
171
+ }
172
+
173
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
174
+ for (int64_t x = 0; x < nblocks; x++) {
175
+ for (int i = 0; i < nrows_interleaved; i++) {
176
+ dst_tmp[i] = src[x + i * nblocks];
177
+ }
178
+ *dst++ = make_block_q4_0x16(dst_tmp, interleave_block);
179
+ }
180
+ src += nrows_interleaved * nblocks;
181
+ }
182
+ return 0;
183
+
184
+ GGML_UNUSED(data_size);
185
+ }
186
+
187
+ static int repack_q4_1_to_q4_1_16_bl(ggml_tensor * t,
188
+ int interleave_block,
189
+ const void * GGML_RESTRICT data,
190
+ size_t data_size) {
191
+ GGML_ASSERT(t->type == GGML_TYPE_Q4_1);
192
+ GGML_ASSERT(interleave_block == 16);
193
+
194
+ constexpr int nrows_interleaved = 16;
195
+
196
+ block_q4_1x16 * dst = (block_q4_1x16 *) t->data;
197
+ const block_q4_1 * src = (const block_q4_1 *) data;
198
+ block_q4_1 dst_tmp[16];
199
+ int nrow = ggml_nrows(t);
200
+ int nblocks = t->ne[0] / QK4_1;
201
+
202
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q4_1));
203
+
204
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK4_1 != 0) {
205
+ return -1;
206
+ }
207
+
208
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
209
+ for (int64_t x = 0; x < nblocks; x++) {
210
+ for (int i = 0; i < nrows_interleaved; i++) {
211
+ dst_tmp[i] = src[x + i * nblocks];
212
+ }
213
+ *dst++ = make_block_q4_1x16(dst_tmp, interleave_block);
214
+ }
215
+ src += nrows_interleaved * nblocks;
216
+ }
217
+ return 0;
218
+
219
+ GGML_UNUSED(data_size);
220
+ }
221
+
222
+ static inline void get_scale_min_k4(int j,
223
+ const uint8_t * GGML_RESTRICT q,
224
+ uint8_t * GGML_RESTRICT d,
225
+ uint8_t * GGML_RESTRICT m) {
226
+ if (j < 4) {
227
+ *d = q[j] & 63;
228
+ *m = q[j + 4] & 63;
229
+ } else {
230
+ *d = (q[j + 4] & 0xF) | ((q[j - 4] >> 6) << 4);
231
+ *m = (q[j + 4] >> 4) | ((q[j - 0] >> 6) << 4);
232
+ }
233
+ }
234
+
235
+ static int repack_q4_k_to_q4_1_16_bl(ggml_tensor * t,
236
+ int interleave_block,
237
+ const void * GGML_RESTRICT data,
238
+ size_t data_size) {
239
+ GGML_ASSERT(t->type == GGML_TYPE_Q4_K);
240
+ GGML_ASSERT(interleave_block == 16);
241
+ GGML_ASSERT(QK_K / QK4_1 == 8);
242
+
243
+ constexpr int nrows_interleaved = 16;
244
+
245
+ block_q4_1x16 * dst = (block_q4_1x16 *) t->data;
246
+ const block_q4_K * src = (const block_q4_K *) data;
247
+ block_q4_1 dst_tmp[16];
248
+ int nrow = ggml_nrows(t);
249
+ int nblocks = t->ne[0] / QK_K;
250
+
251
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK_K != 0) {
252
+ return -1;
253
+ }
254
+
255
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
256
+ for (int64_t x = 0; x < nblocks; x++) {
257
+ for (int j = 0; j < 8; j++) {
258
+ for (int i = 0; i < nrows_interleaved; i++) {
259
+ uint8_t sc, m;
260
+ const float d = GGML_FP16_TO_FP32(src[x + i * nblocks].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d);
261
+ const float min =
262
+ GGML_FP16_TO_FP32(src[x + i * nblocks].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.dmin);
263
+ get_scale_min_k4(j, src[x + i * nblocks].scales, &sc, &m);
264
+ const float d1 = d * sc;
265
+ const float m1 = min * m;
266
+
267
+ dst_tmp[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d = GGML_FP32_TO_FP16(d1);
268
+ dst_tmp[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.m = GGML_FP32_TO_FP16(-m1);
269
+ // src -> [b0, b32] [b1, b33] ... [b31, b63]
270
+ // dst -> [b0, b16] [b1, b17] ... [b15, b31] [b32, b48] [b33, b49] ... [b47, b63]
271
+ const uint8_t * q = src[x + i * nblocks].qs + (j / 2) * QK4_1;
272
+ if (j % 2 == 0) {
273
+ for (int ii = 0; ii < 16; ii++) {
274
+ dst_tmp[i].qs[ii] = (q[ii] & 0x0F) | ((q[ii + 16] & 0x0F) << 4);
275
+ }
276
+ } else {
277
+ for (int ii = 0; ii < 16; ii++) {
278
+ dst_tmp[i].qs[ii] = ((q[ii] & 0xF0) >> 4) | (q[ii + 16] & 0xF0);
279
+ }
280
+ }
281
+ }
282
+ *dst++ = make_block_q4_1x16(dst_tmp, interleave_block);
283
+ }
284
+ }
285
+ src += nrows_interleaved * nblocks;
286
+ }
287
+ return 0;
288
+
289
+ GGML_UNUSED(data_size);
290
+ }
291
+
292
+ static block_q4_0x32 make_block_q4_0x32(block_q4_0 * in, unsigned int blck_size_interleave) {
293
+ block_q4_0x32 out;
294
+ assert(QK4_0 / blck_size_interleave == 1);
295
+ GGML_UNUSED(blck_size_interleave);
296
+
297
+ for (int i = 0; i < 32; i++) {
298
+ out.d[i] = in[i].d;
299
+ }
300
+
301
+ for (int i = 0; i < 32; i++) {
302
+ // [0, 15], in.d & 0x0F
303
+ for (int j = 0; j < QK4_0 / 4; j++) {
304
+ //src [b0 b16] ......... [b8 b24] ......... [b15 b31]
305
+ //dst [b0 b1] ......... [b14 b15]
306
+ out.qs[i * QK4_0 / 2 + j] = (in[i].qs[j * 2] & 0x0F) | ((in[i].qs[j * 2 + 1] & 0x0F) << 4);
307
+ }
308
+ }
309
+
310
+ for (int i = 0; i < 32; i++) {
311
+ // [16, 31], in.d & 0xF0
312
+ for (int j = 0; j < QK4_0 / 4; j++) {
313
+ //src [b0 b16] ......... [b8 b24] ......... [b15 b31]
314
+ //dst [b16 b17] ......... [b30 b31]
315
+ out.qs[i * QK4_0 / 2 + QK4_0 / 4 + j] = ((in[i].qs[j * 2] & 0xF0) >> 4) | (in[i].qs[j * 2 + 1] & 0xF0);
316
+ }
317
+ }
318
+
319
+ return out;
320
+ }
321
+
322
+ static block_q4_1x32 make_block_q4_1x32(block_q4_1 * in, unsigned int blck_size_interleave) {
323
+ block_q4_1x32 out;
324
+ GGML_ASSERT(QK4_1 / blck_size_interleave == 1);
325
+ GGML_UNUSED(blck_size_interleave);
326
+
327
+ for (int i = 0; i < 32; i++) {
328
+ float d = GGML_FP16_TO_FP32(in[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d);
329
+ float m = GGML_FP16_TO_FP32(in[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.m);
330
+ float mid = -std::nearbyintf(m / d);
331
+ mid = std::min(15.0f, std::max(0.0f, mid));
332
+ out.d[i] = GGML_FP32_TO_FP16(d);
333
+ out.zp[i] = static_cast<uint8_t>(mid);
334
+ }
335
+
336
+ for (int i = 0; i < 32; i++) {
337
+ // [0, 15], in.d & 0x0F
338
+ for (int j = 0; j < QK4_1 / 4; j++) {
339
+ //src [b0 b16] ......... [b8 b24] ......... [b15 b31]
340
+ //dst [b0 b1] ......... [b14 b15]
341
+ out.qs[i * QK4_1 / 2 + j] = (in[i].qs[j * 2] & 0x0F) | ((in[i].qs[j * 2 + 1] & 0x0F) << 4);
342
+ }
343
+ }
344
+
345
+ for (int i = 0; i < 32; i++) {
346
+ // [16, 31], in.d & 0xF0
347
+ for (int j = 0; j < QK4_1 / 4; j++) {
348
+ //src [b0 b16] ......... [b8 b24] ......... [b15 b31]
349
+ //dst [b16 b24] ......... [b23 b31]
350
+ out.qs[i * QK4_1 / 2 + QK4_1 / 4 + j] = ((in[i].qs[j * 2] & 0xF0) >> 4) | (in[i].qs[j * 2 + 1] & 0xF0);
351
+ }
352
+ }
353
+
354
+ return out;
355
+ }
356
+
357
+ static block_q8_0x32 make_block_q8_0x32(block_q8_0 * in, unsigned int blck_size_interleave) {
358
+ block_q8_0x32 out;
359
+ GGML_ASSERT(QK8_0 / blck_size_interleave == 1);
360
+ GGML_UNUSED(blck_size_interleave);
361
+
362
+ for (int i = 0; i < 32; i++) {
363
+ out.d[i] = in[i].d;
364
+ }
365
+
366
+ for (int i = 0; i < 32; i++) {
367
+ memcpy(out.qs + i * QK8_0, in[i].qs, QK8_0);
368
+ }
369
+
370
+ return out;
371
+ }
372
+
373
+ static int repack_q2_k_to_q2_k_32_bl(ggml_tensor * t,
374
+ int interleave_block,
375
+ const void * GGML_RESTRICT data,
376
+ size_t data_size) {
377
+ GGML_ASSERT(t->type == GGML_TYPE_Q2_K);
378
+ GGML_ASSERT(interleave_block == 32);
379
+ GGML_ASSERT(QK_K == 256);
380
+
381
+ constexpr int nrows_interleaved = 32;
382
+
383
+ const block_q2_K * src = (const block_q2_K *) data;
384
+
385
+ auto * dst = (spacemit_kernels::nrow_block_q2_k<32> *) t->data;
386
+
387
+ int nrow = ggml_nrows(t);
388
+ int nblocks = t->ne[0] / QK_K;
389
+
390
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q2_K));
391
+
392
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK_K != 0) {
393
+ return -1;
394
+ }
395
+
396
+ uint8_t qs_aux[256] = { 0 };
397
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
398
+ for (int64_t x = 0; x < nblocks; x++) {
399
+ for (int i = 0; i < nrows_interleaved; i++) {
400
+ const block_q2_K * src_block = &src[(b + i) * nblocks + x];
401
+
402
+ // scale for [16, N]
403
+ for (int j = 0; j < 16; j++) {
404
+ auto zp_aux = (dst->scales[j * nrows_interleaved + i]) & 0xF0;
405
+
406
+ dst->scales[j * nrows_interleaved + i] = (src_block->scales[j] & 0x0F) | zp_aux;
407
+ }
408
+
409
+ // zp for [N, 16]
410
+ for (int j = 0; j < 16; j++) {
411
+ auto scale_aux = (dst->scales[16 * i + j]) & 0x0F;
412
+
413
+ dst->scales[16 * i + j] = (src_block->scales[j] & 0xF0) | scale_aux;
414
+ }
415
+
416
+ for (int k = 0; k < 4; k++) {
417
+ for (int j = 0; j < 32; j++) {
418
+ qs_aux[k * 32 + j] = (src_block->qs[j] >> (2 * k)) & 0x03;
419
+ }
420
+ }
421
+
422
+ for (int k = 0; k < 4; k++) {
423
+ for (int j = 0; j < 32; j++) {
424
+ qs_aux[k * 32 + j + 128] = (src_block->qs[j + 32] >> (2 * k)) & 0x03;
425
+ }
426
+ }
427
+
428
+ // from nrows_interleaved * [2 * 32byte]
429
+ // to 4 * [nrows_interleaved * 16byte]
430
+ for (int k = 0; k < 4; k++) {
431
+ for (int j = 0; j < 16; j++) {
432
+ uint8_t qs0 = qs_aux[j + k * 64];
433
+ uint8_t qs16 = qs_aux[j + 16 + k * 64];
434
+ uint8_t qs32 = qs_aux[j + 32 + k * 64];
435
+ uint8_t qs48 = qs_aux[j + 48 + k * 64];
436
+
437
+ dst->qs[(k * nrows_interleaved + i) * 16 + j] =
438
+ (qs0 & 0x03) | ((qs16 & 0x03) << 2) | ((qs32 & 0x03) << 4) | ((qs48 & 0x03) << 6);
439
+ }
440
+ }
441
+
442
+ dst->scales16[i] = src_block->GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d;
443
+ dst->zeros16[i] = src_block->GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.dmin;
444
+ }
445
+ dst++;
446
+ }
447
+ }
448
+
449
+ return 0;
450
+ }
451
+
452
+ static int repack_q3_k_to_q3_k_32_bl(ggml_tensor * t,
453
+ int interleave_block,
454
+ const void * GGML_RESTRICT data,
455
+ size_t data_size) {
456
+ GGML_ASSERT(t->type == GGML_TYPE_Q3_K);
457
+ GGML_ASSERT(interleave_block == 32);
458
+ GGML_ASSERT(QK_K == 256);
459
+
460
+ constexpr int nrows_interleaved = 32;
461
+
462
+ const uint32_t kmask1 = 0x03030303;
463
+ const uint32_t kmask2 = 0x0f0f0f0f;
464
+
465
+ const block_q3_K * src = (const block_q3_K *) data;
466
+
467
+ auto * dst = (spacemit_kernels::nrow_block_q3_k<32> *) t->data;
468
+
469
+ int nrow = ggml_nrows(t);
470
+ int nblocks = t->ne[0] / QK_K;
471
+
472
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q3_K));
473
+
474
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK_K != 0) {
475
+ return -1;
476
+ }
477
+
478
+ uint32_t b_scale_aux[4] = { 0 };
479
+ uint8_t qs_aux[256] = { 0 };
480
+
481
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
482
+ for (int64_t x = 0; x < nblocks; x++) {
483
+ for (int i = 0; i < nrows_interleaved; i++) {
484
+ const block_q3_K * src_block = &src[(b + i) * nblocks + x];
485
+
486
+ uint32_t * auxs = b_scale_aux;
487
+ int8_t * scale = (int8_t *) auxs;
488
+ memcpy(auxs, src_block->scales, 12);
489
+
490
+ uint32_t tmp = auxs[2];
491
+ auxs[2] = ((auxs[0] >> 4) & kmask2) | (((tmp >> 4) & kmask1) << 4);
492
+ auxs[3] = ((auxs[1] >> 4) & kmask2) | (((tmp >> 6) & kmask1) << 4);
493
+ auxs[0] = (auxs[0] & kmask2) | (((tmp >> 0) & kmask1) << 4);
494
+ auxs[1] = (auxs[1] & kmask2) | (((tmp >> 2) & kmask1) << 4);
495
+
496
+ for (int j = 0; j < 16; j++) {
497
+ dst->scales[j * nrows_interleaved + i] = scale[j] - 32;
498
+ }
499
+
500
+ for (int k = 0; k < 4; k++) {
501
+ for (int j = 0; j < 32; j++) {
502
+ qs_aux[k * 32 + j] = (src_block->qs[j] >> (2 * k)) & 0x03;
503
+ }
504
+ }
505
+
506
+ for (int k = 0; k < 4; k++) {
507
+ for (int j = 0; j < 32; j++) {
508
+ qs_aux[k * 32 + j + 128] = (src_block->qs[j + 32] >> (2 * k)) & 0x03;
509
+ }
510
+ }
511
+
512
+ // from nrows_interleaved * [2 * 32byte]
513
+ // to 4 * [nrows_interleaved * 16byte]
514
+ for (int k = 0; k < 4; k++) {
515
+ for (int j = 0; j < 16; j++) {
516
+ uint8_t qs0 = qs_aux[j + k * 64];
517
+ uint8_t qs16 = qs_aux[j + 16 + k * 64];
518
+ uint8_t qs32 = qs_aux[j + 32 + k * 64];
519
+ uint8_t qs48 = qs_aux[j + 48 + k * 64];
520
+
521
+ dst->qs[(k * nrows_interleaved + i) * 16 + j] =
522
+ (qs0 & 0x03) | ((qs16 & 0x03) << 2) | ((qs32 & 0x03) << 4) | ((qs48 & 0x03) << 6);
523
+ }
524
+ }
525
+
526
+ //memcpy(dst->hmask + i * 32, src_block->hmask, 32);
527
+
528
+ // from nrows_interleaved * [32byte]
529
+ // to 16 * [nrows_interleaved * uint16_t]
530
+ uint16_t * dst_mask = ((uint16_t *) dst->hmask) + i;
531
+ for (int j = 0; j < 16; j++, dst_mask += nrows_interleaved) {
532
+ uint8_t b_shift = j / 2;
533
+ uint8_t * b_mask_col = (uint8_t *) (src_block->hmask + (j % 2) * 16);
534
+ // b0 - b15
535
+ uint16_t msk_out_0 = 0;
536
+
537
+ for (int k = 0; k < 8; k++) {
538
+ msk_out_0 |= (uint16_t) ((b_mask_col[k] >> b_shift) & 0x01) << k;
539
+ }
540
+ for (int k = 8; k < 16; k++) {
541
+ msk_out_0 |= (uint16_t) ((b_mask_col[k] >> b_shift) & 0x01) << k;
542
+ }
543
+
544
+ dst_mask[0] = msk_out_0;
545
+ }
546
+
547
+ dst->scales16[i] = src_block->d;
548
+ }
549
+
550
+ dst++;
551
+ }
552
+ }
553
+
554
+ return 0;
555
+ }
556
+
557
+ static int repack_q4_0_to_q4_0_32_bl_ref(ggml_tensor * t,
558
+ int interleave_block,
559
+ const void * GGML_RESTRICT data,
560
+ size_t data_size) {
561
+ GGML_ASSERT(t->type == GGML_TYPE_Q4_0);
562
+ GGML_ASSERT(interleave_block == 32); // unused
563
+
564
+ constexpr int nrows_interleaved = 32;
565
+
566
+ block_q4_0x32 * dst = (block_q4_0x32 *) t->data;
567
+ const block_q4_0 * src = (const block_q4_0 *) data;
568
+ block_q4_0 dst_tmp[32];
569
+ int nrow = ggml_nrows(t);
570
+ int nblocks = t->ne[0] / QK4_0;
571
+
572
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q4_0));
573
+
574
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK4_0 != 0) {
575
+ return -1;
576
+ }
577
+
578
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
579
+ for (int64_t x = 0; x < nblocks; x++) {
580
+ for (int i = 0; i < nrows_interleaved; i++) {
581
+ dst_tmp[i] = src[x + i * nblocks];
582
+ }
583
+ *dst++ = make_block_q4_0x32(dst_tmp, interleave_block);
584
+ }
585
+ src += nrows_interleaved * nblocks;
586
+ }
587
+ return 0;
588
+
589
+ GGML_UNUSED(data_size);
590
+ }
591
+
592
+ static int repack_q4_0_to_q4_0_256_32_bl_ref(ggml_tensor * t,
593
+ int interleave_block,
594
+ const void * GGML_RESTRICT data,
595
+ size_t data_size) {
596
+ GGML_ASSERT(t->type == GGML_TYPE_Q4_0);
597
+ GGML_ASSERT(interleave_block == 32); // unused
598
+
599
+ constexpr int nrows_interleaved = 32;
600
+
601
+ block_q4_0x32x256 * dst = (block_q4_0x32x256 *) t->data;
602
+ const block_q4_0 * src = (const block_q4_0 *) data;
603
+ block_q4_0 dst_tmp[32];
604
+ int nrow = ggml_nrows(t);
605
+ int nblocks = t->ne[0] / QK4_0;
606
+
607
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q4_0));
608
+ GGML_ASSERT(nblocks % 8 == 0); // for 256-block interleaving
609
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK4_0 != 0) {
610
+ return -1;
611
+ }
612
+
613
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
614
+ for (int64_t x = 0; x < nblocks; x += 8) {
615
+ for (int j = 0; j < 8; j++) {
616
+ for (int i = 0; i < nrows_interleaved; i++) {
617
+ dst_tmp[i] = src[x + j + i * nblocks];
618
+ }
619
+ dst->blocks[j] = make_block_q4_0x32(dst_tmp, interleave_block);
620
+ }
621
+ dst++;
622
+ }
623
+ src += nrows_interleaved * nblocks;
624
+ }
625
+ return 0;
626
+
627
+ GGML_UNUSED(data_size);
628
+ }
629
+
630
+ static int repack_q4_0_to_q4_1_256_32_bl_ref(ggml_tensor * t,
631
+ int interleave_block,
632
+ const void * GGML_RESTRICT data,
633
+ size_t data_size) {
634
+ GGML_ASSERT(t->type == GGML_TYPE_Q4_1);
635
+ GGML_ASSERT(interleave_block == 32); // unused
636
+
637
+ constexpr int nrows_interleaved = 32;
638
+
639
+ block_q4_1x32x256 * dst = (block_q4_1x32x256 *) t->data;
640
+ const block_q4_1 * src = (const block_q4_1 *) data;
641
+ block_q4_1 dst_tmp[32];
642
+ int nrow = ggml_nrows(t);
643
+ int nblocks = t->ne[0] / QK4_0;
644
+
645
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q4_1));
646
+ GGML_ASSERT(nblocks % 8 == 0); // for 256-block interleaving
647
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK4_0 != 0) {
648
+ return -1;
649
+ }
650
+
651
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
652
+ for (int64_t x = 0; x < nblocks; x += 8) {
653
+ for (int j = 0; j < 8; j++) {
654
+ for (int i = 0; i < nrows_interleaved; i++) {
655
+ dst_tmp[i] = src[x + j + i * nblocks];
656
+ }
657
+
658
+ block_q4_0x32 * dst_block = &dst->blocks[j];
659
+ uint8_t * dst_zp = dst->zps + j * nrows_interleaved;
660
+
661
+ for (int i = 0; i < nrows_interleaved; i++) {
662
+ float d = GGML_FP16_TO_FP32(dst_tmp[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d);
663
+ float m = GGML_FP16_TO_FP32(dst_tmp[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.m);
664
+ float mid = -std::nearbyintf(m / d);
665
+ mid = std::min(15.0f, std::max(0.0f, mid));
666
+
667
+ dst_block->d[i] = GGML_FP32_TO_FP16(d);
668
+ dst_zp[i] = static_cast<uint8_t>(mid);
669
+ }
670
+
671
+ for (int i = 0; i < nrows_interleaved; i++) {
672
+ for (int k = 0; k < QK4_1 / 4; k++) {
673
+ dst_block->qs[i * QK4_1 / 2 + k] =
674
+ (dst_tmp[i].qs[k * 2] & 0x0F) | ((dst_tmp[i].qs[k * 2 + 1] & 0x0F) << 4);
675
+ }
676
+ }
677
+
678
+ for (int i = 0; i < nrows_interleaved; i++) {
679
+ for (int k = 0; k < QK4_1 / 4; k++) {
680
+ dst_block->qs[i * QK4_1 / 2 + QK4_1 / 4 + k] =
681
+ ((dst_tmp[i].qs[k * 2] & 0xF0) >> 4) | (dst_tmp[i].qs[k * 2 + 1] & 0xF0);
682
+ }
683
+ }
684
+ }
685
+ dst++;
686
+ }
687
+ src += nrows_interleaved * nblocks;
688
+ }
689
+ return 0;
690
+
691
+ GGML_UNUSED(data_size);
692
+ }
693
+
694
+ // RVV optimized version of repack_q4_0_to_q4_0_32_bl
695
+ // Eliminates the intermediate dst_tmp buffer and vectorizes nibble repack.
696
+ static int repack_q4_0_to_q4_0_32_bl(ggml_tensor * t,
697
+ int interleave_block,
698
+ const void * GGML_RESTRICT data,
699
+ size_t data_size) {
700
+ GGML_ASSERT(t->type == GGML_TYPE_Q4_0);
701
+ GGML_ASSERT(interleave_block == 32);
702
+
703
+ constexpr int nrows_interleaved = 32;
704
+ constexpr int qs_bytes = QK4_0 / 2; // 16
705
+
706
+ block_q4_0x32 * dst = (block_q4_0x32 *) t->data;
707
+ const block_q4_0 * src = (const block_q4_0 *) data;
708
+ int nrow = ggml_nrows(t);
709
+ int nblocks = t->ne[0] / QK4_0;
710
+
711
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q4_0));
712
+
713
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK4_0 != 0) {
714
+ return -1;
715
+ }
716
+
717
+ const ptrdiff_t row_stride = (ptrdiff_t) nblocks * sizeof(block_q4_0);
718
+
719
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
720
+ for (int64_t x = 0; x < nblocks; x++) {
721
+ const block_q4_0 * col_src = src + x;
722
+
723
+ // --- 1) Gather 32 scale values (ggml_half d) with stride load ---
724
+ // d is at offset 0 of each block_q4_0, stride between rows = row_stride
725
+ {
726
+ const uint8_t * d_base = (const uint8_t *) &col_src->d;
727
+ ggml_half * d_dst = dst->d;
728
+ size_t remaining = 32;
729
+ size_t offset = 0;
730
+ while (remaining > 0) {
731
+ size_t vl = __riscv_vsetvl_e16m1(remaining);
732
+ vuint16m1_t vd =
733
+ __riscv_vlse16_v_u16m1((const uint16_t *) (d_base + offset * row_stride), row_stride, vl);
734
+ __riscv_vse16_v_u16m1((uint16_t *) (d_dst + offset), vd, vl);
735
+ offset += vl;
736
+ remaining -= vl;
737
+ }
738
+ }
739
+
740
+ // --- 2) Nibble repack qs for each of the 32 rows ---
741
+ // For each row i:
742
+ // src qs[16]: [b0|b16] [b1|b17] ... [b15|b31] (lo nibble = b_j, hi nibble = b_{j+16})
743
+ // dst qs low 8B: (qs[2j] & 0x0F) | ((qs[2j+1] & 0x0F) << 4) for j=0..7
744
+ // dst qs high 8B: ((qs[2j] >> 4)) | (qs[2j+1] & 0xF0) for j=0..7
745
+ {
746
+ const size_t vl8 = __riscv_vsetvl_e8m1(8);
747
+ for (int i = 0; i < 32; i++) {
748
+ const uint8_t * sq = col_src[i * nblocks].qs;
749
+ uint8_t * dq = dst->qs + i * qs_bytes;
750
+
751
+ // stride-2 load to separate even/odd bytes
752
+ vuint8m1_t v_even = __riscv_vlse8_v_u8m1(sq, 2, vl8); // qs[0], qs[2], ..., qs[14]
753
+ vuint8m1_t v_odd = __riscv_vlse8_v_u8m1(sq + 1, 2, vl8); // qs[1], qs[3], ..., qs[15]
754
+
755
+ // low nibble part: (even & 0x0F) | ((odd & 0x0F) << 4)
756
+ vuint8m1_t v_even_lo = __riscv_vand_vx_u8m1(v_even, 0x0F, vl8);
757
+ vuint8m1_t v_odd_lo = __riscv_vand_vx_u8m1(v_odd, 0x0F, vl8);
758
+ vuint8m1_t v_lo = __riscv_vor_vv_u8m1(v_even_lo, __riscv_vsll_vx_u8m1(v_odd_lo, 4, vl8), vl8);
759
+
760
+ // high nibble part: (even >> 4) | (odd & 0xF0)
761
+ vuint8m1_t v_even_hi = __riscv_vsrl_vx_u8m1(v_even, 4, vl8);
762
+ vuint8m1_t v_odd_hi = __riscv_vand_vx_u8m1(v_odd, 0xF0, vl8);
763
+ vuint8m1_t v_hi = __riscv_vor_vv_u8m1(v_even_hi, v_odd_hi, vl8);
764
+
765
+ __riscv_vse8_v_u8m1(dq, v_lo, vl8);
766
+ __riscv_vse8_v_u8m1(dq + 8, v_hi, vl8);
767
+ }
768
+ }
769
+
770
+ dst++;
771
+ }
772
+ src += nrows_interleaved * nblocks;
773
+ }
774
+ return 0;
775
+
776
+ GGML_UNUSED(data_size);
777
+ }
778
+
779
+ static int repack_q4_1_to_q4_1_32_bl_ref(ggml_tensor * t,
780
+ int interleave_block,
781
+ const void * GGML_RESTRICT data,
782
+ size_t data_size) {
783
+ GGML_ASSERT(t->type == GGML_TYPE_Q4_1);
784
+ GGML_ASSERT(interleave_block == 32); // unused
785
+
786
+ constexpr int nrows_interleaved = 32;
787
+
788
+ block_q4_1x32 * dst = (block_q4_1x32 *) t->data;
789
+ const block_q4_1 * src = (const block_q4_1 *) data;
790
+ block_q4_1 dst_tmp[32];
791
+ int nrow = ggml_nrows(t);
792
+ int nblocks = t->ne[0] / QK4_1;
793
+
794
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q4_1));
795
+
796
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK4_1 != 0) {
797
+ return -1;
798
+ }
799
+
800
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
801
+ for (int64_t x = 0; x < nblocks; x++) {
802
+ for (int i = 0; i < nrows_interleaved; i++) {
803
+ dst_tmp[i] = src[x + i * nblocks];
804
+ }
805
+ *dst++ = make_block_q4_1x32(dst_tmp, interleave_block);
806
+ }
807
+ src += nrows_interleaved * nblocks;
808
+ }
809
+ return 0;
810
+
811
+ GGML_UNUSED(data_size);
812
+ }
813
+
814
+ // RVV optimized version of repack_q4_1_to_q4_1_32_bl
815
+ // Eliminates the intermediate dst_tmp buffer and vectorizes nibble repack + zp computation.
816
+ static int repack_q4_1_to_q4_1_32_bl(ggml_tensor * t,
817
+ int interleave_block,
818
+ const void * GGML_RESTRICT data,
819
+ size_t data_size) {
820
+ GGML_ASSERT(t->type == GGML_TYPE_Q4_1);
821
+ GGML_ASSERT(interleave_block == 32);
822
+
823
+ constexpr int nrows_interleaved = 32;
824
+ constexpr int qs_bytes = QK4_1 / 2; // 16
825
+
826
+ block_q4_1x32 * dst = (block_q4_1x32 *) t->data;
827
+ const block_q4_1 * src = (const block_q4_1 *) data;
828
+ int nrow = ggml_nrows(t);
829
+ int nblocks = t->ne[0] / QK4_1;
830
+
831
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q4_1));
832
+
833
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK4_1 != 0) {
834
+ return -1;
835
+ }
836
+
837
+ const ptrdiff_t row_stride = (ptrdiff_t) nblocks * sizeof(block_q4_1);
838
+
839
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
840
+ for (int64_t x = 0; x < nblocks; x++) {
841
+ const block_q4_1 * col_src = src + x;
842
+
843
+ // --- 1) Gather d and m, compute zp = clamp(nearbyint(-m/d), 0, 15) ---
844
+ // block_q4_1 layout: [d(f16), m(f16), qs[16]]
845
+ // d is at byte offset 0, m is at byte offset 2 from each block start
846
+ {
847
+ const uint8_t * dm_base = (const uint8_t *) &col_src->GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d;
848
+ ggml_half * d_dst = dst->d;
849
+ uint8_t * zp_dst = dst->zp;
850
+ size_t remaining = 32;
851
+ size_t offset = 0;
852
+ while (remaining > 0) {
853
+ size_t vl = __riscv_vsetvl_e16m1(remaining);
854
+
855
+ // stride load d (f16) from each row
856
+ vuint16m1_t vd_raw =
857
+ __riscv_vlse16_v_u16m1((const uint16_t *) (dm_base + offset * row_stride), row_stride, vl);
858
+ __riscv_vse16_v_u16m1((uint16_t *) (d_dst + offset), vd_raw, vl);
859
+
860
+ // stride load m (f16) from each row (offset +2 bytes from d)
861
+ vuint16m1_t vm_raw =
862
+ __riscv_vlse16_v_u16m1((const uint16_t *) (dm_base + 2 + offset * row_stride), row_stride, vl);
863
+
864
+ // convert to f32 for zp computation: zp = nearbyint(-m / d)
865
+ vfloat16m1_t vd_f16 = __riscv_vreinterpret_v_u16m1_f16m1(vd_raw);
866
+ vfloat16m1_t vm_f16 = __riscv_vreinterpret_v_u16m1_f16m1(vm_raw);
867
+
868
+ // -m / d in f16 directly (SpaceMIT X60 supports f16 arithmetic)
869
+ vfloat16m1_t v_neg_m = __riscv_vfneg_v_f16m1(vm_f16, vl);
870
+ vfloat16m1_t v_ratio = __riscv_vfdiv_vv_f16m1(v_neg_m, vd_f16, vl);
871
+
872
+ // Convert to f32 for nearbyint, then clamp
873
+ vfloat32m2_t v_ratio_f32 = __riscv_vfwcvt_f_f_v_f32m2(v_ratio, vl);
874
+
875
+ // Use integer rounding: convert f32 -> int (rounds to nearest)
876
+ vint32m2_t v_zp_i32 = __riscv_vfcvt_x_f_v_i32m2(v_ratio_f32, vl);
877
+
878
+ // clamp to [0, 15]
879
+ v_zp_i32 = __riscv_vmax_vx_i32m2(v_zp_i32, 0, vl);
880
+ v_zp_i32 = __riscv_vmin_vx_i32m2(v_zp_i32, 15, vl);
881
+
882
+ // narrow i32 -> u8
883
+ vint16m1_t v_zp_i16 = __riscv_vncvt_x_x_w_i16m1(v_zp_i32, vl);
884
+ vint8mf2_t v_zp_i8 = __riscv_vncvt_x_x_w_i8mf2(v_zp_i16, vl);
885
+ vuint8mf2_t v_zp_u8 = __riscv_vreinterpret_v_i8mf2_u8mf2(v_zp_i8);
886
+ __riscv_vse8_v_u8mf2(zp_dst + offset, v_zp_u8, vl);
887
+
888
+ offset += vl;
889
+ remaining -= vl;
890
+ }
891
+ }
892
+
893
+ // --- 2) Nibble repack qs for each of the 32 rows ---
894
+ {
895
+ const size_t vl8 = __riscv_vsetvl_e8m1(8);
896
+ for (int i = 0; i < 32; i++) {
897
+ const uint8_t * sq = col_src[i * nblocks].qs;
898
+ uint8_t * dq = dst->qs + i * qs_bytes;
899
+
900
+ // stride-2 load to separate even/odd bytes
901
+ vuint8m1_t v_even = __riscv_vlse8_v_u8m1(sq, 2, vl8);
902
+ vuint8m1_t v_odd = __riscv_vlse8_v_u8m1(sq + 1, 2, vl8);
903
+
904
+ // low nibble part: (even & 0x0F) | ((odd & 0x0F) << 4)
905
+ vuint8m1_t v_even_lo = __riscv_vand_vx_u8m1(v_even, 0x0F, vl8);
906
+ vuint8m1_t v_odd_lo = __riscv_vand_vx_u8m1(v_odd, 0x0F, vl8);
907
+ vuint8m1_t v_lo = __riscv_vor_vv_u8m1(v_even_lo, __riscv_vsll_vx_u8m1(v_odd_lo, 4, vl8), vl8);
908
+
909
+ // high nibble part: (even >> 4) | (odd & 0xF0)
910
+ vuint8m1_t v_even_hi = __riscv_vsrl_vx_u8m1(v_even, 4, vl8);
911
+ vuint8m1_t v_odd_hi = __riscv_vand_vx_u8m1(v_odd, 0xF0, vl8);
912
+ vuint8m1_t v_hi = __riscv_vor_vv_u8m1(v_even_hi, v_odd_hi, vl8);
913
+
914
+ __riscv_vse8_v_u8m1(dq, v_lo, vl8);
915
+ __riscv_vse8_v_u8m1(dq + 8, v_hi, vl8);
916
+ }
917
+ }
918
+
919
+ dst++;
920
+ }
921
+ src += nrows_interleaved * nblocks;
922
+ }
923
+ return 0;
924
+
925
+ GGML_UNUSED(data_size);
926
+ }
927
+
928
+ static int repack_q4_k_to_q4_1_32_bl(ggml_tensor * t,
929
+ int interleave_block,
930
+ const void * GGML_RESTRICT data,
931
+ size_t data_size) {
932
+ GGML_ASSERT(t->type == GGML_TYPE_Q4_K);
933
+ GGML_ASSERT(interleave_block == 32);
934
+ GGML_ASSERT(QK_K / QK4_1 == 8);
935
+
936
+ constexpr int nrows_interleaved = 32;
937
+
938
+ block_q4_1x32 * dst = (block_q4_1x32 *) t->data;
939
+ const block_q4_K * src = (const block_q4_K *) data;
940
+ block_q4_1 dst_tmp[32];
941
+ int nrow = ggml_nrows(t);
942
+ int nblocks = t->ne[0] / QK_K;
943
+
944
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK_K != 0) {
945
+ return -1;
946
+ }
947
+
948
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
949
+ for (int64_t x = 0; x < nblocks; x++) {
950
+ for (int j = 0; j < 8; j++) {
951
+ for (int i = 0; i < nrows_interleaved; i++) {
952
+ uint8_t sc, m;
953
+ const float d = GGML_FP16_TO_FP32(src[x + i * nblocks].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d);
954
+ const float min =
955
+ GGML_FP16_TO_FP32(src[x + i * nblocks].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.dmin);
956
+ get_scale_min_k4(j, src[x + i * nblocks].scales, &sc, &m);
957
+ const float d1 = d * sc;
958
+ const float m1 = min * m;
959
+
960
+ dst_tmp[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d = GGML_FP32_TO_FP16(d1);
961
+ dst_tmp[i].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.m = GGML_FP32_TO_FP16(-m1);
962
+ // src -> [b0, b32] [b1, b33] ... [b31, b63]
963
+ // dst -> [b0, b16] [b1, b17] ... [b15, b31] [b32, b48] [b33, b49] ... [b47, b63]
964
+ const uint8_t * q = src[x + i * nblocks].qs + (j / 2) * QK4_1;
965
+ if (j % 2 == 0) {
966
+ for (int ii = 0; ii < 16; ii++) {
967
+ dst_tmp[i].qs[ii] = (q[ii] & 0x0F) | ((q[ii + 16] & 0x0F) << 4);
968
+ }
969
+ } else {
970
+ for (int ii = 0; ii < 16; ii++) {
971
+ dst_tmp[i].qs[ii] = ((q[ii] & 0xF0) >> 4) | (q[ii + 16] & 0xF0);
972
+ }
973
+ }
974
+ }
975
+ *dst++ = make_block_q4_1x32(dst_tmp, interleave_block);
976
+ }
977
+ }
978
+ src += nrows_interleaved * nblocks;
979
+ }
980
+ return 0;
981
+
982
+ GGML_UNUSED(data_size);
983
+ }
984
+
985
+ static int repack_q6_k_to_q8_0_32_bl_ref(ggml_tensor * t,
986
+ int interleave_block,
987
+ const void * GGML_RESTRICT data,
988
+ size_t data_size) {
989
+ GGML_ASSERT(t->type == GGML_TYPE_Q6_K);
990
+ GGML_ASSERT(interleave_block == 32);
991
+ GGML_ASSERT(QK_K / QK4_1 == 8);
992
+
993
+ constexpr int nrows_interleaved = 32;
994
+
995
+ block_q8_0x32 * dst = (block_q8_0x32 *) t->data;
996
+ const block_q6_K * src = (const block_q6_K *) data;
997
+ block_q8_0 dst_tmp[32];
998
+ int8_t aux8[QK4_1];
999
+ int nrow = ggml_nrows(t);
1000
+ int nblocks = t->ne[0] / QK_K;
1001
+
1002
+ if (t->ne[0] % QK_K != 0) {
1003
+ return -1;
1004
+ }
1005
+
1006
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
1007
+ int64_t nrow_real = std::min((int64_t) nrow - b, (int64_t) nrows_interleaved);
1008
+ for (int64_t x = 0; x < nblocks; x++) {
1009
+ for (int bi = 0; bi < 8; bi++) {
1010
+ int i = 0;
1011
+ for (; i < nrow_real; i++) {
1012
+ const uint8_t * q4 = src[x + i * nblocks].ql;
1013
+ const uint8_t * qh = src[x + i * nblocks].qh;
1014
+ const int8_t * scales = src[x + i * nblocks].scales;
1015
+ float d = GGML_FP16_TO_FP32(src[x + i * nblocks].d);
1016
+
1017
+ q4 += 64 * (bi / 4);
1018
+ qh += 32 * (bi / 4);
1019
+ int8_t * GGML_RESTRICT a = aux8;
1020
+
1021
+ int8_t bi_idx = bi % 4;
1022
+
1023
+ if (bi_idx == 0) {
1024
+ for (int l = 0; l < 32; ++l) {
1025
+ a[l] = (int8_t) ((q4[l] & 0xF) | (((qh[l] >> 0) & 3) << 4)) - 32;
1026
+ }
1027
+ } else if (bi_idx == 1) {
1028
+ for (int l = 0; l < 32; ++l) {
1029
+ a[l] = (int8_t) ((q4[l + 32] & 0xF) | (((qh[l] >> 2) & 3) << 4)) - 32;
1030
+ }
1031
+ } else if (bi_idx == 2) {
1032
+ for (int l = 0; l < 32; ++l) {
1033
+ a[l] = (int8_t) ((q4[l + 0] >> 4) | (((qh[l] >> 4) & 3) << 4)) - 32;
1034
+ }
1035
+ } else if (bi_idx == 3) {
1036
+ for (int l = 0; l < 32; ++l) {
1037
+ a[l] = (int8_t) ((q4[l + 32] >> 4) | (((qh[l] >> 6) & 3) << 4)) - 32;
1038
+ }
1039
+ }
1040
+ a = aux8;
1041
+
1042
+ float a_max_abs = 0.0f;
1043
+ float scale_0 = scales[bi * 2 + 0] * d;
1044
+ float scale_1 = scales[bi * 2 + 1] * d;
1045
+ for (int l = 0; l < 16; ++l) {
1046
+ a_max_abs = std::max(a_max_abs, std::abs(a[l] * scale_0));
1047
+ }
1048
+
1049
+ for (int l = 16; l < 32; ++l) {
1050
+ a_max_abs = std::max(a_max_abs, std::abs(a[l] * scale_1));
1051
+ }
1052
+
1053
+ float reflect_scale = a_max_abs / ((1 << 7) - 1);
1054
+ float reflect_scale_0 = scale_0 / reflect_scale;
1055
+ float reflect_scale_1 = scale_1 / reflect_scale;
1056
+
1057
+ for (int l = 0; l < 16; ++l) {
1058
+ float a_temp = std::clamp(std::nearbyintf(a[l] * reflect_scale_0), -128.0f, 127.0f);
1059
+ a[l] = (int8_t) (a_temp);
1060
+ }
1061
+
1062
+ for (int l = 16; l < 32; ++l) {
1063
+ float a_temp = std::clamp(std::nearbyintf(a[l] * reflect_scale_1), -128.0f, 127.0f);
1064
+ a[l] = (int8_t) (a_temp);
1065
+ }
1066
+
1067
+ dst_tmp[i].d = GGML_FP32_TO_FP16(reflect_scale);
1068
+
1069
+ memcpy(dst_tmp[i].qs, a, 32 * sizeof(int8_t));
1070
+ }
1071
+
1072
+ for (; i < nrows_interleaved; i++) {
1073
+ memset(&dst_tmp[i], 0, sizeof(block_q8_0));
1074
+ }
1075
+
1076
+ *dst++ = make_block_q8_0x32(dst_tmp, interleave_block);
1077
+ }
1078
+ }
1079
+ src += nrows_interleaved * nblocks;
1080
+ }
1081
+ return 0;
1082
+
1083
+ GGML_UNUSED(data_size);
1084
+ }
1085
+
1086
+ // RVV optimized version of repack_q6_k_to_q8_0_32_bl
1087
+ // Vectorizes the Q6_K dequant -> requant pipeline using RVV intrinsics.
1088
+ // For each sub-block (bi), dequant 32 Q6_K values to int6 -> apply two sub-block scales ->
1089
+ // find max abs -> compute reflect_scale -> requant to int8 -> gather d with stride load.
1090
+ static int repack_q6_k_to_q8_0_32_bl(ggml_tensor * t,
1091
+ int interleave_block,
1092
+ const void * GGML_RESTRICT data,
1093
+ size_t data_size) {
1094
+ GGML_ASSERT(t->type == GGML_TYPE_Q6_K);
1095
+ GGML_ASSERT(interleave_block == 32);
1096
+ GGML_ASSERT(QK_K / QK4_1 == 8);
1097
+
1098
+ constexpr int nrows_interleaved = 32;
1099
+
1100
+ block_q8_0x32 * dst = (block_q8_0x32 *) t->data;
1101
+ const block_q6_K * src = (const block_q6_K *) data;
1102
+ int nrow = ggml_nrows(t);
1103
+ int nblocks = t->ne[0] / QK_K;
1104
+
1105
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK_K != 0) {
1106
+ return -1;
1107
+ }
1108
+
1109
+ const ptrdiff_t row_stride = (ptrdiff_t) nblocks * sizeof(block_q6_K);
1110
+
1111
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
1112
+ for (int64_t x = 0; x < nblocks; x++) {
1113
+ for (int bi = 0; bi < 8; bi++) {
1114
+ // --- 1) Gather 32 d values with stride load ---
1115
+ // We need to compute reflect_scale per row first, so gather d later.
1116
+ // Process each row: dequant Q6_K sub-block -> requant to Q8_0
1117
+ for (int i = 0; i < nrows_interleaved; i++) {
1118
+ const block_q6_K * src_blk = &src[x + i * nblocks];
1119
+ const uint8_t * q4 = src_blk->ql + 64 * (bi / 4);
1120
+ const uint8_t * qh = src_blk->qh + 32 * (bi / 4);
1121
+ const int8_t * scales = src_blk->scales;
1122
+ float d = GGML_FP16_TO_FP32(src_blk->d);
1123
+
1124
+ int8_t bi_idx = bi % 4;
1125
+
1126
+ // --- Dequant 32 Q6_K values to int6 (range [-32, 31]) using RVV ---
1127
+ // vl = 32 for e8m2 (VLEN=256) or loop for smaller VLEN
1128
+ const size_t vl16 = __riscv_vsetvl_e8m1(16);
1129
+
1130
+ vint8m1_t va_lo, va_hi; // 16 elements each
1131
+
1132
+ if (bi_idx == 0) {
1133
+ // a[l] = (q4[l] & 0xF) | (((qh[l] >> 0) & 3) << 4) - 32
1134
+ vuint8m1_t vq4_lo = __riscv_vle8_v_u8m1(q4, vl16);
1135
+ vuint8m1_t vq4_hi = __riscv_vle8_v_u8m1(q4 + 16, vl16);
1136
+ vuint8m1_t vqh_lo = __riscv_vle8_v_u8m1(qh, vl16);
1137
+ vuint8m1_t vqh_hi = __riscv_vle8_v_u8m1(qh + 16, vl16);
1138
+
1139
+ vuint8m1_t vlo4_lo = __riscv_vand_vx_u8m1(vq4_lo, 0x0F, vl16);
1140
+ vuint8m1_t vlo4_hi = __riscv_vand_vx_u8m1(vq4_hi, 0x0F, vl16);
1141
+ vuint8m1_t vh_lo = __riscv_vsll_vx_u8m1(__riscv_vand_vx_u8m1(vqh_lo, 0x03, vl16), 4, vl16);
1142
+ vuint8m1_t vh_hi = __riscv_vsll_vx_u8m1(__riscv_vand_vx_u8m1(vqh_hi, 0x03, vl16), 4, vl16);
1143
+
1144
+ vuint8m1_t vcomb_lo = __riscv_vor_vv_u8m1(vlo4_lo, vh_lo, vl16);
1145
+ vuint8m1_t vcomb_hi = __riscv_vor_vv_u8m1(vlo4_hi, vh_hi, vl16);
1146
+
1147
+ va_lo = __riscv_vsub_vx_i8m1(__riscv_vreinterpret_v_u8m1_i8m1(vcomb_lo), 32, vl16);
1148
+ va_hi = __riscv_vsub_vx_i8m1(__riscv_vreinterpret_v_u8m1_i8m1(vcomb_hi), 32, vl16);
1149
+ } else if (bi_idx == 1) {
1150
+ // a[l] = (q4[l+32] & 0xF) | (((qh[l] >> 2) & 3) << 4) - 32
1151
+ vuint8m1_t vq4_lo = __riscv_vle8_v_u8m1(q4 + 32, vl16);
1152
+ vuint8m1_t vq4_hi = __riscv_vle8_v_u8m1(q4 + 48, vl16);
1153
+ vuint8m1_t vqh_lo = __riscv_vle8_v_u8m1(qh, vl16);
1154
+ vuint8m1_t vqh_hi = __riscv_vle8_v_u8m1(qh + 16, vl16);
1155
+
1156
+ vuint8m1_t vlo4_lo = __riscv_vand_vx_u8m1(vq4_lo, 0x0F, vl16);
1157
+ vuint8m1_t vlo4_hi = __riscv_vand_vx_u8m1(vq4_hi, 0x0F, vl16);
1158
+ vuint8m1_t vh_lo = __riscv_vsll_vx_u8m1(
1159
+ __riscv_vand_vx_u8m1(__riscv_vsrl_vx_u8m1(vqh_lo, 2, vl16), 0x03, vl16), 4, vl16);
1160
+ vuint8m1_t vh_hi = __riscv_vsll_vx_u8m1(
1161
+ __riscv_vand_vx_u8m1(__riscv_vsrl_vx_u8m1(vqh_hi, 2, vl16), 0x03, vl16), 4, vl16);
1162
+
1163
+ vuint8m1_t vcomb_lo = __riscv_vor_vv_u8m1(vlo4_lo, vh_lo, vl16);
1164
+ vuint8m1_t vcomb_hi = __riscv_vor_vv_u8m1(vlo4_hi, vh_hi, vl16);
1165
+
1166
+ va_lo = __riscv_vsub_vx_i8m1(__riscv_vreinterpret_v_u8m1_i8m1(vcomb_lo), 32, vl16);
1167
+ va_hi = __riscv_vsub_vx_i8m1(__riscv_vreinterpret_v_u8m1_i8m1(vcomb_hi), 32, vl16);
1168
+ } else if (bi_idx == 2) {
1169
+ // a[l] = (q4[l] >> 4) | (((qh[l] >> 4) & 3) << 4) - 32
1170
+ vuint8m1_t vq4_lo = __riscv_vle8_v_u8m1(q4, vl16);
1171
+ vuint8m1_t vq4_hi = __riscv_vle8_v_u8m1(q4 + 16, vl16);
1172
+ vuint8m1_t vqh_lo = __riscv_vle8_v_u8m1(qh, vl16);
1173
+ vuint8m1_t vqh_hi = __riscv_vle8_v_u8m1(qh + 16, vl16);
1174
+
1175
+ vuint8m1_t vhi4_lo = __riscv_vsrl_vx_u8m1(vq4_lo, 4, vl16);
1176
+ vuint8m1_t vhi4_hi = __riscv_vsrl_vx_u8m1(vq4_hi, 4, vl16);
1177
+ vuint8m1_t vh_lo = __riscv_vsll_vx_u8m1(
1178
+ __riscv_vand_vx_u8m1(__riscv_vsrl_vx_u8m1(vqh_lo, 4, vl16), 0x03, vl16), 4, vl16);
1179
+ vuint8m1_t vh_hi = __riscv_vsll_vx_u8m1(
1180
+ __riscv_vand_vx_u8m1(__riscv_vsrl_vx_u8m1(vqh_hi, 4, vl16), 0x03, vl16), 4, vl16);
1181
+
1182
+ vuint8m1_t vcomb_lo = __riscv_vor_vv_u8m1(vhi4_lo, vh_lo, vl16);
1183
+ vuint8m1_t vcomb_hi = __riscv_vor_vv_u8m1(vhi4_hi, vh_hi, vl16);
1184
+
1185
+ va_lo = __riscv_vsub_vx_i8m1(__riscv_vreinterpret_v_u8m1_i8m1(vcomb_lo), 32, vl16);
1186
+ va_hi = __riscv_vsub_vx_i8m1(__riscv_vreinterpret_v_u8m1_i8m1(vcomb_hi), 32, vl16);
1187
+ } else { // bi_idx == 3
1188
+ // a[l] = (q4[l+32] >> 4) | (((qh[l] >> 6) & 3) << 4) - 32
1189
+ vuint8m1_t vq4_lo = __riscv_vle8_v_u8m1(q4 + 32, vl16);
1190
+ vuint8m1_t vq4_hi = __riscv_vle8_v_u8m1(q4 + 48, vl16);
1191
+ vuint8m1_t vqh_lo = __riscv_vle8_v_u8m1(qh, vl16);
1192
+ vuint8m1_t vqh_hi = __riscv_vle8_v_u8m1(qh + 16, vl16);
1193
+
1194
+ vuint8m1_t vhi4_lo = __riscv_vsrl_vx_u8m1(vq4_lo, 4, vl16);
1195
+ vuint8m1_t vhi4_hi = __riscv_vsrl_vx_u8m1(vq4_hi, 4, vl16);
1196
+ vuint8m1_t vh_lo = __riscv_vsll_vx_u8m1(
1197
+ __riscv_vand_vx_u8m1(__riscv_vsrl_vx_u8m1(vqh_lo, 6, vl16), 0x03, vl16), 4, vl16);
1198
+ vuint8m1_t vh_hi = __riscv_vsll_vx_u8m1(
1199
+ __riscv_vand_vx_u8m1(__riscv_vsrl_vx_u8m1(vqh_hi, 6, vl16), 0x03, vl16), 4, vl16);
1200
+
1201
+ vuint8m1_t vcomb_lo = __riscv_vor_vv_u8m1(vhi4_lo, vh_lo, vl16);
1202
+ vuint8m1_t vcomb_hi = __riscv_vor_vv_u8m1(vhi4_hi, vh_hi, vl16);
1203
+
1204
+ va_lo = __riscv_vsub_vx_i8m1(__riscv_vreinterpret_v_u8m1_i8m1(vcomb_lo), 32, vl16);
1205
+ va_hi = __riscv_vsub_vx_i8m1(__riscv_vreinterpret_v_u8m1_i8m1(vcomb_hi), 32, vl16);
1206
+ }
1207
+
1208
+ // --- Widen to i16 for scaled abs computation ---
1209
+ float scale_0 = scales[bi * 2 + 0] * d;
1210
+ float scale_1 = scales[bi * 2 + 1] * d;
1211
+
1212
+ // Widen i8 -> i16 -> f32 for abs*scale computation
1213
+ vint16m2_t va_lo_w = __riscv_vsext_vf2_i16m2(va_lo, vl16);
1214
+ vint16m2_t va_hi_w = __riscv_vsext_vf2_i16m2(va_hi, vl16);
1215
+
1216
+ // Compute |a[l] * scale_0| for lo half, |a[l] * scale_1| for hi half
1217
+ vfloat32m4_t vf_lo = __riscv_vfcvt_f_x_v_f32m4(__riscv_vsext_vf2_i32m4(va_lo_w, vl16), vl16);
1218
+ vfloat32m4_t vf_hi = __riscv_vfcvt_f_x_v_f32m4(__riscv_vsext_vf2_i32m4(va_hi_w, vl16), vl16);
1219
+
1220
+ vfloat32m4_t vabs_lo = __riscv_vfabs_v_f32m4(__riscv_vfmul_vf_f32m4(vf_lo, scale_0, vl16), vl16);
1221
+ vfloat32m4_t vabs_hi = __riscv_vfabs_v_f32m4(__riscv_vfmul_vf_f32m4(vf_hi, scale_1, vl16), vl16);
1222
+
1223
+ // Find max abs across both halves
1224
+ vfloat32m4_t vabs_max = __riscv_vfmax_vv_f32m4(vabs_lo, vabs_hi, vl16);
1225
+
1226
+ // Reduce to scalar max
1227
+ vfloat32m1_t vzero = __riscv_vfmv_v_f_f32m1(0.0f, 1);
1228
+ vfloat32m1_t vmax_red = __riscv_vfredmax_vs_f32m4_f32m1(vabs_max, vzero, vl16);
1229
+ float a_max_abs = __riscv_vfmv_f_s_f32m1_f32(vmax_red);
1230
+
1231
+ float reflect_scale = a_max_abs / 127.0f;
1232
+ float reflect_scale_0 = scale_0 / reflect_scale;
1233
+ float reflect_scale_1 = scale_1 / reflect_scale;
1234
+
1235
+ // --- Requant: a[l] = clamp(nearbyint(a[l] * reflect_scale_x), -128, 127) ---
1236
+ vfloat32m4_t vscaled_lo = __riscv_vfmul_vf_f32m4(vf_lo, reflect_scale_0, vl16);
1237
+ vfloat32m4_t vscaled_hi = __riscv_vfmul_vf_f32m4(vf_hi, reflect_scale_1, vl16);
1238
+
1239
+ // fcvt.x rounds to nearest (using current rounding mode)
1240
+ vint32m4_t vi_lo = __riscv_vfcvt_x_f_v_i32m4(vscaled_lo, vl16);
1241
+ vint32m4_t vi_hi = __riscv_vfcvt_x_f_v_i32m4(vscaled_hi, vl16);
1242
+
1243
+ // Clamp to [-128, 127]
1244
+ vi_lo = __riscv_vmax_vx_i32m4(vi_lo, -128, vl16);
1245
+ vi_lo = __riscv_vmin_vx_i32m4(vi_lo, 127, vl16);
1246
+ vi_hi = __riscv_vmax_vx_i32m4(vi_hi, -128, vl16);
1247
+ vi_hi = __riscv_vmin_vx_i32m4(vi_hi, 127, vl16);
1248
+
1249
+ // Narrow i32 -> i16 -> i8
1250
+ vint16m2_t vi16_lo = __riscv_vncvt_x_x_w_i16m2(vi_lo, vl16);
1251
+ vint16m2_t vi16_hi = __riscv_vncvt_x_x_w_i16m2(vi_hi, vl16);
1252
+ vint8m1_t vi8_lo = __riscv_vncvt_x_x_w_i8m1(vi16_lo, vl16);
1253
+ vint8m1_t vi8_hi = __riscv_vncvt_x_x_w_i8m1(vi16_hi, vl16);
1254
+
1255
+ // Store d and qs directly into dst block
1256
+ dst->d[i] = GGML_FP32_TO_FP16(reflect_scale);
1257
+ int8_t * dq = (int8_t *) dst->qs + i * QK8_0;
1258
+ __riscv_vse8_v_i8m1(dq, vi8_lo, vl16);
1259
+ __riscv_vse8_v_i8m1(dq + 16, vi8_hi, vl16);
1260
+ }
1261
+ dst++;
1262
+ }
1263
+ }
1264
+ src += nrows_interleaved * nblocks;
1265
+ }
1266
+ return 0;
1267
+
1268
+ GGML_UNUSED(data_size);
1269
+ }
1270
+
1271
+ static int repack_q8_0_to_q8_0_32_bl_ref(ggml_tensor * t,
1272
+ int interleave_block,
1273
+ const void * GGML_RESTRICT data,
1274
+ size_t data_size) {
1275
+ GGML_ASSERT(t->type == GGML_TYPE_Q8_0);
1276
+ GGML_ASSERT(interleave_block == 32); // unused
1277
+
1278
+ constexpr int nrows_interleaved = 32;
1279
+
1280
+ block_q8_0x32 * dst = (block_q8_0x32 *) t->data;
1281
+ const block_q8_0 * src = (const block_q8_0 *) data;
1282
+ block_q8_0 dst_tmp[32];
1283
+ int nrow = ggml_nrows(t);
1284
+ int nblocks = t->ne[0] / QK8_0;
1285
+
1286
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q8_0));
1287
+
1288
+ if (t->ne[0] % QK8_0 != 0) {
1289
+ return -1;
1290
+ }
1291
+
1292
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
1293
+ int64_t nrows_real = std::min((int64_t) nrow - b, (int64_t) nrows_interleaved);
1294
+ for (int64_t x = 0; x < nblocks; x++) {
1295
+ int i = 0;
1296
+ for (; i < nrows_real; i++) {
1297
+ dst_tmp[i] = src[x + i * nblocks];
1298
+ }
1299
+ for (; i < nrows_interleaved; i++) {
1300
+ memset(&dst_tmp[i], 0, sizeof(block_q8_0));
1301
+ }
1302
+ *dst++ = make_block_q8_0x32(dst_tmp, interleave_block);
1303
+ }
1304
+ src += nrows_interleaved * nblocks;
1305
+ }
1306
+ return 0;
1307
+
1308
+ GGML_UNUSED(data_size);
1309
+ }
1310
+
1311
+ // RVV optimized version of repack_q8_0_to_q8_0_32_bl
1312
+ // Eliminates the intermediate dst_tmp buffer and vectorizes scale gather + qs copy.
1313
+ static int repack_q8_0_to_q8_0_32_bl(ggml_tensor * t,
1314
+ int interleave_block,
1315
+ const void * GGML_RESTRICT data,
1316
+ size_t data_size) {
1317
+ GGML_ASSERT(t->type == GGML_TYPE_Q8_0);
1318
+ GGML_ASSERT(interleave_block == 32);
1319
+
1320
+ constexpr int nrows_interleaved = 32;
1321
+
1322
+ block_q8_0x32 * dst = (block_q8_0x32 *) t->data;
1323
+ const block_q8_0 * src = (const block_q8_0 *) data;
1324
+ int nrow = ggml_nrows(t);
1325
+ int nblocks = t->ne[0] / QK8_0;
1326
+
1327
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q8_0));
1328
+
1329
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK8_0 != 0) {
1330
+ return -1;
1331
+ }
1332
+
1333
+ const ptrdiff_t row_stride = (ptrdiff_t) nblocks * sizeof(block_q8_0);
1334
+
1335
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
1336
+ for (int64_t x = 0; x < nblocks; x++) {
1337
+ const block_q8_0 * col_src = src + x;
1338
+
1339
+ // --- 1) Gather 32 scale values (ggml_half d) with stride load ---
1340
+ {
1341
+ const uint8_t * d_base = (const uint8_t *) &col_src->d;
1342
+ ggml_half * d_dst = dst->d;
1343
+ size_t remaining = 32;
1344
+ size_t offset = 0;
1345
+ while (remaining > 0) {
1346
+ size_t vl = __riscv_vsetvl_e16m1(remaining);
1347
+ vuint16m1_t vd =
1348
+ __riscv_vlse16_v_u16m1((const uint16_t *) (d_base + offset * row_stride), row_stride, vl);
1349
+ __riscv_vse16_v_u16m1((uint16_t *) (d_dst + offset), vd, vl);
1350
+ offset += vl;
1351
+ remaining -= vl;
1352
+ }
1353
+ }
1354
+
1355
+ // --- 2) Copy qs for each of the 32 rows (32 bytes per row) ---
1356
+ {
1357
+ for (int i = 0; i < 32; i++) {
1358
+ const int8_t * sq = col_src[i * nblocks].qs;
1359
+ int8_t * dq = (int8_t *) dst->qs + i * QK8_0;
1360
+
1361
+ size_t len = QK8_0;
1362
+ size_t idx = 0;
1363
+ while (len > 0) {
1364
+ size_t vl = __riscv_vsetvl_e8m2(len);
1365
+ vint8m2_t vs = __riscv_vle8_v_i8m2(sq + idx, vl);
1366
+ __riscv_vse8_v_i8m2(dq + idx, vs, vl);
1367
+ idx += vl;
1368
+ len -= vl;
1369
+ }
1370
+ }
1371
+ }
1372
+
1373
+ dst++;
1374
+ }
1375
+ src += nrows_interleaved * nblocks;
1376
+ }
1377
+ return 0;
1378
+
1379
+ GGML_UNUSED(data_size);
1380
+ }
1381
+
1382
+ static void convert_mxfp4_to_5bit(const block_mxfp4 & src, spacemit_kernels::nrow_block_mxfp4<1> & dst) {
1383
+ dst.e[0] = src.e;
1384
+
1385
+ // Decode all 32 mxfp4 values to signed integers via kvalues_mxfp4
1386
+ int8_t vals[32];
1387
+ for (int j = 0; j < QK_MXFP4 / 2; j++) {
1388
+ vals[j] = kvalues_mxfp4[src.qs[j] & 0xF];
1389
+ vals[j + QK_MXFP4 / 2] = kvalues_mxfp4[src.qs[j] >> 4];
1390
+ }
1391
+
1392
+ // vals [b0, b1, b2, b3, ..., b30, b31]
1393
+ // Pack abs into qs with reorder: [b0,b1]..[b14,b15]..[b30,b31]
1394
+ for (int j = 0; j < QK_MXFP4 / 2; j++) {
1395
+ uint8_t lo0 = static_cast<uint8_t>(std::abs(vals[j * 2]));
1396
+ uint8_t lo1 = static_cast<uint8_t>(std::abs(vals[j * 2 + 1]));
1397
+ dst.qs[j] = (lo0 & 0x0F) | ((lo1 & 0x0F) << 4);
1398
+ }
1399
+
1400
+ // Pack sign bits into qh[4] (32 bits total, 1 bit per weight)
1401
+ // reorder: [0,1,2,...,15,16,17,...,31] after the qs reorder above
1402
+ uint32_t sign_bits = 0;
1403
+ for (int j = 0; j < 32; j++) {
1404
+ if (vals[j] < 0) {
1405
+ sign_bits |= (1u << j);
1406
+ }
1407
+ }
1408
+ memcpy(dst.qh, &sign_bits, 4);
1409
+ }
1410
+
1411
+ static spacemit_kernels::nrow_block_mxfp4<32> make_block_mxfp4x32(spacemit_kernels::nrow_block_mxfp4<1> * in,
1412
+ unsigned int blck_size_interleave) {
1413
+ spacemit_kernels::nrow_block_mxfp4<32> out;
1414
+ GGML_ASSERT(QK_MXFP4 / blck_size_interleave == 1);
1415
+ GGML_UNUSED(blck_size_interleave);
1416
+
1417
+ for (int i = 0; i < 32; i++) {
1418
+ out.e[i] = in[i].e[0];
1419
+ }
1420
+
1421
+ // qs: copy per-row 16 bytes
1422
+ for (int i = 0; i < 32; i++) {
1423
+ memcpy(out.qs + i * 16, in[i].qs, 16);
1424
+ }
1425
+
1426
+ // qh: copy per-row 4 bytes
1427
+ for (int i = 0; i < 32; i++) {
1428
+ memcpy(out.qh + i * 4, in[i].qh, 4);
1429
+ }
1430
+
1431
+ return out;
1432
+ }
1433
+
1434
+ static int repack_mxfp4_to_mxfp4_32_bl(ggml_tensor * t,
1435
+ int interleave_block,
1436
+ const void * GGML_RESTRICT data,
1437
+ size_t data_size) {
1438
+ GGML_ASSERT(t->type == GGML_TYPE_MXFP4);
1439
+ GGML_ASSERT(interleave_block == 32);
1440
+
1441
+ constexpr int nrows_interleaved = 32;
1442
+
1443
+ spacemit_kernels::nrow_block_mxfp4<32> * dst = (spacemit_kernels::nrow_block_mxfp4<32> *) t->data;
1444
+ const block_mxfp4 * src = (const block_mxfp4 *) data;
1445
+ spacemit_kernels::nrow_block_mxfp4<1> dst_tmp[32];
1446
+ int nrow = ggml_nrows(t);
1447
+ int nblocks = t->ne[0] / QK_MXFP4;
1448
+
1449
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_mxfp4));
1450
+
1451
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK_MXFP4 != 0) {
1452
+ return -1;
1453
+ }
1454
+
1455
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
1456
+ for (int64_t x = 0; x < nblocks; x++) {
1457
+ for (int i = 0; i < nrows_interleaved; i++) {
1458
+ convert_mxfp4_to_5bit(src[x + i * nblocks], dst_tmp[i]);
1459
+ }
1460
+ *dst++ = make_block_mxfp4x32(dst_tmp, interleave_block);
1461
+ }
1462
+ src += nrows_interleaved * nblocks;
1463
+ }
1464
+ return 0;
1465
+ }
1466
+
1467
+ static spacemit_kernels::nrow_block_q5_1<32> make_block_q5_1x32(spacemit_kernels::nrow_block_q5_1<1> * in,
1468
+ unsigned int blck_size_interleave) {
1469
+ spacemit_kernels::nrow_block_q5_1<32> out;
1470
+ GGML_ASSERT(QK5_1 / blck_size_interleave == 1);
1471
+ GGML_UNUSED(blck_size_interleave);
1472
+
1473
+ for (int i = 0; i < 32; i++) {
1474
+ out.scales16[i] = in[i].scales16[0];
1475
+ out.zp[i] = in[i].zp[0];
1476
+ }
1477
+
1478
+ // qs: low 4 bits, reorder from [b0,b16],[b1,b17]... to [b0,b1]...[b14,b15] and [b16,b17]...[b30,b31]
1479
+ for (int i = 0; i < 32; i++) {
1480
+ // low half [0..15]
1481
+ for (int j = 0; j < QK5_1 / 4; j++) {
1482
+ out.qs[i * QK5_1 / 2 + j] = (in[i].qs[j * 2] & 0x0F) | ((in[i].qs[j * 2 + 1] & 0x0F) << 4);
1483
+ }
1484
+ // high half [16..31]
1485
+ for (int j = 0; j < QK5_1 / 4; j++) {
1486
+ out.qs[i * QK5_1 / 2 + QK5_1 / 4 + j] = ((in[i].qs[j * 2] & 0xF0) >> 4) | (in[i].qs[j * 2 + 1] & 0xF0);
1487
+ }
1488
+ }
1489
+
1490
+ // qh: 5th bit, copy directly
1491
+ for (int i = 0; i < 32; i++) {
1492
+ for (int j = 0; j < 4; j++) {
1493
+ out.qh[i * 4 + j] = in[i].qh[j];
1494
+ }
1495
+ }
1496
+
1497
+ return out;
1498
+ }
1499
+
1500
+ static spacemit_kernels::nrow_block_q5_0<32> make_block_q5_0x32(spacemit_kernels::nrow_block_q5_0<1> * in,
1501
+ unsigned int blck_size_interleave) {
1502
+ spacemit_kernels::nrow_block_q5_0<32> out;
1503
+ GGML_ASSERT(QK5_0 / blck_size_interleave == 1);
1504
+ GGML_UNUSED(blck_size_interleave);
1505
+
1506
+ for (int i = 0; i < 32; i++) {
1507
+ out.scales16[i] = in[i].scales16[0];
1508
+ }
1509
+
1510
+ // qs: low 4 bits, reorder from [b0,b16],[b1,b17]... to [b0,b1]...[b14,b15] and [b16,b17]...[b30,b31]
1511
+ for (int i = 0; i < 32; i++) {
1512
+ // low half [0..15]
1513
+ for (int j = 0; j < QK5_0 / 4; j++) {
1514
+ out.qs[i * QK5_0 / 2 + j] = (in[i].qs[j * 2] & 0x0F) | ((in[i].qs[j * 2 + 1] & 0x0F) << 4);
1515
+ }
1516
+ // high half [16..31]
1517
+ for (int j = 0; j < QK5_0 / 4; j++) {
1518
+ out.qs[i * QK5_0 / 2 + QK5_0 / 4 + j] = ((in[i].qs[j * 2] & 0xF0) >> 4) | (in[i].qs[j * 2 + 1] & 0xF0);
1519
+ }
1520
+ }
1521
+
1522
+ // qh: 5th bit, copy directly
1523
+ for (int i = 0; i < 32; i++) {
1524
+ for (int j = 0; j < 4; j++) {
1525
+ out.qh[i * 4 + j] = in[i].qh[j];
1526
+ }
1527
+ }
1528
+
1529
+ return out;
1530
+ }
1531
+
1532
+ static int repack_q5_0_to_q5_0_32_bl(ggml_tensor * t,
1533
+ int interleave_block,
1534
+ const void * GGML_RESTRICT data,
1535
+ size_t data_size) {
1536
+ GGML_ASSERT(t->type == GGML_TYPE_Q5_0);
1537
+ GGML_ASSERT(interleave_block == 32); // unused
1538
+
1539
+ constexpr int nrows_interleaved = 32;
1540
+
1541
+ spacemit_kernels::nrow_block_q5_0<32> * dst = (spacemit_kernels::nrow_block_q5_0<32> *) t->data;
1542
+ const block_q5_0 * src = (const block_q5_0 *) data;
1543
+ spacemit_kernels::nrow_block_q5_0<1> dst_tmp[32];
1544
+ int nrow = ggml_nrows(t);
1545
+ int nblocks = t->ne[0] / QK5_0;
1546
+
1547
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q5_0));
1548
+
1549
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK5_0 != 0) {
1550
+ return -1;
1551
+ }
1552
+
1553
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
1554
+ for (int64_t x = 0; x < nblocks; x++) {
1555
+ for (int i = 0; i < nrows_interleaved; i++) {
1556
+ const block_q5_0 & s = src[x + i * nblocks];
1557
+
1558
+ dst_tmp[i].scales16[0] = s.d;
1559
+ memcpy(dst_tmp[i].qs, s.qs, sizeof(dst_tmp[i].qs));
1560
+ memcpy(dst_tmp[i].qh, s.qh, sizeof(dst_tmp[i].qh));
1561
+ }
1562
+ *dst++ = make_block_q5_0x32(dst_tmp, interleave_block);
1563
+ }
1564
+ src += nrows_interleaved * nblocks;
1565
+ }
1566
+ return 0;
1567
+ }
1568
+
1569
+ static int repack_q5_1_to_q5_1_32_bl(ggml_tensor * t,
1570
+ int interleave_block,
1571
+ const void * GGML_RESTRICT data,
1572
+ size_t data_size) {
1573
+ GGML_ASSERT(t->type == GGML_TYPE_Q5_1);
1574
+ GGML_ASSERT(interleave_block == 32); // unused
1575
+
1576
+ constexpr int nrows_interleaved = 32;
1577
+
1578
+ spacemit_kernels::nrow_block_q5_1<32> * dst = (spacemit_kernels::nrow_block_q5_1<32> *) t->data;
1579
+ const block_q5_1 * src = (const block_q5_1 *) data;
1580
+ spacemit_kernels::nrow_block_q5_1<1> dst_tmp[32];
1581
+ int nrow = ggml_nrows(t);
1582
+ int nblocks = t->ne[0] / QK5_1;
1583
+
1584
+ GGML_ASSERT(data_size == nrow * nblocks * sizeof(block_q5_1));
1585
+
1586
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK5_1 != 0) {
1587
+ return -1;
1588
+ }
1589
+
1590
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
1591
+ for (int64_t x = 0; x < nblocks; x++) {
1592
+ for (int i = 0; i < nrows_interleaved; i++) {
1593
+ const block_q5_1 & s = src[x + i * nblocks];
1594
+
1595
+ float d = GGML_FP16_TO_FP32(s.GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d);
1596
+ float m = GGML_FP16_TO_FP32(s.GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.m);
1597
+
1598
+ if (d == 0.0f) {
1599
+ dst_tmp[i].scales16[0] = GGML_FP32_TO_FP16(std::fabs(m));
1600
+ dst_tmp[i].zp[0] = m < 0.0f ? 1 : 0;
1601
+ memset(dst_tmp[i].qh, 0, sizeof(dst_tmp[i].qh));
1602
+ memset(dst_tmp[i].qs, m > 0.0f ? 0x11 : 0x00, sizeof(dst_tmp[i].qs));
1603
+ continue;
1604
+ }
1605
+
1606
+ float mid = std::nearbyintf(-m / d);
1607
+ mid = std::min(31.0f, std::max(0.0f, mid));
1608
+
1609
+ dst_tmp[i].scales16[0] = GGML_FP32_TO_FP16(d);
1610
+ dst_tmp[i].zp[0] = static_cast<uint8_t>(mid);
1611
+
1612
+ // qs: copy low 4 bits directly (same nibble packing)
1613
+ memcpy(dst_tmp[i].qs, s.qs, QK5_1 / 2);
1614
+
1615
+ // qh: copy 5th bit directly
1616
+ memcpy(dst_tmp[i].qh, s.qh, 4);
1617
+ }
1618
+ *dst++ = make_block_q5_1x32(dst_tmp, interleave_block);
1619
+ }
1620
+ src += nrows_interleaved * nblocks;
1621
+ }
1622
+ return 0;
1623
+ }
1624
+
1625
+ static int repack_q5_k_to_q5_1_32_bl(ggml_tensor * t,
1626
+ int interleave_block,
1627
+ const void * GGML_RESTRICT data,
1628
+ size_t data_size) {
1629
+ GGML_ASSERT(t->type == GGML_TYPE_Q5_K);
1630
+ GGML_ASSERT(interleave_block == 32);
1631
+ GGML_ASSERT(QK_K / QK5_1 == 8);
1632
+
1633
+ constexpr int nrows_interleaved = 32;
1634
+
1635
+ spacemit_kernels::nrow_block_q5_1<32> * dst = (spacemit_kernels::nrow_block_q5_1<32> *) t->data;
1636
+ const block_q5_K * src = (const block_q5_K *) data;
1637
+ spacemit_kernels::nrow_block_q5_1<1> dst_tmp[32];
1638
+ int nrow = ggml_nrows(t);
1639
+ int nblocks = t->ne[0] / QK_K;
1640
+
1641
+ if (t->ne[1] % nrows_interleaved != 0 || t->ne[0] % QK_K != 0) {
1642
+ return -1;
1643
+ }
1644
+
1645
+ for (int b = 0; b < nrow; b += nrows_interleaved) {
1646
+ for (int64_t x = 0; x < nblocks; x++) {
1647
+ for (int j = 0; j < 8; j++) {
1648
+ for (int i = 0; i < nrows_interleaved; i++) {
1649
+ uint8_t sc, m;
1650
+ const float d = GGML_FP16_TO_FP32(src[x + i * nblocks].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.d);
1651
+ const float min =
1652
+ GGML_FP16_TO_FP32(src[x + i * nblocks].GGML_COMMON_AGGR_U.GGML_COMMON_AGGR_S.dmin);
1653
+ get_scale_min_k4(j, src[x + i * nblocks].scales, &sc, &m);
1654
+
1655
+ float d1 = d * sc;
1656
+ float m1 = min * m;
1657
+
1658
+ float mid = std::nearbyintf(m1 / d1);
1659
+ mid = std::min(31.0f, std::max(0.0f, mid));
1660
+ dst_tmp[i].scales16[0] = GGML_FP32_TO_FP16(d1);
1661
+ dst_tmp[i].zp[0] = static_cast<uint8_t>(mid);
1662
+
1663
+ // src -> [b0, b32] [b1, b33] ... [b31, b63]
1664
+ // dst -> [b0, b16] [b1, b17] ... [b15, b31] [b32, b48] [b33, b49] ... [b47, b63]
1665
+ const uint8_t * q = src[x + i * nblocks].qs + (j / 2) * QK5_1;
1666
+ if (j % 2 == 0) {
1667
+ for (int ii = 0; ii < 16; ii++) {
1668
+ dst_tmp[i].qs[ii] = (q[ii] & 0x0F) | ((q[ii + 16] & 0x0F) << 4);
1669
+ }
1670
+ } else {
1671
+ for (int ii = 0; ii < 16; ii++) {
1672
+ dst_tmp[i].qs[ii] = ((q[ii] & 0xF0) >> 4) | (q[ii + 16] & 0xF0);
1673
+ }
1674
+ }
1675
+
1676
+ // Extract the 5th bit (qh) for this sub-block
1677
+ // block_q5_K.qh[32]: for sub-block j, the 5th bit is at bit position j in qh[l]
1678
+ // qs was reordered: dst_qs maps to src weights [0,16,1,17,...,15,31]
1679
+ // So qh must follow the same reorder to stay aligned with qs
1680
+ // dst qh[4] = 32 bits for 32 weights in the reordered layout:
1681
+ // byte 0: weights 0..7 (from src_qh[0..7])
1682
+ // byte 1: weights 8..15 (from src_qh[8..15])
1683
+ // byte 2: weights 16..23 (from src_qh[16..23])
1684
+ // byte 3: weights 24..31 (from src_qh[24..31])
1685
+ const uint8_t * src_qh = src[x + i * nblocks].qh;
1686
+ for (int bi = 0; bi < 4; bi++) {
1687
+ uint8_t qh_byte = 0;
1688
+ for (int k = 0; k < 8; k++) {
1689
+ int src_idx = bi * 8 + k;
1690
+ qh_byte |= ((src_qh[src_idx] >> j) & 1) << k;
1691
+ }
1692
+ dst_tmp[i].qh[bi] = qh_byte;
1693
+ }
1694
+ }
1695
+ *dst++ = make_block_q5_1x32(dst_tmp, interleave_block);
1696
+ }
1697
+ }
1698
+ src += nrows_interleaved * nblocks;
1699
+ }
1700
+ return 0;
1701
+ }
1702
+
1703
+ namespace ggml::cpu::riscv64_spacemit {
1704
+
1705
+ template <typename BLOC_TYPE, int64_t INTER_SIZE, int64_t NB_COLS> int repack(ggml_tensor *, const void *, size_t);
1706
+
1707
+ template <> int repack<block_q4_0, 32, 16>(ggml_tensor * t, const void * data, size_t data_size) {
1708
+ return repack_q4_0_to_q4_0_16_bl(t, 16, data, data_size);
1709
+ }
1710
+
1711
+ template <> int repack<block_q4_1, 32, 16>(ggml_tensor * t, const void * data, size_t data_size) {
1712
+ return repack_q4_1_to_q4_1_16_bl(t, 16, data, data_size);
1713
+ }
1714
+
1715
+ template <> int repack<block_q4_K, 32, 16>(ggml_tensor * t, const void * data, size_t data_size) {
1716
+ return repack_q4_k_to_q4_1_16_bl(t, 16, data, data_size);
1717
+ }
1718
+
1719
+ template <> int repack<block_q2_K, 256, 32>(ggml_tensor * t, const void * data, size_t data_size) {
1720
+ return repack_q2_k_to_q2_k_32_bl(t, 32, data, data_size);
1721
+ }
1722
+
1723
+ template <> int repack<block_q3_K, 256, 32>(ggml_tensor * t, const void * data, size_t data_size) {
1724
+ return repack_q3_k_to_q3_k_32_bl(t, 32, data, data_size);
1725
+ }
1726
+
1727
+ template <> int repack<block_q4_0, 32, 32>(ggml_tensor * t, const void * data, size_t data_size) {
1728
+ #if 0
1729
+ return repack_q4_0_to_q4_0_32_bl_ref(t, 32, data, data_size);
1730
+ #else
1731
+ return repack_q4_0_to_q4_0_32_bl(t, 32, data, data_size);
1732
+ #endif
1733
+ }
1734
+
1735
+ template <> int repack<block_q4_0, 256, 32>(ggml_tensor * t, const void * data, size_t data_size) {
1736
+ #if 1
1737
+ return repack_q4_0_to_q4_0_256_32_bl_ref(t, 32, data, data_size);
1738
+ #else
1739
+ //return repack_q4_0_to_q4_0_256_32_bl(t, 32, data, data_size);
1740
+ #endif
1741
+ }
1742
+
1743
+ template <> int repack<block_q4_1, 32, 32>(ggml_tensor * t, const void * data, size_t data_size) {
1744
+ #if 0
1745
+ return repack_q4_1_to_q4_1_32_bl_ref(t, 32, data, data_size);
1746
+ #else
1747
+ return repack_q4_1_to_q4_1_32_bl(t, 32, data, data_size);
1748
+ #endif
1749
+ }
1750
+
1751
+ template <> int repack<block_q4_1, 256, 32>(ggml_tensor * t, const void * data, size_t data_size) {
1752
+ #if 1
1753
+ return repack_q4_0_to_q4_1_256_32_bl_ref(t, 32, data, data_size);
1754
+ #else
1755
+ return repack_q4_1_to_q4_1_256_32_bl(t, 32, data, data_size);
1756
+ #endif
1757
+ }
1758
+
1759
+ template <> int repack<block_q4_K, 32, 32>(ggml_tensor * t, const void * data, size_t data_size) {
1760
+ return repack_q4_k_to_q4_1_32_bl(t, 32, data, data_size);
1761
+ }
1762
+
1763
+ template <> int repack<block_q6_K, 32, 32>(ggml_tensor * t, const void * data, size_t data_size) {
1764
+ #if 1
1765
+ return repack_q6_k_to_q8_0_32_bl_ref(t, 32, data, data_size);
1766
+ #else
1767
+ return repack_q6_k_to_q8_0_32_bl(t, 32, data, data_size);
1768
+ #endif
1769
+ }
1770
+
1771
+ template <> int repack<block_q8_0, 32, 32>(ggml_tensor * t, const void * data, size_t data_size) {
1772
+ #if 1
1773
+ return repack_q8_0_to_q8_0_32_bl_ref(t, 32, data, data_size);
1774
+ #else
1775
+ return repack_q8_0_to_q8_0_32_bl(t, 32, data, data_size);
1776
+ #endif
1777
+ }
1778
+
1779
+ template <> int repack<block_mxfp4, 32, 32>(ggml_tensor * t, const void * data, size_t data_size) {
1780
+ return repack_mxfp4_to_mxfp4_32_bl(t, 32, data, data_size);
1781
+ }
1782
+
1783
+ template <> int repack<block_q5_0, 32, 32>(ggml_tensor * t, const void * data, size_t data_size) {
1784
+ return repack_q5_0_to_q5_0_32_bl(t, 32, data, data_size);
1785
+ }
1786
+
1787
+ template <> int repack<block_q5_1, 32, 32>(ggml_tensor * t, const void * data, size_t data_size) {
1788
+ return repack_q5_1_to_q5_1_32_bl(t, 32, data, data_size);
1789
+ }
1790
+
1791
+ template <> int repack<block_q5_K, 32, 32>(ggml_tensor * t, const void * data, size_t data_size) {
1792
+ return repack_q5_k_to_q5_1_32_bl(t, 32, data, data_size);
1793
+ }
1794
+
1795
+ } // namespace ggml::cpu::riscv64_spacemit