npm - @fugood/llama.node - Versions diffs - 0.6.2 → 1.0.0-beta.1 - Mend

@fugood/llama.node 0.6.2 → 1.0.0-beta.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (378) hide show

package/CMakeLists.txt +40 -30
package/README.md +4 -1
package/lib/binding.js +41 -29
package/lib/binding.ts +26 -25
package/package.json +45 -10
package/scripts/build.js +47 -0
package/scripts/llama.cpp.patch +109 -0
package/src/anyascii.c +22223 -0
package/src/anyascii.h +42 -0
package/src/tts_utils.cpp +20 -7
package/src/tts_utils.h +2 -0
package/bin/darwin/arm64/llama-node.node +0 -0
package/bin/darwin/x64/llama-node.node +0 -0
package/bin/linux/arm64/llama-node.node +0 -0
package/bin/linux/x64/llama-node.node +0 -0
package/bin/linux-cuda/arm64/llama-node.node +0 -0
package/bin/linux-cuda/x64/llama-node.node +0 -0
package/bin/linux-vulkan/arm64/llama-node.node +0 -0
package/bin/linux-vulkan/x64/llama-node.node +0 -0
package/bin/win32/x64/llama-node.node +0 -0
package/bin/win32/x64/node.lib +0 -0
package/bin/win32-vulkan/arm64/llama-node.node +0 -0
package/bin/win32-vulkan/arm64/node.lib +0 -0
package/bin/win32-vulkan/x64/llama-node.node +0 -0
package/bin/win32-vulkan/x64/node.lib +0 -0
package/patches/node-api-headers+1.1.0.patch +0 -26
package/src/llama.cpp/.github/workflows/build-linux-cross.yml +0 -233
package/src/llama.cpp/.github/workflows/build.yml +0 -1078
package/src/llama.cpp/.github/workflows/close-issue.yml +0 -28
package/src/llama.cpp/.github/workflows/docker.yml +0 -178
package/src/llama.cpp/.github/workflows/editorconfig.yml +0 -29
package/src/llama.cpp/.github/workflows/gguf-publish.yml +0 -44
package/src/llama.cpp/.github/workflows/labeler.yml +0 -17
package/src/llama.cpp/.github/workflows/python-check-requirements.yml +0 -33
package/src/llama.cpp/.github/workflows/python-lint.yml +0 -30
package/src/llama.cpp/.github/workflows/python-type-check.yml +0 -40
package/src/llama.cpp/.github/workflows/release.yml +0 -739
package/src/llama.cpp/.github/workflows/server.yml +0 -237
package/src/llama.cpp/.github/workflows/winget.yml +0 -42
package/src/llama.cpp/cmake/arm64-apple-clang.cmake +0 -16
package/src/llama.cpp/cmake/arm64-windows-llvm.cmake +0 -16
package/src/llama.cpp/cmake/build-info.cmake +0 -64
package/src/llama.cpp/cmake/common.cmake +0 -35
package/src/llama.cpp/cmake/git-vars.cmake +0 -22
package/src/llama.cpp/cmake/x64-windows-llvm.cmake +0 -5
package/src/llama.cpp/common/build-info.cpp.in +0 -4
package/src/llama.cpp/docs/build.md +0 -561
package/src/llama.cpp/examples/CMakeLists.txt +0 -43
package/src/llama.cpp/examples/batched/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/batched/batched.cpp +0 -246
package/src/llama.cpp/examples/chat-13B.bat +0 -57
package/src/llama.cpp/examples/convert-llama2c-to-ggml/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/convert-llama2c-to-ggml/convert-llama2c-to-ggml.cpp +0 -941
package/src/llama.cpp/examples/deprecation-warning/deprecation-warning.cpp +0 -35
package/src/llama.cpp/examples/embedding/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/embedding/embedding.cpp +0 -323
package/src/llama.cpp/examples/eval-callback/CMakeLists.txt +0 -10
package/src/llama.cpp/examples/eval-callback/eval-callback.cpp +0 -194
package/src/llama.cpp/examples/gen-docs/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/gen-docs/gen-docs.cpp +0 -83
package/src/llama.cpp/examples/gguf/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/gguf/gguf.cpp +0 -265
package/src/llama.cpp/examples/gguf-hash/CMakeLists.txt +0 -22
package/src/llama.cpp/examples/gguf-hash/deps/rotate-bits/rotate-bits.h +0 -46
package/src/llama.cpp/examples/gguf-hash/deps/sha1/sha1.c +0 -295
package/src/llama.cpp/examples/gguf-hash/deps/sha1/sha1.h +0 -52
package/src/llama.cpp/examples/gguf-hash/deps/sha256/sha256.c +0 -221
package/src/llama.cpp/examples/gguf-hash/deps/sha256/sha256.h +0 -24
package/src/llama.cpp/examples/gguf-hash/deps/xxhash/xxhash.c +0 -42
package/src/llama.cpp/examples/gguf-hash/deps/xxhash/xxhash.h +0 -7093
package/src/llama.cpp/examples/gguf-hash/gguf-hash.cpp +0 -694
package/src/llama.cpp/examples/gritlm/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/gritlm/gritlm.cpp +0 -229
package/src/llama.cpp/examples/jeopardy/questions.txt +0 -100
package/src/llama.cpp/examples/llama.android/app/build.gradle.kts +0 -65
package/src/llama.cpp/examples/llama.android/build.gradle.kts +0 -6
package/src/llama.cpp/examples/llama.android/llama/build.gradle.kts +0 -71
package/src/llama.cpp/examples/llama.android/llama/src/main/cpp/CMakeLists.txt +0 -53
package/src/llama.cpp/examples/llama.android/llama/src/main/cpp/llama-android.cpp +0 -452
package/src/llama.cpp/examples/llama.android/settings.gradle.kts +0 -18
package/src/llama.cpp/examples/lookahead/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/lookahead/lookahead.cpp +0 -472
package/src/llama.cpp/examples/lookup/CMakeLists.txt +0 -23
package/src/llama.cpp/examples/lookup/lookup-create.cpp +0 -40
package/src/llama.cpp/examples/lookup/lookup-merge.cpp +0 -47
package/src/llama.cpp/examples/lookup/lookup-stats.cpp +0 -157
package/src/llama.cpp/examples/lookup/lookup.cpp +0 -242
package/src/llama.cpp/examples/parallel/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/parallel/parallel.cpp +0 -492
package/src/llama.cpp/examples/passkey/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/passkey/passkey.cpp +0 -277
package/src/llama.cpp/examples/retrieval/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/retrieval/retrieval.cpp +0 -304
package/src/llama.cpp/examples/save-load-state/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/save-load-state/save-load-state.cpp +0 -246
package/src/llama.cpp/examples/simple/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/simple/simple.cpp +0 -206
package/src/llama.cpp/examples/simple-chat/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/simple-chat/simple-chat.cpp +0 -206
package/src/llama.cpp/examples/simple-cmake-pkg/CMakeLists.txt +0 -11
package/src/llama.cpp/examples/speculative/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/speculative/speculative.cpp +0 -644
package/src/llama.cpp/examples/speculative-simple/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/speculative-simple/speculative-simple.cpp +0 -261
package/src/llama.cpp/examples/sycl/CMakeLists.txt +0 -9
package/src/llama.cpp/examples/sycl/build.sh +0 -23
package/src/llama.cpp/examples/sycl/ls-sycl-device.cpp +0 -13
package/src/llama.cpp/examples/sycl/run-llama2.sh +0 -27
package/src/llama.cpp/examples/sycl/run-llama3.sh +0 -28
package/src/llama.cpp/examples/sycl/win-build-sycl.bat +0 -33
package/src/llama.cpp/examples/sycl/win-run-llama2.bat +0 -9
package/src/llama.cpp/examples/sycl/win-run-llama3.bat +0 -9
package/src/llama.cpp/examples/training/CMakeLists.txt +0 -5
package/src/llama.cpp/examples/training/finetune.cpp +0 -96
package/src/llama.cpp/ggml/cmake/GitVars.cmake +0 -22
package/src/llama.cpp/ggml/cmake/common.cmake +0 -26
package/src/llama.cpp/ggml/src/ggml-alloc.c +0 -1042
package/src/llama.cpp/ggml/src/ggml-backend-impl.h +0 -255
package/src/llama.cpp/ggml/src/ggml-backend-reg.cpp +0 -586
package/src/llama.cpp/ggml/src/ggml-backend.cpp +0 -2008
package/src/llama.cpp/ggml/src/ggml-blas/CMakeLists.txt +0 -87
package/src/llama.cpp/ggml/src/ggml-blas/ggml-blas.cpp +0 -517
package/src/llama.cpp/ggml/src/ggml-cann/CMakeLists.txt +0 -74
package/src/llama.cpp/ggml/src/ggml-cann/acl_tensor.cpp +0 -179
package/src/llama.cpp/ggml/src/ggml-cann/acl_tensor.h +0 -258
package/src/llama.cpp/ggml/src/ggml-cann/aclnn_ops.cpp +0 -2863
package/src/llama.cpp/ggml/src/ggml-cann/aclnn_ops.h +0 -1110
package/src/llama.cpp/ggml/src/ggml-cann/common.h +0 -420
package/src/llama.cpp/ggml/src/ggml-cann/ggml-cann.cpp +0 -2570
package/src/llama.cpp/ggml/src/ggml-common.h +0 -1857
package/src/llama.cpp/ggml/src/ggml-cpu/cmake/FindSIMD.cmake +0 -100
package/src/llama.cpp/ggml/src/ggml-cuda/CMakeLists.txt +0 -184
package/src/llama.cpp/ggml/src/ggml-cuda/vendors/cuda.h +0 -15
package/src/llama.cpp/ggml/src/ggml-cuda/vendors/hip.h +0 -243
package/src/llama.cpp/ggml/src/ggml-cuda/vendors/musa.h +0 -140
package/src/llama.cpp/ggml/src/ggml-hip/CMakeLists.txt +0 -131
package/src/llama.cpp/ggml/src/ggml-impl.h +0 -601
package/src/llama.cpp/ggml/src/ggml-kompute/CMakeLists.txt +0 -166
package/src/llama.cpp/ggml/src/ggml-kompute/ggml-kompute.cpp +0 -2251
package/src/llama.cpp/ggml/src/ggml-metal/CMakeLists.txt +0 -120
package/src/llama.cpp/ggml/src/ggml-metal/ggml-metal-impl.h +0 -622
package/src/llama.cpp/ggml/src/ggml-musa/CMakeLists.txt +0 -113
package/src/llama.cpp/ggml/src/ggml-opencl/CMakeLists.txt +0 -96
package/src/llama.cpp/ggml/src/ggml-opencl/ggml-opencl.cpp +0 -5124
package/src/llama.cpp/ggml/src/ggml-opt.cpp +0 -1037
package/src/llama.cpp/ggml/src/ggml-quants.c +0 -5232
package/src/llama.cpp/ggml/src/ggml-quants.h +0 -100
package/src/llama.cpp/ggml/src/ggml-rpc/CMakeLists.txt +0 -9
package/src/llama.cpp/ggml/src/ggml-rpc/ggml-rpc.cpp +0 -1813
package/src/llama.cpp/ggml/src/ggml-sycl/CMakeLists.txt +0 -189
package/src/llama.cpp/ggml/src/ggml-sycl/backend.hpp +0 -37
package/src/llama.cpp/ggml/src/ggml-sycl/binbcast.cpp +0 -239
package/src/llama.cpp/ggml/src/ggml-sycl/binbcast.hpp +0 -39
package/src/llama.cpp/ggml/src/ggml-sycl/common.cpp +0 -83
package/src/llama.cpp/ggml/src/ggml-sycl/common.hpp +0 -493
package/src/llama.cpp/ggml/src/ggml-sycl/concat.cpp +0 -197
package/src/llama.cpp/ggml/src/ggml-sycl/concat.hpp +0 -20
package/src/llama.cpp/ggml/src/ggml-sycl/conv.cpp +0 -100
package/src/llama.cpp/ggml/src/ggml-sycl/conv.hpp +0 -20
package/src/llama.cpp/ggml/src/ggml-sycl/convert.cpp +0 -623
package/src/llama.cpp/ggml/src/ggml-sycl/convert.hpp +0 -34
package/src/llama.cpp/ggml/src/ggml-sycl/cpy.cpp +0 -701
package/src/llama.cpp/ggml/src/ggml-sycl/cpy.hpp +0 -11
package/src/llama.cpp/ggml/src/ggml-sycl/dequantize.hpp +0 -791
package/src/llama.cpp/ggml/src/ggml-sycl/dmmv.cpp +0 -1160
package/src/llama.cpp/ggml/src/ggml-sycl/dmmv.hpp +0 -27
package/src/llama.cpp/ggml/src/ggml-sycl/dpct/helper.hpp +0 -2957
package/src/llama.cpp/ggml/src/ggml-sycl/element_wise.cpp +0 -1536
package/src/llama.cpp/ggml/src/ggml-sycl/element_wise.hpp +0 -75
package/src/llama.cpp/ggml/src/ggml-sycl/gemm.hpp +0 -99
package/src/llama.cpp/ggml/src/ggml-sycl/getrows.cpp +0 -311
package/src/llama.cpp/ggml/src/ggml-sycl/getrows.hpp +0 -20
package/src/llama.cpp/ggml/src/ggml-sycl/ggml-sycl.cpp +0 -4443
package/src/llama.cpp/ggml/src/ggml-sycl/gla.cpp +0 -105
package/src/llama.cpp/ggml/src/ggml-sycl/gla.hpp +0 -8
package/src/llama.cpp/ggml/src/ggml-sycl/im2col.cpp +0 -136
package/src/llama.cpp/ggml/src/ggml-sycl/im2col.hpp +0 -21
package/src/llama.cpp/ggml/src/ggml-sycl/mmq.cpp +0 -3030
package/src/llama.cpp/ggml/src/ggml-sycl/mmq.hpp +0 -33
package/src/llama.cpp/ggml/src/ggml-sycl/mmvq.cpp +0 -1108
package/src/llama.cpp/ggml/src/ggml-sycl/mmvq.hpp +0 -27
package/src/llama.cpp/ggml/src/ggml-sycl/norm.cpp +0 -474
package/src/llama.cpp/ggml/src/ggml-sycl/norm.hpp +0 -26
package/src/llama.cpp/ggml/src/ggml-sycl/outprod.cpp +0 -46
package/src/llama.cpp/ggml/src/ggml-sycl/outprod.hpp +0 -10
package/src/llama.cpp/ggml/src/ggml-sycl/presets.hpp +0 -74
package/src/llama.cpp/ggml/src/ggml-sycl/quants.hpp +0 -83
package/src/llama.cpp/ggml/src/ggml-sycl/rope.cpp +0 -362
package/src/llama.cpp/ggml/src/ggml-sycl/rope.hpp +0 -20
package/src/llama.cpp/ggml/src/ggml-sycl/softmax.cpp +0 -264
package/src/llama.cpp/ggml/src/ggml-sycl/softmax.hpp +0 -20
package/src/llama.cpp/ggml/src/ggml-sycl/sycl_hw.cpp +0 -13
package/src/llama.cpp/ggml/src/ggml-sycl/sycl_hw.hpp +0 -23
package/src/llama.cpp/ggml/src/ggml-sycl/tsembd.cpp +0 -73
package/src/llama.cpp/ggml/src/ggml-sycl/tsembd.hpp +0 -20
package/src/llama.cpp/ggml/src/ggml-sycl/vecdotq.hpp +0 -1215
package/src/llama.cpp/ggml/src/ggml-sycl/wkv.cpp +0 -305
package/src/llama.cpp/ggml/src/ggml-sycl/wkv.hpp +0 -10
package/src/llama.cpp/ggml/src/ggml-threading.cpp +0 -12
package/src/llama.cpp/ggml/src/ggml-threading.h +0 -14
package/src/llama.cpp/ggml/src/ggml-vulkan/CMakeLists.txt +0 -196
package/src/llama.cpp/ggml/src/ggml-vulkan/ggml-vulkan.cpp +0 -10699
package/src/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/CMakeLists.txt +0 -39
package/src/llama.cpp/ggml/src/ggml-vulkan/vulkan-shaders/vulkan-shaders-gen.cpp +0 -751
package/src/llama.cpp/ggml/src/ggml.c +0 -6550
package/src/llama.cpp/ggml/src/gguf.cpp +0 -1330
package/src/llama.cpp/models/.editorconfig +0 -1
package/src/llama.cpp/models/ggml-vocab-aquila.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-baichuan.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-bert-bge.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-bert-bge.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-bert-bge.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-chameleon.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-chameleon.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-command-r.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-command-r.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-command-r.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-deepseek-coder.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-deepseek-coder.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-deepseek-coder.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-deepseek-llm.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-deepseek-llm.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-deepseek-llm.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-deepseek-r1-qwen.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-deepseek-r1-qwen.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-falcon.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-falcon.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-falcon.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-gpt-2.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-gpt-2.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-gpt-2.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-gpt-4o.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-gpt-4o.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-gpt-neox.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-llama-bpe.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-llama-bpe.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-llama-bpe.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-llama-spm.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-llama-spm.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-llama-spm.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-llama4.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-llama4.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-mpt.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-mpt.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-mpt.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-phi-3.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-phi-3.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-phi-3.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-pixtral.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-pixtral.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-qwen2.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-qwen2.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-qwen2.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-refact.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-refact.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-refact.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-roberta-bpe.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-roberta-bpe.gguf.out +0 -46
package/src/llama.cpp/models/ggml-vocab-starcoder.gguf +0 -0
package/src/llama.cpp/models/ggml-vocab-starcoder.gguf.inp +0 -112
package/src/llama.cpp/models/ggml-vocab-starcoder.gguf.out +0 -46
package/src/llama.cpp/pocs/CMakeLists.txt +0 -14
package/src/llama.cpp/pocs/vdot/CMakeLists.txt +0 -9
package/src/llama.cpp/pocs/vdot/q8dot.cpp +0 -173
package/src/llama.cpp/pocs/vdot/vdot.cpp +0 -311
package/src/llama.cpp/prompts/LLM-questions.txt +0 -49
package/src/llama.cpp/prompts/alpaca.txt +0 -1
package/src/llama.cpp/prompts/assistant.txt +0 -31
package/src/llama.cpp/prompts/chat-with-baichuan.txt +0 -4
package/src/llama.cpp/prompts/chat-with-bob.txt +0 -7
package/src/llama.cpp/prompts/chat-with-qwen.txt +0 -1
package/src/llama.cpp/prompts/chat-with-vicuna-v0.txt +0 -7
package/src/llama.cpp/prompts/chat-with-vicuna-v1.txt +0 -7
package/src/llama.cpp/prompts/chat.txt +0 -28
package/src/llama.cpp/prompts/dan-modified.txt +0 -1
package/src/llama.cpp/prompts/dan.txt +0 -1
package/src/llama.cpp/prompts/mnemonics.txt +0 -93
package/src/llama.cpp/prompts/parallel-questions.txt +0 -43
package/src/llama.cpp/prompts/reason-act.txt +0 -18
package/src/llama.cpp/requirements/requirements-all.txt +0 -15
package/src/llama.cpp/requirements/requirements-compare-llama-bench.txt +0 -2
package/src/llama.cpp/requirements/requirements-convert_hf_to_gguf.txt +0 -7
package/src/llama.cpp/requirements/requirements-convert_hf_to_gguf_update.txt +0 -7
package/src/llama.cpp/requirements/requirements-convert_legacy_llama.txt +0 -5
package/src/llama.cpp/requirements/requirements-convert_llama_ggml_to_gguf.txt +0 -1
package/src/llama.cpp/requirements/requirements-convert_lora_to_gguf.txt +0 -4
package/src/llama.cpp/requirements/requirements-gguf_editor_gui.txt +0 -3
package/src/llama.cpp/requirements/requirements-pydantic.txt +0 -3
package/src/llama.cpp/requirements/requirements-test-tokenizer-random.txt +0 -1
package/src/llama.cpp/requirements/requirements-tool_bench.txt +0 -12
package/src/llama.cpp/requirements.txt +0 -13
package/src/llama.cpp/scripts/build-info.sh +0 -30
package/src/llama.cpp/scripts/install-oneapi.bat +0 -19
package/src/llama.cpp/scripts/xxd.cmake +0 -16
package/src/llama.cpp/tests/CMakeLists.txt +0 -177
package/src/llama.cpp/tests/get-model.cpp +0 -21
package/src/llama.cpp/tests/get-model.h +0 -2
package/src/llama.cpp/tests/test-arg-parser.cpp +0 -178
package/src/llama.cpp/tests/test-autorelease.cpp +0 -24
package/src/llama.cpp/tests/test-backend-ops.cpp +0 -4793
package/src/llama.cpp/tests/test-barrier.cpp +0 -94
package/src/llama.cpp/tests/test-c.c +0 -7
package/src/llama.cpp/tests/test-chat-template.cpp +0 -417
package/src/llama.cpp/tests/test-chat.cpp +0 -985
package/src/llama.cpp/tests/test-double-float.cpp +0 -57
package/src/llama.cpp/tests/test-gbnf-validator.cpp +0 -109
package/src/llama.cpp/tests/test-gguf.cpp +0 -1338
package/src/llama.cpp/tests/test-grammar-integration.cpp +0 -1308
package/src/llama.cpp/tests/test-grammar-llguidance.cpp +0 -1201
package/src/llama.cpp/tests/test-grammar-parser.cpp +0 -519
package/src/llama.cpp/tests/test-json-schema-to-grammar.cpp +0 -1304
package/src/llama.cpp/tests/test-llama-grammar.cpp +0 -408
package/src/llama.cpp/tests/test-log.cpp +0 -39
package/src/llama.cpp/tests/test-model-load-cancel.cpp +0 -27
package/src/llama.cpp/tests/test-mtmd-c-api.c +0 -63
package/src/llama.cpp/tests/test-opt.cpp +0 -904
package/src/llama.cpp/tests/test-quantize-fns.cpp +0 -186
package/src/llama.cpp/tests/test-quantize-perf.cpp +0 -365
package/src/llama.cpp/tests/test-quantize-stats.cpp +0 -424
package/src/llama.cpp/tests/test-regex-partial.cpp +0 -288
package/src/llama.cpp/tests/test-rope.cpp +0 -262
package/src/llama.cpp/tests/test-sampling.cpp +0 -399
package/src/llama.cpp/tests/test-tokenizer-0.cpp +0 -312
package/src/llama.cpp/tests/test-tokenizer-1-bpe.cpp +0 -155
package/src/llama.cpp/tests/test-tokenizer-1-spm.cpp +0 -125
package/src/llama.cpp/tools/CMakeLists.txt +0 -39
package/src/llama.cpp/tools/batched-bench/CMakeLists.txt +0 -5
package/src/llama.cpp/tools/batched-bench/batched-bench.cpp +0 -204
package/src/llama.cpp/tools/cvector-generator/CMakeLists.txt +0 -5
package/src/llama.cpp/tools/cvector-generator/completions.txt +0 -582
package/src/llama.cpp/tools/cvector-generator/cvector-generator.cpp +0 -508
package/src/llama.cpp/tools/cvector-generator/mean.hpp +0 -48
package/src/llama.cpp/tools/cvector-generator/negative.txt +0 -4
package/src/llama.cpp/tools/cvector-generator/pca.hpp +0 -315
package/src/llama.cpp/tools/cvector-generator/positive.txt +0 -4
package/src/llama.cpp/tools/export-lora/CMakeLists.txt +0 -5
package/src/llama.cpp/tools/export-lora/export-lora.cpp +0 -434
package/src/llama.cpp/tools/gguf-split/CMakeLists.txt +0 -5
package/src/llama.cpp/tools/gguf-split/gguf-split.cpp +0 -583
package/src/llama.cpp/tools/imatrix/CMakeLists.txt +0 -5
package/src/llama.cpp/tools/imatrix/imatrix.cpp +0 -667
package/src/llama.cpp/tools/llama-bench/CMakeLists.txt +0 -5
package/src/llama.cpp/tools/llama-bench/llama-bench.cpp +0 -2024
package/src/llama.cpp/tools/main/CMakeLists.txt +0 -5
package/src/llama.cpp/tools/main/main.cpp +0 -977
package/src/llama.cpp/tools/mtmd/CMakeLists.txt +0 -58
package/src/llama.cpp/tools/mtmd/clip-impl.h +0 -462
package/src/llama.cpp/tools/mtmd/clip.cpp +0 -4024
package/src/llama.cpp/tools/mtmd/clip.h +0 -101
package/src/llama.cpp/tools/mtmd/deprecation-warning.cpp +0 -22
package/src/llama.cpp/tools/mtmd/miniaudio.h +0 -93468
package/src/llama.cpp/tools/mtmd/mtmd-audio.cpp +0 -855
package/src/llama.cpp/tools/mtmd/mtmd-audio.h +0 -62
package/src/llama.cpp/tools/mtmd/mtmd-cli.cpp +0 -377
package/src/llama.cpp/tools/mtmd/mtmd-helper.cpp +0 -297
package/src/llama.cpp/tools/mtmd/mtmd.cpp +0 -942
package/src/llama.cpp/tools/mtmd/mtmd.h +0 -362
package/src/llama.cpp/tools/mtmd/requirements.txt +0 -5
package/src/llama.cpp/tools/perplexity/CMakeLists.txt +0 -5
package/src/llama.cpp/tools/perplexity/perplexity.cpp +0 -2063
package/src/llama.cpp/tools/quantize/CMakeLists.txt +0 -6
package/src/llama.cpp/tools/quantize/quantize.cpp +0 -519
package/src/llama.cpp/tools/rpc/CMakeLists.txt +0 -4
package/src/llama.cpp/tools/rpc/rpc-server.cpp +0 -322
package/src/llama.cpp/tools/run/CMakeLists.txt +0 -16
package/src/llama.cpp/tools/run/linenoise.cpp/linenoise.cpp +0 -1995
package/src/llama.cpp/tools/run/linenoise.cpp/linenoise.h +0 -137
package/src/llama.cpp/tools/run/run.cpp +0 -1261
package/src/llama.cpp/tools/server/CMakeLists.txt +0 -51
package/src/llama.cpp/tools/server/bench/requirements.txt +0 -2
package/src/llama.cpp/tools/server/httplib.h +0 -10506
package/src/llama.cpp/tools/server/server.cpp +0 -4966
package/src/llama.cpp/tools/server/tests/requirements.txt +0 -8
package/src/llama.cpp/tools/server/utils.hpp +0 -1337
package/src/llama.cpp/tools/tokenize/CMakeLists.txt +0 -5
package/src/llama.cpp/tools/tokenize/tokenize.cpp +0 -416
package/src/llama.cpp/tools/tts/CMakeLists.txt +0 -5
package/src/llama.cpp/tools/tts/tts.cpp +0 -1092

package/src/llama.cpp/tools/main/main.cpp DELETED Viewed

@@ -1,977 +0,0 @@
-#include "arg.h"
-#include "common.h"
-#include "console.h"
-#include "log.h"
-#include "sampling.h"
-#include "llama.h"
-#include "chat.h"
-#include <cstdio>
-#include <cstring>
-#include <ctime>
-#include <fstream>
-#include <iostream>
-#include <sstream>
-#include <string>
-#include <vector>
-#if defined (__unix__) || (defined (__APPLE__) && defined (__MACH__))
-#include <signal.h>
-#include <unistd.h>
-#elif defined (_WIN32)
-#define WIN32_LEAN_AND_MEAN
-#ifndef NOMINMAX
-#define NOMINMAX
-#endif
-#include <windows.h>
-#include <signal.h>
-#endif
-#if defined(_MSC_VER)
-#pragma warning(disable: 4244 4267) // possible loss of data
-#endif
-static llama_context           ** g_ctx;
-static llama_model             ** g_model;
-static common_sampler          ** g_smpl;
-static common_params            * g_params;
-static std::vector<llama_token> * g_input_tokens;
-static std::ostringstream       * g_output_ss;
-static std::vector<llama_token> * g_output_tokens;
-static bool is_interacting  = false;
-static bool need_insert_eot = false;
-static void print_usage(int argc, char ** argv) {
-    (void) argc;
-    LOG("\nexample usage:\n");
-    LOG("\n  text generation:     %s -m your_model.gguf -p \"I believe the meaning of life is\" -n 128 -no-cnv\n", argv[0]);
-    LOG("\n  chat (conversation): %s -m your_model.gguf -sys \"You are a helpful assistant\"\n", argv[0]);
-    LOG("\n");
-}
-static bool file_exists(const std::string & path) {
-    std::ifstream f(path.c_str());
-    return f.good();
-}
-static bool file_is_empty(const std::string & path) {
-    std::ifstream f;
-    f.exceptions(std::ifstream::failbit | std::ifstream::badbit);
-    f.open(path.c_str(), std::ios::in | std::ios::binary | std::ios::ate);
-    return f.tellg() == 0;
-}
-#if defined (__unix__) || (defined (__APPLE__) && defined (__MACH__)) || defined (_WIN32)
-static void sigint_handler(int signo) {
-    if (signo == SIGINT) {
-        if (!is_interacting && g_params->interactive) {
-            is_interacting  = true;
-            need_insert_eot = true;
-        } else {
-            console::cleanup();
-            LOG("\n");
-            common_perf_print(*g_ctx, *g_smpl);
-            // make sure all logs are flushed
-            LOG("Interrupted by user\n");
-            common_log_pause(common_log_main());
-            _exit(130);
-        }
-    }
-}
-#endif
-int main(int argc, char ** argv) {
-    common_params params;
-    g_params = &params;
-    if (!common_params_parse(argc, argv, params, LLAMA_EXAMPLE_MAIN, print_usage)) {
-        return 1;
-    }
-    common_init();
-    auto & sparams = params.sampling;
-    // save choice to use color for later
-    // (note for later: this is a slightly awkward choice)
-    console::init(params.simple_io, params.use_color);
-    atexit([]() { console::cleanup(); });
-    if (params.embedding) {
-        LOG_ERR("************\n");
-        LOG_ERR("%s: please use the 'embedding' tool for embedding calculations\n", __func__);
-        LOG_ERR("************\n\n");
-        return 0;
-    }
-    if (params.n_ctx != 0 && params.n_ctx < 8) {
-        LOG_WRN("%s: warning: minimum context size is 8, using minimum size.\n", __func__);
-        params.n_ctx = 8;
-    }
-    if (params.rope_freq_base != 0.0) {
-        LOG_WRN("%s: warning: changing RoPE frequency base to %g.\n", __func__, params.rope_freq_base);
-    }
-    if (params.rope_freq_scale != 0.0) {
-        LOG_WRN("%s: warning: scaling RoPE frequency by %g.\n", __func__, params.rope_freq_scale);
-    }
-    LOG_INF("%s: llama backend init\n", __func__);
-    llama_backend_init();
-    llama_numa_init(params.numa);
-    llama_model * model = nullptr;
-    llama_context * ctx = nullptr;
-    common_sampler * smpl = nullptr;
-    g_model = &model;
-    g_ctx = &ctx;
-    g_smpl = &smpl;
-    std::vector<common_chat_msg> chat_msgs;
-    // load the model and apply lora adapter, if any
-    LOG_INF("%s: load the model and apply lora adapter, if any\n", __func__);
-    common_init_result llama_init = common_init_from_params(params);
-    model = llama_init.model.get();
-    ctx = llama_init.context.get();
-    if (model == NULL) {
-        LOG_ERR("%s: error: unable to load model\n", __func__);
-        return 1;
-    }
-    const llama_vocab * vocab = llama_model_get_vocab(model);
-    auto chat_templates = common_chat_templates_init(model, params.chat_template);
-    LOG_INF("%s: llama threadpool init, n_threads = %d\n", __func__, (int) params.cpuparams.n_threads);
-    auto * cpu_dev = ggml_backend_dev_by_type(GGML_BACKEND_DEVICE_TYPE_CPU);
-    if (!cpu_dev) {
-        LOG_ERR("%s: no CPU backend found\n", __func__);
-        return 1;
-    }
-    auto * reg = ggml_backend_dev_backend_reg(cpu_dev);
-    auto * ggml_threadpool_new_fn = (decltype(ggml_threadpool_new) *) ggml_backend_reg_get_proc_address(reg, "ggml_threadpool_new");
-    auto * ggml_threadpool_free_fn = (decltype(ggml_threadpool_free) *) ggml_backend_reg_get_proc_address(reg, "ggml_threadpool_free");
-    struct ggml_threadpool_params tpp_batch =
-            ggml_threadpool_params_from_cpu_params(params.cpuparams_batch);
-    struct ggml_threadpool_params tpp =
-            ggml_threadpool_params_from_cpu_params(params.cpuparams);
-    set_process_priority(params.cpuparams.priority);
-    struct ggml_threadpool * threadpool_batch = NULL;
-    if (!ggml_threadpool_params_match(&tpp, &tpp_batch)) {
-        threadpool_batch = ggml_threadpool_new_fn(&tpp_batch);
-        if (!threadpool_batch) {
-            LOG_ERR("%s: batch threadpool create failed : n_threads %d\n", __func__, tpp_batch.n_threads);
-            return 1;
-        }
-        // Start the non-batch threadpool in the paused state
-        tpp.paused = true;
-    }
-    struct ggml_threadpool * threadpool = ggml_threadpool_new_fn(&tpp);
-    if (!threadpool) {
-        LOG_ERR("%s: threadpool create failed : n_threads %d\n", __func__, tpp.n_threads);
-        return 1;
-    }
-    llama_attach_threadpool(ctx, threadpool, threadpool_batch);
-    const int n_ctx_train = llama_model_n_ctx_train(model);
-    const int n_ctx = llama_n_ctx(ctx);
-    if (n_ctx > n_ctx_train) {
-        LOG_WRN("%s: model was trained on only %d context tokens (%d specified)\n", __func__, n_ctx_train, n_ctx);
-    }
-    // auto enable conversation mode if chat template is available
-    const bool has_chat_template = common_chat_templates_was_explicit(chat_templates.get());
-    if (params.conversation_mode == COMMON_CONVERSATION_MODE_AUTO) {
-        if (has_chat_template) {
-            LOG_INF("%s: chat template is available, enabling conversation mode (disable it with -no-cnv)\n", __func__);
-            params.conversation_mode = COMMON_CONVERSATION_MODE_ENABLED;
-        } else {
-            params.conversation_mode = COMMON_CONVERSATION_MODE_DISABLED;
-        }
-    }
-    // in case user force-activate conversation mode (via -cnv) without proper chat template, we show a warning
-    if (params.conversation_mode && !has_chat_template) {
-        LOG_WRN("%s: chat template is not available or is not supported. This may cause the model to output suboptimal responses\n", __func__);
-    }
-    // print chat template example in conversation mode
-    if (params.conversation_mode) {
-        if (params.enable_chat_template) {
-            if (!params.prompt.empty() && params.system_prompt.empty()) {
-                LOG_WRN("*** User-specified prompt will pre-start conversation, did you mean to set --system-prompt (-sys) instead?\n");
-            }
-            LOG_INF("%s: chat template example:\n%s\n", __func__, common_chat_format_example(chat_templates.get(), params.use_jinja).c_str());
-        } else {
-            LOG_INF("%s: in-suffix/prefix is specified, chat template will be disabled\n", __func__);
-        }
-    }
-    // print system information
-    {
-        LOG_INF("\n");
-        LOG_INF("%s\n", common_params_get_system_info(params).c_str());
-        LOG_INF("\n");
-    }
-    std::string path_session = params.path_prompt_cache;
-    std::vector<llama_token> session_tokens;
-    if (!path_session.empty()) {
-        LOG_INF("%s: attempting to load saved session from '%s'\n", __func__, path_session.c_str());
-        if (!file_exists(path_session)) {
-            LOG_INF("%s: session file does not exist, will create.\n", __func__);
-        } else if (file_is_empty(path_session)) {
-            LOG_INF("%s: The session file is empty. A new session will be initialized.\n", __func__);
-        } else {
-            // The file exists and is not empty
-            session_tokens.resize(n_ctx);
-            size_t n_token_count_out = 0;
-            if (!llama_state_load_file(ctx, path_session.c_str(), session_tokens.data(), session_tokens.capacity(), &n_token_count_out)) {
-                LOG_ERR("%s: failed to load session file '%s'\n", __func__, path_session.c_str());
-                return 1;
-            }
-            session_tokens.resize(n_token_count_out);
-            LOG_INF("%s: loaded a session with prompt size of %d tokens\n", __func__, (int)session_tokens.size());
-        }
-    }
-    const bool add_bos = llama_vocab_get_add_bos(vocab) && !params.use_jinja;
-    if (!llama_model_has_encoder(model)) {
-        GGML_ASSERT(!llama_vocab_get_add_eos(vocab));
-    }
-    LOG_DBG("n_ctx: %d, add_bos: %d\n", n_ctx, add_bos);
-    std::vector<llama_token> embd_inp;
-    bool waiting_for_first_input = false;
-    auto chat_add_and_format = [&chat_msgs, &chat_templates](const std::string & role, const std::string & content) {
-        common_chat_msg new_msg;
-        new_msg.role = role;
-        new_msg.content = content;
-        auto formatted = common_chat_format_single(chat_templates.get(), chat_msgs, new_msg, role == "user", g_params->use_jinja);
-        chat_msgs.push_back(new_msg);
-        LOG_DBG("formatted: '%s'\n", formatted.c_str());
-        return formatted;
-    };
-    std::string prompt;
-    {
-        if (params.conversation_mode && params.enable_chat_template) {
-            if (!params.system_prompt.empty()) {
-                // format the system prompt (will use template default if empty)
-                chat_add_and_format("system", params.system_prompt);
-            }
-            if (!params.prompt.empty()) {
-                // format and append the user prompt
-                chat_add_and_format("user", params.prompt);
-            } else {
-                waiting_for_first_input = true;
-            }
-            if (!params.system_prompt.empty() || !params.prompt.empty()) {
-                common_chat_templates_inputs inputs;
-                inputs.messages = chat_msgs;
-                inputs.add_generation_prompt = !params.prompt.empty();
-                prompt = common_chat_templates_apply(chat_templates.get(), inputs).prompt;
-            }
-        } else {
-            // otherwise use the prompt as is
-            prompt = params.prompt;
-        }
-        if (params.interactive_first || !prompt.empty() || session_tokens.empty()) {
-            LOG_DBG("tokenize the prompt\n");
-            embd_inp = common_tokenize(ctx, prompt, true, true);
-        } else {
-            LOG_DBG("use session tokens\n");
-            embd_inp = session_tokens;
-        }
-        LOG_DBG("prompt: \"%s\"\n", prompt.c_str());
-        LOG_DBG("tokens: %s\n", string_from(ctx, embd_inp).c_str());
-    }
-    // Should not run without any tokens
-    if (!waiting_for_first_input && embd_inp.empty()) {
-        if (add_bos) {
-            embd_inp.push_back(llama_vocab_bos(vocab));
-            LOG_WRN("embd_inp was considered empty and bos was added: %s\n", string_from(ctx, embd_inp).c_str());
-        } else {
-            LOG_ERR("input is empty\n");
-            return -1;
-        }
-    }
-    // Tokenize negative prompt
-    if ((int) embd_inp.size() > n_ctx - 4) {
-        LOG_ERR("%s: prompt is too long (%d tokens, max %d)\n", __func__, (int) embd_inp.size(), n_ctx - 4);
-        return 1;
-    }
-    // debug message about similarity of saved session, if applicable
-    size_t n_matching_session_tokens = 0;
-    if (!session_tokens.empty()) {
-        for (llama_token id : session_tokens) {
-            if (n_matching_session_tokens >= embd_inp.size() || id != embd_inp[n_matching_session_tokens]) {
-                break;
-            }
-            n_matching_session_tokens++;
-        }
-        if (params.prompt.empty() && n_matching_session_tokens == embd_inp.size()) {
-            LOG_INF("%s: using full prompt from session file\n", __func__);
-        } else if (n_matching_session_tokens >= embd_inp.size()) {
-            LOG_INF("%s: session file has exact match for prompt!\n", __func__);
-        } else if (n_matching_session_tokens < (embd_inp.size() / 2)) {
-            LOG_WRN("%s: session file has low similarity to prompt (%zu / %zu tokens); will mostly be reevaluated\n",
-                    __func__, n_matching_session_tokens, embd_inp.size());
-        } else {
-            LOG_INF("%s: session file matches %zu / %zu tokens of prompt\n",
-                    __func__, n_matching_session_tokens, embd_inp.size());
-        }
-        // remove any "future" tokens that we might have inherited from the previous session
-        llama_kv_self_seq_rm(ctx, -1, n_matching_session_tokens, -1);
-    }
-    LOG_DBG("recalculate the cached logits (check): embd_inp.size() %zu, n_matching_session_tokens %zu, embd_inp.size() %zu, session_tokens.size() %zu\n",
-         embd_inp.size(), n_matching_session_tokens, embd_inp.size(), session_tokens.size());
-    // if we will use the cache for the full prompt without reaching the end of the cache, force
-    // reevaluation of the last token to recalculate the cached logits
-    if (!embd_inp.empty() && n_matching_session_tokens == embd_inp.size() && session_tokens.size() > embd_inp.size()) {
-        LOG_DBG("recalculate the cached logits (do): session_tokens.resize( %zu )\n", embd_inp.size() - 1);
-        session_tokens.resize(embd_inp.size() - 1);
-    }
-    // number of tokens to keep when resetting context
-    if (params.n_keep < 0 || params.n_keep > (int) embd_inp.size()) {
-        params.n_keep = (int)embd_inp.size();
-    } else {
-        params.n_keep += add_bos; // always keep the BOS token
-    }
-    if (params.conversation_mode) {
-        if (params.single_turn && !params.prompt.empty()) {
-            params.interactive = false;
-            params.interactive_first = false;
-        } else {
-            params.interactive_first = true;
-        }
-    }
-    // enable interactive mode if interactive start is specified
-    if (params.interactive_first) {
-        params.interactive = true;
-    }
-    if (params.verbose_prompt) {
-        LOG_INF("%s: prompt: '%s'\n", __func__, params.prompt.c_str());
-        LOG_INF("%s: number of tokens in prompt = %zu\n", __func__, embd_inp.size());
-        for (int i = 0; i < (int) embd_inp.size(); i++) {
-            LOG_INF("%6d -> '%s'\n", embd_inp[i], common_token_to_piece(ctx, embd_inp[i]).c_str());
-        }
-        if (params.n_keep > add_bos) {
-            LOG_INF("%s: static prompt based on n_keep: '", __func__);
-            for (int i = 0; i < params.n_keep; i++) {
-                LOG_CNT("%s", common_token_to_piece(ctx, embd_inp[i]).c_str());
-            }
-            LOG_CNT("'\n");
-        }
-        LOG_INF("\n");
-    }
-    // ctrl+C handling
-    {
-#if defined (__unix__) || (defined (__APPLE__) && defined (__MACH__))
-        struct sigaction sigint_action;
-        sigint_action.sa_handler = sigint_handler;
-        sigemptyset (&sigint_action.sa_mask);
-        sigint_action.sa_flags = 0;
-        sigaction(SIGINT, &sigint_action, NULL);
-#elif defined (_WIN32)
-        auto console_ctrl_handler = +[](DWORD ctrl_type) -> BOOL {
-            return (ctrl_type == CTRL_C_EVENT) ? (sigint_handler(SIGINT), true) : false;
-        };
-        SetConsoleCtrlHandler(reinterpret_cast<PHANDLER_ROUTINE>(console_ctrl_handler), true);
-#endif
-    }
-    if (params.interactive) {
-        LOG_INF("%s: interactive mode on.\n", __func__);
-        if (!params.antiprompt.empty()) {
-            for (const auto & antiprompt : params.antiprompt) {
-                LOG_INF("Reverse prompt: '%s'\n", antiprompt.c_str());
-                if (params.verbose_prompt) {
-                    auto tmp = common_tokenize(ctx, antiprompt, false, true);
-                    for (int i = 0; i < (int) tmp.size(); i++) {
-                        LOG_INF("%6d -> '%s'\n", tmp[i], common_token_to_piece(ctx, tmp[i]).c_str());
-                    }
-                }
-            }
-        }
-        if (params.input_prefix_bos) {
-            LOG_INF("Input prefix with BOS\n");
-        }
-        if (!params.input_prefix.empty()) {
-            LOG_INF("Input prefix: '%s'\n", params.input_prefix.c_str());
-            if (params.verbose_prompt) {
-                auto tmp = common_tokenize(ctx, params.input_prefix, true, true);
-                for (int i = 0; i < (int) tmp.size(); i++) {
-                    LOG_INF("%6d -> '%s'\n", tmp[i], common_token_to_piece(ctx, tmp[i]).c_str());
-                }
-            }
-        }
-        if (!params.input_suffix.empty()) {
-            LOG_INF("Input suffix: '%s'\n", params.input_suffix.c_str());
-            if (params.verbose_prompt) {
-                auto tmp = common_tokenize(ctx, params.input_suffix, false, true);
-                for (int i = 0; i < (int) tmp.size(); i++) {
-                    LOG_INF("%6d -> '%s'\n", tmp[i], common_token_to_piece(ctx, tmp[i]).c_str());
-                }
-            }
-        }
-    }
-    smpl = common_sampler_init(model, sparams);
-    if (!smpl) {
-        LOG_ERR("%s: failed to initialize sampling subsystem\n", __func__);
-        return 1;
-    }
-    LOG_INF("sampler seed: %u\n",     common_sampler_get_seed(smpl));
-    LOG_INF("sampler params: \n%s\n", sparams.print().c_str());
-    LOG_INF("sampler chain: %s\n",    common_sampler_print(smpl).c_str());
-    LOG_INF("generate: n_ctx = %d, n_batch = %d, n_predict = %d, n_keep = %d\n", n_ctx, params.n_batch, params.n_predict, params.n_keep);
-    // group-attention state
-    // number of grouped KV tokens so far (used only if params.grp_attn_n > 1)
-    int ga_i = 0;
-    const int ga_n = params.grp_attn_n;
-    const int ga_w = params.grp_attn_w;
-    if (ga_n != 1) {
-        GGML_ASSERT(ga_n > 0                    && "grp_attn_n must be positive");                     // NOLINT
-        GGML_ASSERT(ga_w % ga_n == 0            && "grp_attn_w must be a multiple of grp_attn_n");     // NOLINT
-      //GGML_ASSERT(n_ctx_train % ga_w == 0     && "n_ctx_train must be a multiple of grp_attn_w");    // NOLINT
-      //GGML_ASSERT(n_ctx >= n_ctx_train * ga_n && "n_ctx must be at least n_ctx_train * grp_attn_n"); // NOLINT
-        LOG_INF("self-extend: n_ctx_train = %d, grp_attn_n = %d, grp_attn_w = %d\n", n_ctx_train, ga_n, ga_w);
-    }
-    LOG_INF("\n");
-    if (params.interactive) {
-        const char * control_message;
-        if (params.multiline_input) {
-            control_message = " - To return control to the AI, end your input with '\\'.\n"
-                              " - To return control without starting a new line, end your input with '/'.\n";
-        } else {
-            control_message = " - Press Return to return control to the AI.\n"
-                              " - To return control without starting a new line, end your input with '/'.\n"
-                              " - If you want to submit another line, end your input with '\\'.\n";
-        }
-        LOG_INF("== Running in interactive mode. ==\n");
-#if defined (__unix__) || (defined (__APPLE__) && defined (__MACH__)) || defined (_WIN32)
-        LOG_INF(       " - Press Ctrl+C to interject at any time.\n");
-#endif
-        LOG_INF(       "%s", control_message);
-        if (params.conversation_mode && params.enable_chat_template && params.system_prompt.empty()) {
-            LOG_INF(   " - Not using system message. To change it, set a different value via -sys PROMPT\n");
-        }
-        LOG_INF("\n");
-        is_interacting = params.interactive_first;
-    }
-    bool is_antiprompt        = false;
-    bool input_echo           = true;
-    bool display              = true;
-    bool need_to_save_session = !path_session.empty() && n_matching_session_tokens < embd_inp.size();
-    int n_past             = 0;
-    int n_remain           = params.n_predict;
-    int n_consumed         = 0;
-    int n_session_consumed = 0;
-    std::vector<int>   input_tokens;  g_input_tokens  = &input_tokens;
-    std::vector<int>   output_tokens; g_output_tokens = &output_tokens;
-    std::ostringstream output_ss;     g_output_ss     = &output_ss;
-    std::ostringstream assistant_ss; // for storing current assistant message, used in conversation mode
-    // the first thing we will do is to output the prompt, so set color accordingly
-    console::set_display(console::prompt);
-    display = params.display_prompt;
-    std::vector<llama_token> embd;
-    // single-token antiprompts
-    std::vector<llama_token> antiprompt_token;
-    for (const std::string & antiprompt : params.antiprompt) {
-        auto ids = ::common_tokenize(ctx, antiprompt, false, true);
-        if (ids.size() == 1) {
-            antiprompt_token.push_back(ids[0]);
-        }
-    }
-    if (llama_model_has_encoder(model)) {
-        int enc_input_size = embd_inp.size();
-        llama_token * enc_input_buf = embd_inp.data();
-        if (llama_encode(ctx, llama_batch_get_one(enc_input_buf, enc_input_size))) {
-            LOG_ERR("%s : failed to eval\n", __func__);
-            return 1;
-        }
-        llama_token decoder_start_token_id = llama_model_decoder_start_token(model);
-        if (decoder_start_token_id == LLAMA_TOKEN_NULL) {
-            decoder_start_token_id = llama_vocab_bos(vocab);
-        }
-        embd_inp.clear();
-        embd_inp.push_back(decoder_start_token_id);
-    }
-    while ((n_remain != 0 && !is_antiprompt) || params.interactive) {
-        // predict
-        if (!embd.empty()) {
-            // Note: (n_ctx - 4) here is to match the logic for commandline prompt handling via
-            // --prompt or --file which uses the same value.
-            int max_embd_size = n_ctx - 4;
-            // Ensure the input doesn't exceed the context size by truncating embd if necessary.
-            if ((int) embd.size() > max_embd_size) {
-                const int skipped_tokens = (int) embd.size() - max_embd_size;
-                embd.resize(max_embd_size);
-                console::set_display(console::error);
-                LOG_WRN("<<input too long: skipped %d token%s>>", skipped_tokens, skipped_tokens != 1 ? "s" : "");
-                console::set_display(console::reset);
-            }
-            if (ga_n == 1) {
-                // infinite text generation via context shifting
-                // if we run out of context:
-                // - take the n_keep first tokens from the original prompt (via n_past)
-                // - take half of the last (n_ctx - n_keep) tokens and recompute the logits in batches
-                if (n_past + (int) embd.size() >= n_ctx) {
-                    if (!params.ctx_shift){
-                        LOG_DBG("\n\n%s: context full and context shift is disabled => stopping\n", __func__);
-                        break;
-                    }
-                    if (params.n_predict == -2) {
-                        LOG_DBG("\n\n%s: context full and n_predict == -%d => stopping\n", __func__, params.n_predict);
-                        break;
-                    }
-                    const int n_left    = n_past - params.n_keep;
-                    const int n_discard = n_left/2;
-                    LOG_DBG("context full, swapping: n_past = %d, n_left = %d, n_ctx = %d, n_keep = %d, n_discard = %d\n",
-                            n_past, n_left, n_ctx, params.n_keep, n_discard);
-                    llama_kv_self_seq_rm (ctx, 0, params.n_keep            , params.n_keep + n_discard);
-                    llama_kv_self_seq_add(ctx, 0, params.n_keep + n_discard, n_past, -n_discard);
-                    n_past -= n_discard;
-                    LOG_DBG("after swap: n_past = %d\n", n_past);
-                    LOG_DBG("embd: %s\n", string_from(ctx, embd).c_str());
-                    LOG_DBG("clear session path\n");
-                    path_session.clear();
-                }
-            } else {
-                // context extension via Self-Extend
-                while (n_past >= ga_i + ga_w) {
-                    const int ib = (ga_n*ga_i)/ga_w;
-                    const int bd = (ga_w/ga_n)*(ga_n - 1);
-                    const int dd = (ga_w/ga_n) - ib*bd - ga_w;
-                    LOG_DBG("\n");
-                    LOG_DBG("shift: [%6d, %6d] + %6d -> [%6d, %6d]\n", ga_i, n_past, ib*bd, ga_i + ib*bd, n_past + ib*bd);
-                    LOG_DBG("div:   [%6d, %6d] / %6d -> [%6d, %6d]\n", ga_i + ib*bd, ga_i + ib*bd + ga_w, ga_n, (ga_i + ib*bd)/ga_n, (ga_i + ib*bd + ga_w)/ga_n);
-                    LOG_DBG("shift: [%6d, %6d] + %6d -> [%6d, %6d]\n", ga_i + ib*bd + ga_w, n_past + ib*bd, dd, ga_i + ib*bd + ga_w + dd, n_past + ib*bd + dd);
-                    llama_kv_self_seq_add(ctx, 0, ga_i,                n_past,              ib*bd);
-                    llama_kv_self_seq_div(ctx, 0, ga_i + ib*bd,        ga_i + ib*bd + ga_w, ga_n);
-                    llama_kv_self_seq_add(ctx, 0, ga_i + ib*bd + ga_w, n_past + ib*bd,      dd);
-                    n_past -= bd;
-                    ga_i += ga_w/ga_n;
-                    LOG_DBG("\nn_past_old = %d, n_past = %d, ga_i = %d\n\n", n_past + bd, n_past, ga_i);
-                }
-            }
-            // try to reuse a matching prefix from the loaded session instead of re-eval (via n_past)
-            if (n_session_consumed < (int) session_tokens.size()) {
-                size_t i = 0;
-                for ( ; i < embd.size(); i++) {
-                    if (embd[i] != session_tokens[n_session_consumed]) {
-                        session_tokens.resize(n_session_consumed);
-                        break;
-                    }
-                    n_past++;
-                    n_session_consumed++;
-                    if (n_session_consumed >= (int) session_tokens.size()) {
-                        ++i;
-                        break;
-                    }
-                }
-                if (i > 0) {
-                    embd.erase(embd.begin(), embd.begin() + i);
-                }
-            }
-            for (int i = 0; i < (int) embd.size(); i += params.n_batch) {
-                int n_eval = (int) embd.size() - i;
-                if (n_eval > params.n_batch) {
-                    n_eval = params.n_batch;
-                }
-                LOG_DBG("eval: %s\n", string_from(ctx, embd).c_str());
-                if (llama_decode(ctx, llama_batch_get_one(&embd[i], n_eval))) {
-                    LOG_ERR("%s : failed to eval\n", __func__);
-                    return 1;
-                }
-                n_past += n_eval;
-                LOG_DBG("n_past = %d\n", n_past);
-                // Display total tokens alongside total time
-                if (params.n_print > 0 && n_past % params.n_print == 0) {
-                    LOG_DBG("\n\033[31mTokens consumed so far = %d / %d \033[0m\n", n_past, n_ctx);
-                }
-            }
-            if (!embd.empty() && !path_session.empty()) {
-                session_tokens.insert(session_tokens.end(), embd.begin(), embd.end());
-                n_session_consumed = session_tokens.size();
-            }
-        }
-        embd.clear();
-        if ((int) embd_inp.size() <= n_consumed && !is_interacting) {
-            // optionally save the session on first sample (for faster prompt loading next time)
-            if (!path_session.empty() && need_to_save_session && !params.prompt_cache_ro) {
-                need_to_save_session = false;
-                llama_state_save_file(ctx, path_session.c_str(), session_tokens.data(), session_tokens.size());
-                LOG_DBG("saved session to %s\n", path_session.c_str());
-            }
-            const llama_token id = common_sampler_sample(smpl, ctx, -1);
-            common_sampler_accept(smpl, id, /* accept_grammar= */ true);
-            // LOG_DBG("last: %s\n", string_from(ctx, smpl->prev.to_vector()).c_str());
-            embd.push_back(id);
-            // echo this to console
-            input_echo = true;
-            // decrement remaining sampling budget
-            --n_remain;
-            LOG_DBG("n_remain: %d\n", n_remain);
-        } else {
-            // some user input remains from prompt or interaction, forward it to processing
-            LOG_DBG("embd_inp.size(): %d, n_consumed: %d\n", (int) embd_inp.size(), n_consumed);
-            while ((int) embd_inp.size() > n_consumed) {
-                embd.push_back(embd_inp[n_consumed]);
-                // push the prompt in the sampling context in order to apply repetition penalties later
-                // for the prompt, we don't apply grammar rules
-                common_sampler_accept(smpl, embd_inp[n_consumed], /* accept_grammar= */ false);
-                ++n_consumed;
-                if ((int) embd.size() >= params.n_batch) {
-                    break;
-                }
-            }
-        }
-        // display text
-        if (input_echo && display) {
-            for (auto id : embd) {
-                const std::string token_str = common_token_to_piece(ctx, id, params.special);
-                // Console/Stream Output
-                LOG("%s", token_str.c_str());
-                // Record Displayed Tokens To Log
-                // Note: Generated tokens are created one by one hence this check
-                if (embd.size() > 1) {
-                    // Incoming Requested Tokens
-                    input_tokens.push_back(id);
-                } else {
-                    // Outgoing Generated Tokens
-                    output_tokens.push_back(id);
-                    output_ss << token_str;
-                }
-            }
-        }
-        // reset color to default if there is no pending user input
-        if (input_echo && (int) embd_inp.size() == n_consumed) {
-            console::set_display(console::reset);
-            display = true;
-        }
-        // if not currently processing queued inputs;
-        if ((int) embd_inp.size() <= n_consumed) {
-            // check for reverse prompt in the last n_prev tokens
-            if (!params.antiprompt.empty()) {
-                const int n_prev = 32;
-                const std::string last_output = common_sampler_prev_str(smpl, ctx, n_prev);
-                is_antiprompt = false;
-                // Check if each of the reverse prompts appears at the end of the output.
-                // If we're not running interactively, the reverse prompt might be tokenized with some following characters
-                // so we'll compensate for that by widening the search window a bit.
-                for (std::string & antiprompt : params.antiprompt) {
-                    size_t extra_padding = params.interactive ? 0 : 2;
-                    size_t search_start_pos = last_output.length() > static_cast<size_t>(antiprompt.length() + extra_padding)
-                        ? last_output.length() - static_cast<size_t>(antiprompt.length() + extra_padding)
-                        : 0;
-                    if (last_output.find(antiprompt, search_start_pos) != std::string::npos) {
-                        if (params.interactive) {
-                            is_interacting = true;
-                        }
-                        is_antiprompt = true;
-                        break;
-                    }
-                }
-                // check for reverse prompt using special tokens
-                llama_token last_token = common_sampler_last(smpl);
-                for (auto token : antiprompt_token) {
-                    if (token == last_token) {
-                        if (params.interactive) {
-                            is_interacting = true;
-                        }
-                        is_antiprompt = true;
-                        break;
-                    }
-                }
-                if (is_antiprompt) {
-                    LOG_DBG("found antiprompt: %s\n", last_output.c_str());
-                }
-            }
-            // deal with end of generation tokens in interactive mode
-            if (!waiting_for_first_input && llama_vocab_is_eog(vocab, common_sampler_last(smpl))) {
-                LOG_DBG("found an EOG token\n");
-                if (params.interactive) {
-                    if (!params.antiprompt.empty()) {
-                        // tokenize and inject first reverse prompt
-                        const auto first_antiprompt = common_tokenize(ctx, params.antiprompt.front(), false, true);
-                        embd_inp.insert(embd_inp.end(), first_antiprompt.begin(), first_antiprompt.end());
-                        is_antiprompt = true;
-                    }
-                    if (params.enable_chat_template) {
-                        chat_add_and_format("assistant", assistant_ss.str());
-                    }
-                    is_interacting = true;
-                    LOG("\n");
-                }
-            }
-            // if current token is not EOG, we add it to current assistant message
-            if (params.conversation_mode && !waiting_for_first_input) {
-                const auto id = common_sampler_last(smpl);
-                assistant_ss << common_token_to_piece(ctx, id, false);
-                if (!prompt.empty()) {
-                    prompt.clear();
-                    is_interacting = false;
-                }
-            }
-            if ((n_past > 0 || waiting_for_first_input) && is_interacting) {
-                LOG_DBG("waiting for user input\n");
-                if (params.conversation_mode) {
-                    LOG("\n> ");
-                }
-                if (params.input_prefix_bos) {
-                    LOG_DBG("adding input prefix BOS token\n");
-                    embd_inp.push_back(llama_vocab_bos(vocab));
-                }
-                std::string buffer;
-                if (!params.input_prefix.empty() && !params.conversation_mode) {
-                    LOG_DBG("appending input prefix: '%s'\n", params.input_prefix.c_str());
-                    LOG("%s", params.input_prefix.c_str());
-                }
-                // color user input only
-                console::set_display(console::user_input);
-                display = params.display_prompt;
-                std::string line;
-                bool another_line = true;
-                do {
-                    another_line = console::readline(line, params.multiline_input);
-                    buffer += line;
-                } while (another_line);
-                // done taking input, reset color
-                console::set_display(console::reset);
-                display = true;
-                if (buffer.empty()) { // Ctrl+D on empty line exits
-                    LOG("EOF by user\n");
-                    break;
-                }
-                if (buffer.back() == '\n') {
-                    // Implement #587:
-                    // If the user wants the text to end in a newline,
-                    // this should be accomplished by explicitly adding a newline by using \ followed by return,
-                    // then returning control by pressing return again.
-                    buffer.pop_back();
-                }
-                if (buffer.empty()) { // Enter key on empty line lets the user pass control back
-                    LOG_DBG("empty line, passing control back\n");
-                } else { // Add tokens to embd only if the input buffer is non-empty
-                    // append input suffix if any
-                    if (!params.input_suffix.empty() && !params.conversation_mode) {
-                        LOG_DBG("appending input suffix: '%s'\n", params.input_suffix.c_str());
-                        LOG("%s", params.input_suffix.c_str());
-                    }
-                    LOG_DBG("buffer: '%s'\n", buffer.c_str());
-                    const size_t original_size = embd_inp.size();
-                    if (params.escape) {
-                        string_process_escapes(buffer);
-                    }
-                    bool format_chat = params.conversation_mode && params.enable_chat_template;
-                    std::string user_inp = format_chat
-                        ? chat_add_and_format("user", std::move(buffer))
-                        : std::move(buffer);
-                    // TODO: one inconvenient of current chat template implementation is that we can't distinguish between user input and special tokens (prefix/postfix)
-                    const auto line_pfx = common_tokenize(ctx, params.input_prefix, false, true);
-                    const auto line_inp = common_tokenize(ctx, user_inp,            false, format_chat);
-                    const auto line_sfx = common_tokenize(ctx, params.input_suffix, false, true);
-                    LOG_DBG("input tokens: %s\n", string_from(ctx, line_inp).c_str());
-                    // if user stop generation mid-way, we must add EOT to finish model's last response
-                    if (need_insert_eot && format_chat) {
-                        llama_token eot = llama_vocab_eot(vocab);
-                        embd_inp.push_back(eot == LLAMA_TOKEN_NULL ? llama_vocab_eos(vocab) : eot);
-                        need_insert_eot = false;
-                    }
-                    embd_inp.insert(embd_inp.end(), line_pfx.begin(), line_pfx.end());
-                    embd_inp.insert(embd_inp.end(), line_inp.begin(), line_inp.end());
-                    embd_inp.insert(embd_inp.end(), line_sfx.begin(), line_sfx.end());
-                    for (size_t i = original_size; i < embd_inp.size(); ++i) {
-                        const llama_token token = embd_inp[i];
-                        output_tokens.push_back(token);
-                        output_ss << common_token_to_piece(ctx, token);
-                    }
-                    // reset assistant message
-                    assistant_ss.str("");
-                    n_remain -= line_inp.size();
-                    LOG_DBG("n_remain: %d\n", n_remain);
-                }
-                input_echo = false; // do not echo this again
-            }
-            if (n_past > 0 || waiting_for_first_input) {
-                if (is_interacting) {
-                    common_sampler_reset(smpl);
-                }
-                is_interacting = false;
-                if (waiting_for_first_input && params.single_turn) {
-                    params.interactive = false;
-                    params.interactive_first = false;
-                }
-                waiting_for_first_input = false;
-            }
-        }
-        // end of generation
-        if (!embd.empty() && llama_vocab_is_eog(vocab, embd.back()) && !(params.interactive)) {
-            LOG(" [end of text]\n");
-            break;
-        }
-        // In interactive mode, respect the maximum number of tokens and drop back to user input when reached.
-        // We skip this logic when n_predict == -1 (infinite) or -2 (stop at context size).
-        if (params.interactive && n_remain <= 0 && params.n_predict >= 0) {
-            n_remain = params.n_predict;
-            is_interacting = true;
-        }
-    }
-    if (!path_session.empty() && params.prompt_cache_all && !params.prompt_cache_ro) {
-        LOG("\n%s: saving final output to session file '%s'\n", __func__, path_session.c_str());
-        llama_state_save_file(ctx, path_session.c_str(), session_tokens.data(), session_tokens.size());
-    }
-    LOG("\n\n");
-    common_perf_print(ctx, smpl);
-    common_sampler_free(smpl);
-    llama_backend_free();
-    ggml_threadpool_free_fn(threadpool);
-    ggml_threadpool_free_fn(threadpool_batch);
-    return 0;
-}