@hawkeyexl/inference 0.3.1 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/chunk-UFHBIS5K.js +185 -0
- package/dist/chunk-UFHBIS5K.js.map +1 -0
- package/dist/index.d.ts +3 -152
- package/dist/index.js +628 -95
- package/dist/index.js.map +1 -1
- package/dist/llama-cpp-CdVKP63Z.d.ts +308 -0
- package/dist/llama-worker.d.ts +1 -0
- package/dist/llama-worker.js +9 -0
- package/dist/llama-worker.js.map +1 -0
- package/package.json +2 -2
package/dist/index.js
CHANGED
|
@@ -1,3 +1,8 @@
|
|
|
1
|
+
import {
|
|
2
|
+
WORKER_ENV_FLAG,
|
|
3
|
+
openBackend
|
|
4
|
+
} from "./chunk-UFHBIS5K.js";
|
|
5
|
+
|
|
1
6
|
// src/types.ts
|
|
2
7
|
var InferenceError = class extends Error {
|
|
3
8
|
constructor(message) {
|
|
@@ -596,6 +601,11 @@ function isModelPathOrUri(model) {
|
|
|
596
601
|
return /^(hf|huggingface):/i.test(model) || /^https?:\/\//i.test(model) || /^(hf|huggingface)\.co\//i.test(model) || model.endsWith(".gguf");
|
|
597
602
|
}
|
|
598
603
|
|
|
604
|
+
// src/providers/llama-host.ts
|
|
605
|
+
import { fork } from "child_process";
|
|
606
|
+
import { existsSync as existsSync3 } from "fs";
|
|
607
|
+
import { fileURLToPath } from "url";
|
|
608
|
+
|
|
599
609
|
// src/providers/llama-install.ts
|
|
600
610
|
import {
|
|
601
611
|
existsSync as existsSync2,
|
|
@@ -616,6 +626,9 @@ var LOCK_STALE_MS = INSTALL_TIMEOUT_MS + 6e4;
|
|
|
616
626
|
function defaultLlamaRuntimeDirectory(env = process.env) {
|
|
617
627
|
return env["INFERENCE_RUNTIME_DIR"] || join3(homedir2(), ".hawkeyexl-inference", "runtime");
|
|
618
628
|
}
|
|
629
|
+
function nodeLlamaCppShimUrl(directory = defaultLlamaRuntimeDirectory()) {
|
|
630
|
+
return pathToFileURL(join3(directory, SHIM)).href;
|
|
631
|
+
}
|
|
619
632
|
function isModuleNotFound(e) {
|
|
620
633
|
const code = e?.code;
|
|
621
634
|
return code === "ERR_MODULE_NOT_FOUND" || code === "MODULE_NOT_FOUND";
|
|
@@ -786,6 +799,554 @@ function delay(ms) {
|
|
|
786
799
|
return new Promise((resolve) => setTimeout(resolve, ms));
|
|
787
800
|
}
|
|
788
801
|
|
|
802
|
+
// src/providers/llama-host.ts
|
|
803
|
+
var LLAMA_GPUS = ["auto", "cuda", "vulkan", "metal", false];
|
|
804
|
+
function isLlamaGpu(value) {
|
|
805
|
+
return LLAMA_GPUS.includes(value);
|
|
806
|
+
}
|
|
807
|
+
var STDERR_TAIL_LINES = 20;
|
|
808
|
+
var SHUTDOWN_GRACE_MS = 5e3;
|
|
809
|
+
var STDERR_DRAIN_MS = 250;
|
|
810
|
+
var GPU_OFF_VALUES = ["false", "off", "none", "disable", "disabled"];
|
|
811
|
+
var LlamaWorkerCrash = class extends Error {
|
|
812
|
+
constructor(host, code, signal, blameless = false) {
|
|
813
|
+
super("The local-model worker exited.");
|
|
814
|
+
this.host = host;
|
|
815
|
+
this.code = code;
|
|
816
|
+
this.signal = signal;
|
|
817
|
+
this.blameless = blameless;
|
|
818
|
+
this.name = "LlamaWorkerCrash";
|
|
819
|
+
}
|
|
820
|
+
host;
|
|
821
|
+
code;
|
|
822
|
+
signal;
|
|
823
|
+
blameless;
|
|
824
|
+
};
|
|
825
|
+
var WorkerHost = class {
|
|
826
|
+
child;
|
|
827
|
+
/** The backend that initialised, once it has. */
|
|
828
|
+
gpu;
|
|
829
|
+
/** The backend initialising now, so a crash during `init` is attributable. */
|
|
830
|
+
trying;
|
|
831
|
+
dead = false;
|
|
832
|
+
/** How a crash of this host was handled, decided once for every caller. */
|
|
833
|
+
verdict;
|
|
834
|
+
exited;
|
|
835
|
+
closing = false;
|
|
836
|
+
nextId = 1;
|
|
837
|
+
pending = /* @__PURE__ */ new Map();
|
|
838
|
+
stderrTail = [];
|
|
839
|
+
partialLine = "";
|
|
840
|
+
constructor(entry) {
|
|
841
|
+
this.child = fork(entry, [], {
|
|
842
|
+
// stderr is piped rather than inherited so a crash can name its last
|
|
843
|
+
// line; it is still forwarded live, so llama.cpp's output is not lost.
|
|
844
|
+
stdio: ["ignore", "inherit", "pipe", "ipc"],
|
|
845
|
+
serialization: "json",
|
|
846
|
+
// Never the parent's flags (an inspector port, a test runner's loader).
|
|
847
|
+
// From `src/` Node strips the worker's types itself; say nothing about it.
|
|
848
|
+
execArgv: entry.endsWith(".ts") ? ["--disable-warning=ExperimentalWarning"] : [],
|
|
849
|
+
env: { ...process.env, [WORKER_ENV_FLAG]: "1" }
|
|
850
|
+
});
|
|
851
|
+
const stderr = this.child.stderr;
|
|
852
|
+
stderr.setEncoding("utf8");
|
|
853
|
+
stderr.on("data", (chunk) => {
|
|
854
|
+
process.stderr.write(chunk);
|
|
855
|
+
this.collect(chunk);
|
|
856
|
+
});
|
|
857
|
+
this.child.on("message", (message) => this.onMessage(message));
|
|
858
|
+
this.exited = new Promise((resolve) => {
|
|
859
|
+
this.child.once("error", (e) => {
|
|
860
|
+
this.dead = true;
|
|
861
|
+
this.rejectAll(
|
|
862
|
+
new InferenceError(`Could not start the local-model worker (${e.message}).`)
|
|
863
|
+
);
|
|
864
|
+
resolve();
|
|
865
|
+
});
|
|
866
|
+
this.child.once("exit", (code, signal) => {
|
|
867
|
+
this.dead = true;
|
|
868
|
+
void this.drainStderr().then(() => {
|
|
869
|
+
this.onExit(code, signal);
|
|
870
|
+
resolve();
|
|
871
|
+
});
|
|
872
|
+
});
|
|
873
|
+
});
|
|
874
|
+
this.idle();
|
|
875
|
+
}
|
|
876
|
+
get pid() {
|
|
877
|
+
return this.dead ? void 0 : this.child.pid;
|
|
878
|
+
}
|
|
879
|
+
get stderr() {
|
|
880
|
+
return this.partialLine ? [...this.stderrTail, this.partialLine] : this.stderrTail;
|
|
881
|
+
}
|
|
882
|
+
request(request) {
|
|
883
|
+
if (this.dead) {
|
|
884
|
+
return Promise.reject(new LlamaWorkerCrash(this, null, null, true));
|
|
885
|
+
}
|
|
886
|
+
const id = this.nextId++;
|
|
887
|
+
return new Promise((resolve, reject) => {
|
|
888
|
+
this.pending.set(id, {
|
|
889
|
+
resolve,
|
|
890
|
+
reject
|
|
891
|
+
});
|
|
892
|
+
if (this.pending.size === 1) this.busy();
|
|
893
|
+
this.child.send({ ...request, id }, (e) => {
|
|
894
|
+
if (e && this.pending.delete(id)) {
|
|
895
|
+
if (this.pending.size === 0) this.idle();
|
|
896
|
+
reject(new LlamaWorkerCrash(this, null, null, true));
|
|
897
|
+
}
|
|
898
|
+
});
|
|
899
|
+
});
|
|
900
|
+
}
|
|
901
|
+
async shutdown() {
|
|
902
|
+
if (this.dead) return;
|
|
903
|
+
this.closing = true;
|
|
904
|
+
this.busy();
|
|
905
|
+
try {
|
|
906
|
+
this.child.send({ id: this.nextId++, op: "shutdown" });
|
|
907
|
+
} catch {
|
|
908
|
+
}
|
|
909
|
+
const timer = setTimeout(() => this.child.kill(), SHUTDOWN_GRACE_MS);
|
|
910
|
+
await this.exited;
|
|
911
|
+
clearTimeout(timer);
|
|
912
|
+
}
|
|
913
|
+
onMessage(message) {
|
|
914
|
+
if ("event" in message) {
|
|
915
|
+
this.trying = message.gpu;
|
|
916
|
+
return;
|
|
917
|
+
}
|
|
918
|
+
const pending = this.pending.get(message.id);
|
|
919
|
+
if (!pending) return;
|
|
920
|
+
this.pending.delete(message.id);
|
|
921
|
+
if (this.pending.size === 0) this.idle();
|
|
922
|
+
if (message.ok) {
|
|
923
|
+
pending.resolve(message.value);
|
|
924
|
+
} else {
|
|
925
|
+
pending.reject(rebuildError(message.error));
|
|
926
|
+
}
|
|
927
|
+
}
|
|
928
|
+
onExit(code, signal) {
|
|
929
|
+
if (this.pending.size === 0) return;
|
|
930
|
+
this.rejectAll(
|
|
931
|
+
this.closing ? new InferenceError(
|
|
932
|
+
"The local-model worker was shut down by disposeLlamaModels while a call was still running."
|
|
933
|
+
) : new LlamaWorkerCrash(this, code, signal)
|
|
934
|
+
);
|
|
935
|
+
}
|
|
936
|
+
rejectAll(reason) {
|
|
937
|
+
const pending = [...this.pending.values()];
|
|
938
|
+
this.pending.clear();
|
|
939
|
+
this.idle();
|
|
940
|
+
for (const p of pending) p.reject(reason);
|
|
941
|
+
}
|
|
942
|
+
collect(chunk) {
|
|
943
|
+
const lines = (this.partialLine + chunk).split(/\r?\n/);
|
|
944
|
+
this.partialLine = lines.pop() ?? "";
|
|
945
|
+
this.stderrTail.push(...lines);
|
|
946
|
+
this.stderrTail.splice(0, Math.max(0, this.stderrTail.length - STDERR_TAIL_LINES));
|
|
947
|
+
}
|
|
948
|
+
/** `exit` can arrive before the last of stderr; give it a moment. */
|
|
949
|
+
drainStderr() {
|
|
950
|
+
const stderr = this.child.stderr;
|
|
951
|
+
if (stderr.readableEnded || stderr.destroyed) return Promise.resolve();
|
|
952
|
+
return new Promise((resolve) => {
|
|
953
|
+
const timer = setTimeout(resolve, STDERR_DRAIN_MS);
|
|
954
|
+
timer.unref();
|
|
955
|
+
const done = () => {
|
|
956
|
+
clearTimeout(timer);
|
|
957
|
+
resolve();
|
|
958
|
+
};
|
|
959
|
+
stderr.once("end", done);
|
|
960
|
+
stderr.once("close", done);
|
|
961
|
+
});
|
|
962
|
+
}
|
|
963
|
+
/**
|
|
964
|
+
* Hold the parent open only while it is waiting on the worker. An idle
|
|
965
|
+
* worker must not keep a consumer's process alive after its work is done —
|
|
966
|
+
* and when that process exits, the worker sees `disconnect` and exits too.
|
|
967
|
+
*/
|
|
968
|
+
busy() {
|
|
969
|
+
this.child.ref();
|
|
970
|
+
this.child.channel?.ref();
|
|
971
|
+
this.child.stderr?.ref?.();
|
|
972
|
+
}
|
|
973
|
+
idle() {
|
|
974
|
+
this.child.unref();
|
|
975
|
+
this.child.channel?.unref();
|
|
976
|
+
this.child.stderr?.unref?.();
|
|
977
|
+
}
|
|
978
|
+
};
|
|
979
|
+
function rebuildError(error) {
|
|
980
|
+
if (error.name === "InferenceError") return new InferenceError(error.message);
|
|
981
|
+
const rebuilt = new Error(error.message);
|
|
982
|
+
rebuilt.name = error.name;
|
|
983
|
+
return rebuilt;
|
|
984
|
+
}
|
|
985
|
+
function backendName(gpu) {
|
|
986
|
+
switch (gpu) {
|
|
987
|
+
case "cuda":
|
|
988
|
+
return "CUDA";
|
|
989
|
+
case "vulkan":
|
|
990
|
+
return "Vulkan";
|
|
991
|
+
case "metal":
|
|
992
|
+
return "Metal";
|
|
993
|
+
default:
|
|
994
|
+
return "CPU";
|
|
995
|
+
}
|
|
996
|
+
}
|
|
997
|
+
var GGML_ABORT_LINE = /\.(?:cu|cpp|cc|c|h|m|mm):\d+: \S/;
|
|
998
|
+
function describeExit(crash) {
|
|
999
|
+
const how = crash.signal ? `signal ${crash.signal}` : `exit code ${String(crash.code)}`;
|
|
1000
|
+
const lines = crash.host.stderr.map((l) => l.trim()).filter((l) => l !== "");
|
|
1001
|
+
const line = [...lines].reverse().find((l) => GGML_ABORT_LINE.test(l)) ?? lines.find((l) => /error|abort|assert|fatal/i.test(l)) ?? lines[lines.length - 1];
|
|
1002
|
+
return line ? `${how}: ${line}` : how;
|
|
1003
|
+
}
|
|
1004
|
+
var failedBackends = /* @__PURE__ */ new Map();
|
|
1005
|
+
var Slot = class {
|
|
1006
|
+
constructor(gpu, source, entry) {
|
|
1007
|
+
this.gpu = gpu;
|
|
1008
|
+
this.source = source;
|
|
1009
|
+
this.entry = entry;
|
|
1010
|
+
}
|
|
1011
|
+
gpu;
|
|
1012
|
+
source;
|
|
1013
|
+
entry;
|
|
1014
|
+
host;
|
|
1015
|
+
starting;
|
|
1016
|
+
/** The crash a new worker is replacing, so its start can say what changed. */
|
|
1017
|
+
switchedFrom;
|
|
1018
|
+
get pid() {
|
|
1019
|
+
return this.host?.pid;
|
|
1020
|
+
}
|
|
1021
|
+
/**
|
|
1022
|
+
* Run `fn` against a live worker, retrying on the next backend when the
|
|
1023
|
+
* worker crashes under it. Ordinary errors pass through untouched.
|
|
1024
|
+
*/
|
|
1025
|
+
async run(fn) {
|
|
1026
|
+
for (; ; ) {
|
|
1027
|
+
try {
|
|
1028
|
+
return await fn(await this.acquire());
|
|
1029
|
+
} catch (e) {
|
|
1030
|
+
if (!(e instanceof LlamaWorkerCrash)) throw e;
|
|
1031
|
+
const verdict = e.host.verdict ??= this.decide(e);
|
|
1032
|
+
if (verdict instanceof Error) throw verdict;
|
|
1033
|
+
}
|
|
1034
|
+
}
|
|
1035
|
+
}
|
|
1036
|
+
async shutdown() {
|
|
1037
|
+
const host = this.host;
|
|
1038
|
+
this.host = void 0;
|
|
1039
|
+
this.starting = void 0;
|
|
1040
|
+
await host?.shutdown();
|
|
1041
|
+
}
|
|
1042
|
+
acquire() {
|
|
1043
|
+
if (this.starting && !this.host?.dead) return this.starting;
|
|
1044
|
+
const starting = (async () => {
|
|
1045
|
+
const source = await backendSource();
|
|
1046
|
+
const host = new WorkerHost(this.entry);
|
|
1047
|
+
this.host = host;
|
|
1048
|
+
const { gpu } = await host.request({
|
|
1049
|
+
op: "init",
|
|
1050
|
+
moduleUrl: source.moduleUrl,
|
|
1051
|
+
options: this.options()
|
|
1052
|
+
});
|
|
1053
|
+
host.gpu = gpu;
|
|
1054
|
+
if (this.switchedFrom) {
|
|
1055
|
+
warnSwitch(this.switchedFrom, gpu);
|
|
1056
|
+
this.switchedFrom = void 0;
|
|
1057
|
+
}
|
|
1058
|
+
return host;
|
|
1059
|
+
})();
|
|
1060
|
+
this.starting = starting;
|
|
1061
|
+
starting.catch((e) => {
|
|
1062
|
+
if (e instanceof LlamaWorkerCrash || this.starting !== starting) return;
|
|
1063
|
+
const host = this.host;
|
|
1064
|
+
this.starting = void 0;
|
|
1065
|
+
this.host = void 0;
|
|
1066
|
+
void host?.shutdown();
|
|
1067
|
+
});
|
|
1068
|
+
return starting;
|
|
1069
|
+
}
|
|
1070
|
+
options() {
|
|
1071
|
+
if (this.gpu !== "auto") return { gpu: this.gpu };
|
|
1072
|
+
const exclude = [...failedBackends.keys()];
|
|
1073
|
+
return exclude.length > 0 ? { gpu: { type: "auto", exclude }, build: "never" } : { gpu: "auto" };
|
|
1074
|
+
}
|
|
1075
|
+
decide(crash) {
|
|
1076
|
+
if (this.host === crash.host) {
|
|
1077
|
+
this.host = void 0;
|
|
1078
|
+
this.starting = void 0;
|
|
1079
|
+
}
|
|
1080
|
+
if (crash.blameless) return "retry";
|
|
1081
|
+
const gpu = crash.host.gpu ?? crash.host.trying;
|
|
1082
|
+
const exit = describeExit(crash);
|
|
1083
|
+
if (gpu === void 0) {
|
|
1084
|
+
return new InferenceError(
|
|
1085
|
+
`llama.cpp's local-model worker crashed before it chose a backend (${exit}). The request was not answered.`
|
|
1086
|
+
);
|
|
1087
|
+
}
|
|
1088
|
+
const name = backendName(gpu);
|
|
1089
|
+
if (this.gpu !== "auto") {
|
|
1090
|
+
return new InferenceError(
|
|
1091
|
+
`llama.cpp's ${name} backend crashed the local-model worker (${exit}). ${name} was chosen explicitly (${this.source ?? "llamaCpp.gpu"}), so the library did not switch backends. Choose another \u2014 ${alternativesTo(gpu)} \u2014 or unset it to let the library fall back on its own.`
|
|
1092
|
+
);
|
|
1093
|
+
}
|
|
1094
|
+
if (gpu === false) {
|
|
1095
|
+
const all = [...failedBackends].map(([g, e]) => `${backendName(g)} (${e})`);
|
|
1096
|
+
return new InferenceError(
|
|
1097
|
+
`llama.cpp crashed the local-model worker on every backend this machine offers \u2014 ${[...all, `CPU (${exit})`].join(", ")}. The request was not answered.`
|
|
1098
|
+
);
|
|
1099
|
+
}
|
|
1100
|
+
failedBackends.set(gpu, exit);
|
|
1101
|
+
this.switchedFrom = { gpu, exit };
|
|
1102
|
+
return "retry";
|
|
1103
|
+
}
|
|
1104
|
+
};
|
|
1105
|
+
function alternativesTo(gpu) {
|
|
1106
|
+
switch (gpu) {
|
|
1107
|
+
case "cuda":
|
|
1108
|
+
return "NODE_LLAMA_CPP_GPU=vulkan, or false for the CPU";
|
|
1109
|
+
case "vulkan":
|
|
1110
|
+
return "NODE_LLAMA_CPP_GPU=cuda, or false for the CPU";
|
|
1111
|
+
case "metal":
|
|
1112
|
+
return "NODE_LLAMA_CPP_GPU=false for the CPU";
|
|
1113
|
+
default:
|
|
1114
|
+
return "NODE_LLAMA_CPP_GPU=auto for a GPU backend";
|
|
1115
|
+
}
|
|
1116
|
+
}
|
|
1117
|
+
function warnSwitch(from, to) {
|
|
1118
|
+
const crashed = `inference: llama.cpp's ${backendName(from.gpu)} backend crashed the local-model worker (${from.exit}).`;
|
|
1119
|
+
if (to === false) {
|
|
1120
|
+
console.warn(
|
|
1121
|
+
`${crashed} Retrying on the CPU, which is much slower \u2014 expect minutes per call \u2014 and staying there for the rest of this process. Pin a backend with NODE_LLAMA_CPP_GPU or llamaCpp.gpu to fail fast instead.`
|
|
1122
|
+
);
|
|
1123
|
+
return;
|
|
1124
|
+
}
|
|
1125
|
+
const name = backendName(to);
|
|
1126
|
+
console.warn(
|
|
1127
|
+
`${crashed} Retrying on ${name}, and staying on ${name} for the rest of this process. Set NODE_LLAMA_CPP_GPU=${to} (or llamaCpp.gpu: "${to}") to start there.`
|
|
1128
|
+
);
|
|
1129
|
+
}
|
|
1130
|
+
function requestedGpu(option, env = process.env) {
|
|
1131
|
+
if (option !== void 0) {
|
|
1132
|
+
return option === "auto" ? { gpu: "auto" } : { gpu: option, source: `llamaCpp.gpu: ${JSON.stringify(option)}` };
|
|
1133
|
+
}
|
|
1134
|
+
const raw = env["NODE_LLAMA_CPP_GPU"];
|
|
1135
|
+
if (raw == null || raw === "" || raw === "auto") return { gpu: "auto" };
|
|
1136
|
+
if (GPU_OFF_VALUES.includes(raw)) {
|
|
1137
|
+
return { gpu: false, source: `NODE_LLAMA_CPP_GPU=${raw}` };
|
|
1138
|
+
}
|
|
1139
|
+
if (raw === "cuda" || raw === "vulkan" || raw === "metal") {
|
|
1140
|
+
return { gpu: raw, source: `NODE_LLAMA_CPP_GPU=${raw}` };
|
|
1141
|
+
}
|
|
1142
|
+
return { gpu: "auto" };
|
|
1143
|
+
}
|
|
1144
|
+
var slots = /* @__PURE__ */ new Map();
|
|
1145
|
+
function slotFor(gpu, source, entry) {
|
|
1146
|
+
const key = JSON.stringify([gpu, source ?? null]);
|
|
1147
|
+
let slot = slots.get(key);
|
|
1148
|
+
if (!slot) {
|
|
1149
|
+
slot = new Slot(gpu, source, entry);
|
|
1150
|
+
slots.set(key, slot);
|
|
1151
|
+
}
|
|
1152
|
+
return slot;
|
|
1153
|
+
}
|
|
1154
|
+
var sourceOverride;
|
|
1155
|
+
var defaultSource;
|
|
1156
|
+
function backendSource() {
|
|
1157
|
+
if (sourceOverride) return Promise.resolve(sourceOverride);
|
|
1158
|
+
return defaultSource ??= nodeLlamaCppSource().catch((e) => {
|
|
1159
|
+
defaultSource = void 0;
|
|
1160
|
+
throw e;
|
|
1161
|
+
});
|
|
1162
|
+
}
|
|
1163
|
+
async function nodeLlamaCppSource() {
|
|
1164
|
+
let mod;
|
|
1165
|
+
let moduleUrl = "node-llama-cpp";
|
|
1166
|
+
try {
|
|
1167
|
+
mod = await import("node-llama-cpp");
|
|
1168
|
+
} catch (e) {
|
|
1169
|
+
if (!isModuleNotFound(e)) {
|
|
1170
|
+
throw new InferenceError(
|
|
1171
|
+
`node-llama-cpp is installed but failed to load (${e instanceof Error ? e.message : String(e)}). This is the copy resolved from your own node_modules, so reinstalling it here will not help \u2014 check the Node version and the platform build.`
|
|
1172
|
+
);
|
|
1173
|
+
}
|
|
1174
|
+
mod = await importNodeLlamaCpp();
|
|
1175
|
+
moduleUrl = nodeLlamaCppShimUrl();
|
|
1176
|
+
}
|
|
1177
|
+
return {
|
|
1178
|
+
moduleUrl,
|
|
1179
|
+
// `directory` is this library's own, not node-llama-cpp's global default —
|
|
1180
|
+
// owning it is what makes `clearLlamaModels` safe.
|
|
1181
|
+
resolveModelFile: (uri, directory) => mod.resolveModelFile(uri, { directory })
|
|
1182
|
+
};
|
|
1183
|
+
}
|
|
1184
|
+
function workerEntry() {
|
|
1185
|
+
for (const name of ["llama-worker.js", "llama-worker.ts"]) {
|
|
1186
|
+
const path = fileURLToPath(new URL(`./${name}`, import.meta.url));
|
|
1187
|
+
if (existsSync3(path)) return path;
|
|
1188
|
+
}
|
|
1189
|
+
return void 0;
|
|
1190
|
+
}
|
|
1191
|
+
var WorkerModelProxy = class {
|
|
1192
|
+
constructor(slot, path) {
|
|
1193
|
+
this.slot = slot;
|
|
1194
|
+
this.path = path;
|
|
1195
|
+
}
|
|
1196
|
+
slot;
|
|
1197
|
+
path;
|
|
1198
|
+
trainContextSize;
|
|
1199
|
+
ids = /* @__PURE__ */ new Map();
|
|
1200
|
+
async open() {
|
|
1201
|
+
const loaded = await this.slot.run((host) => this.on(host));
|
|
1202
|
+
this.trainContextSize = loaded.trainContextSize;
|
|
1203
|
+
}
|
|
1204
|
+
/** This model's id in `host`, loading it there first if a fallback moved us. */
|
|
1205
|
+
on(host) {
|
|
1206
|
+
for (const known of this.ids.keys()) if (known.dead) this.ids.delete(known);
|
|
1207
|
+
let loaded = this.ids.get(host);
|
|
1208
|
+
if (!loaded) {
|
|
1209
|
+
loaded = host.request({ op: "loadModel", path: this.path });
|
|
1210
|
+
this.ids.set(host, loaded);
|
|
1211
|
+
const settled = loaded;
|
|
1212
|
+
settled.catch(() => {
|
|
1213
|
+
if (this.ids.get(host) === settled) this.ids.delete(host);
|
|
1214
|
+
});
|
|
1215
|
+
}
|
|
1216
|
+
return loaded;
|
|
1217
|
+
}
|
|
1218
|
+
countTokens(text) {
|
|
1219
|
+
return this.slot.run(
|
|
1220
|
+
async (host) => host.request({
|
|
1221
|
+
op: "countTokens",
|
|
1222
|
+
modelId: (await this.on(host)).modelId,
|
|
1223
|
+
text
|
|
1224
|
+
})
|
|
1225
|
+
);
|
|
1226
|
+
}
|
|
1227
|
+
async createSession(systemPrompt, contextSize) {
|
|
1228
|
+
const session = new WorkerSessionProxy(this.slot, this, systemPrompt, contextSize);
|
|
1229
|
+
await session.open();
|
|
1230
|
+
return {
|
|
1231
|
+
contextSize: session.contextSize,
|
|
1232
|
+
prompt: (text, options) => session.prompt(text, options),
|
|
1233
|
+
dispose: () => session.dispose()
|
|
1234
|
+
};
|
|
1235
|
+
}
|
|
1236
|
+
async dispose() {
|
|
1237
|
+
await Promise.all(
|
|
1238
|
+
[...this.ids].map(async ([host, loaded]) => {
|
|
1239
|
+
if (host.dead) return;
|
|
1240
|
+
const { modelId } = await loaded;
|
|
1241
|
+
await host.request({ op: "disposeModel", modelId });
|
|
1242
|
+
}).map((p) => p.catch(() => void 0))
|
|
1243
|
+
);
|
|
1244
|
+
this.ids.clear();
|
|
1245
|
+
}
|
|
1246
|
+
};
|
|
1247
|
+
var WorkerSessionProxy = class {
|
|
1248
|
+
constructor(slot, model, systemPrompt, requestedSize) {
|
|
1249
|
+
this.slot = slot;
|
|
1250
|
+
this.model = model;
|
|
1251
|
+
this.systemPrompt = systemPrompt;
|
|
1252
|
+
this.requestedSize = requestedSize;
|
|
1253
|
+
}
|
|
1254
|
+
slot;
|
|
1255
|
+
model;
|
|
1256
|
+
systemPrompt;
|
|
1257
|
+
requestedSize;
|
|
1258
|
+
contextSize;
|
|
1259
|
+
ids = /* @__PURE__ */ new Map();
|
|
1260
|
+
async open() {
|
|
1261
|
+
await this.slot.run((host) => this.on(host));
|
|
1262
|
+
}
|
|
1263
|
+
/** This session's id in `host`. A retried prompt gets a fresh one there. */
|
|
1264
|
+
on(host) {
|
|
1265
|
+
let id = this.ids.get(host);
|
|
1266
|
+
if (!id) {
|
|
1267
|
+
id = (async () => {
|
|
1268
|
+
const opened = await host.request({
|
|
1269
|
+
op: "createSession",
|
|
1270
|
+
modelId: (await this.model.on(host)).modelId,
|
|
1271
|
+
systemPrompt: this.systemPrompt,
|
|
1272
|
+
...this.requestedSize != null ? { contextSize: this.requestedSize } : {}
|
|
1273
|
+
});
|
|
1274
|
+
this.contextSize = opened.contextSize;
|
|
1275
|
+
return opened.sessionId;
|
|
1276
|
+
})();
|
|
1277
|
+
this.ids.set(host, id);
|
|
1278
|
+
const settled = id;
|
|
1279
|
+
settled.catch(() => {
|
|
1280
|
+
if (this.ids.get(host) === settled) this.ids.delete(host);
|
|
1281
|
+
});
|
|
1282
|
+
}
|
|
1283
|
+
return id;
|
|
1284
|
+
}
|
|
1285
|
+
prompt(text, options) {
|
|
1286
|
+
return this.slot.run(
|
|
1287
|
+
async (host) => host.request({
|
|
1288
|
+
op: "prompt",
|
|
1289
|
+
sessionId: await this.on(host),
|
|
1290
|
+
text,
|
|
1291
|
+
options
|
|
1292
|
+
})
|
|
1293
|
+
);
|
|
1294
|
+
}
|
|
1295
|
+
async dispose() {
|
|
1296
|
+
await Promise.all(
|
|
1297
|
+
[...this.ids].map(async ([host, id]) => {
|
|
1298
|
+
if (host.dead) return;
|
|
1299
|
+
await host.request({ op: "disposeSession", sessionId: await id });
|
|
1300
|
+
}).map((p) => p.catch(() => void 0))
|
|
1301
|
+
);
|
|
1302
|
+
this.ids.clear();
|
|
1303
|
+
}
|
|
1304
|
+
};
|
|
1305
|
+
var warnedInProcess = false;
|
|
1306
|
+
function inProcessRuntime(gpu) {
|
|
1307
|
+
if (!warnedInProcess) {
|
|
1308
|
+
warnedInProcess = true;
|
|
1309
|
+
console.warn(
|
|
1310
|
+
`inference: the local-model worker (llama-worker.js) is missing beside this library \u2014 was it bundled? Running llama.cpp in-process instead, so a native crash in llama.cpp will end this process.`
|
|
1311
|
+
);
|
|
1312
|
+
}
|
|
1313
|
+
let opened;
|
|
1314
|
+
const backend = () => opened ??= backendSource().then(
|
|
1315
|
+
async (source) => openBackend(await import(source.moduleUrl), { gpu }, { trying: () => void 0 })
|
|
1316
|
+
).catch((e) => {
|
|
1317
|
+
opened = void 0;
|
|
1318
|
+
throw e;
|
|
1319
|
+
});
|
|
1320
|
+
return {
|
|
1321
|
+
resolveModelFile: (uri, directory) => backendSource().then((source) => source.resolveModelFile(uri, directory)),
|
|
1322
|
+
loadModel: (path) => backend().then((b) => b.loadModel(path)),
|
|
1323
|
+
getMemoryBudgetBytes: () => backend().then((b) => b.memoryBudget())
|
|
1324
|
+
};
|
|
1325
|
+
}
|
|
1326
|
+
function createWorkerRuntime(options = {}) {
|
|
1327
|
+
const { gpu, source } = requestedGpu(options.gpu);
|
|
1328
|
+
const entry = workerEntry();
|
|
1329
|
+
if (!entry) return inProcessRuntime(gpu);
|
|
1330
|
+
const slot = slotFor(gpu, source, entry);
|
|
1331
|
+
return {
|
|
1332
|
+
resolveModelFile: (uri, directory) => backendSource().then((s) => s.resolveModelFile(uri, directory)),
|
|
1333
|
+
async loadModel(path) {
|
|
1334
|
+
const model = new WorkerModelProxy(slot, path);
|
|
1335
|
+
await model.open();
|
|
1336
|
+
return {
|
|
1337
|
+
trainContextSize: model.trainContextSize,
|
|
1338
|
+
countTokens: (text) => model.countTokens(text),
|
|
1339
|
+
createSession: (systemPrompt, contextSize) => model.createSession(systemPrompt, contextSize),
|
|
1340
|
+
dispose: () => model.dispose()
|
|
1341
|
+
};
|
|
1342
|
+
},
|
|
1343
|
+
getMemoryBudgetBytes: () => slot.run((host) => host.request({ op: "memoryBudget" }))
|
|
1344
|
+
};
|
|
1345
|
+
}
|
|
1346
|
+
async function shutdownLlamaWorkers() {
|
|
1347
|
+
await Promise.all([...slots.values()].map((slot) => slot.shutdown()));
|
|
1348
|
+
}
|
|
1349
|
+
|
|
789
1350
|
// src/providers/llama-cpp.ts
|
|
790
1351
|
var loadedModels = /* @__PURE__ */ new Map();
|
|
791
1352
|
async function disposeLlamaModels() {
|
|
@@ -794,7 +1355,11 @@ async function disposeLlamaModels() {
|
|
|
794
1355
|
await Promise.all(
|
|
795
1356
|
pending.map((p) => p.then((m) => m.dispose()).catch(() => void 0))
|
|
796
1357
|
);
|
|
1358
|
+
await shutdownLlamaWorkers();
|
|
797
1359
|
}
|
|
1360
|
+
var DEFAULT_CONTEXT_SIZE = 8192;
|
|
1361
|
+
var CHAT_TEMPLATE_OVERHEAD_TOKENS = 512;
|
|
1362
|
+
var DEFAULT_RESPONSE_RESERVE_TOKENS = 2048;
|
|
798
1363
|
var LlamaCppProvider = class {
|
|
799
1364
|
constructor(model, options = {}) {
|
|
800
1365
|
this.model = model;
|
|
@@ -803,8 +1368,19 @@ var LlamaCppProvider = class {
|
|
|
803
1368
|
`llama-cpp model "${model}" is a selector. Constructing a provider directly needs a concrete model (e.g. "${aliasForTier("balanced")}") \u2014 use makeProviderAsync to resolve a selector against this machine.`
|
|
804
1369
|
);
|
|
805
1370
|
}
|
|
1371
|
+
if (options.contextSize !== void 0 && !(Number.isInteger(options.contextSize) && options.contextSize > 0)) {
|
|
1372
|
+
throw new InferenceError(
|
|
1373
|
+
`llamaCpp.contextSize must be a positive integer number of tokens, got ${String(options.contextSize)}.`
|
|
1374
|
+
);
|
|
1375
|
+
}
|
|
1376
|
+
if (options.gpu !== void 0 && !isLlamaGpu(options.gpu)) {
|
|
1377
|
+
throw new InferenceError(
|
|
1378
|
+
`llamaCpp.gpu must be "auto", "cuda", "vulkan", "metal" or false, got ${JSON.stringify(options.gpu) ?? String(options.gpu)}.`
|
|
1379
|
+
);
|
|
1380
|
+
}
|
|
1381
|
+
this.contextSize = options.contextSize;
|
|
806
1382
|
this.uri = resolveLlamaModelRef(model);
|
|
807
|
-
this.runtime = options.runtime ?? defaultLlamaRuntime();
|
|
1383
|
+
this.runtime = options.runtime ?? defaultLlamaRuntime(options.gpu !== void 0 ? { gpu: options.gpu } : {});
|
|
808
1384
|
this.thoughtTokens = options.thoughtTokens ?? 0;
|
|
809
1385
|
this.maxTokens = options.maxTokens;
|
|
810
1386
|
this.modelsDirectory = options.modelsDirectory ?? defaultLlamaModelsDirectory();
|
|
@@ -815,6 +1391,7 @@ var LlamaCppProvider = class {
|
|
|
815
1391
|
runtime;
|
|
816
1392
|
thoughtTokens;
|
|
817
1393
|
maxTokens;
|
|
1394
|
+
contextSize;
|
|
818
1395
|
modelsDirectory;
|
|
819
1396
|
/**
|
|
820
1397
|
* Loaded-model key: the same URI in two directories is two different files.
|
|
@@ -831,14 +1408,24 @@ var LlamaCppProvider = class {
|
|
|
831
1408
|
}
|
|
832
1409
|
async completeJSON(req) {
|
|
833
1410
|
const model = await this.load();
|
|
834
|
-
const
|
|
1411
|
+
const systemPrompt = systemPromptFor(req);
|
|
1412
|
+
const plan = await this.contextFor(model, systemPrompt, req.user);
|
|
1413
|
+
const session = await model.createSession(systemPrompt, plan.contextSize);
|
|
1414
|
+
const contextSize = session.contextSize ?? plan.contextSize;
|
|
1415
|
+
const implicitMaxTokens = this.maxTokens == null && plan.promptTokens != null ? contextSize - plan.promptTokens : void 0;
|
|
1416
|
+
const maxTokens = this.maxTokens ?? implicitMaxTokens;
|
|
835
1417
|
try {
|
|
836
1418
|
const result = await session.prompt(req.user, {
|
|
837
1419
|
schema: req.schema,
|
|
838
1420
|
temperature: req.temperature,
|
|
839
1421
|
thoughtTokens: this.thoughtTokens,
|
|
840
|
-
...
|
|
1422
|
+
...maxTokens != null ? { maxTokens } : {}
|
|
841
1423
|
});
|
|
1424
|
+
if (result.stopReason === "maxTokens" && implicitMaxTokens != null) {
|
|
1425
|
+
throw new Error(
|
|
1426
|
+
`llama-cpp generation filled the ${contextSize}-token context before completing the JSON \u2014 raise llamaCpp.contextSize, or set llamaCpp.maxTokens to bound the response.`
|
|
1427
|
+
);
|
|
1428
|
+
}
|
|
842
1429
|
if (result.stopReason === "maxTokens") {
|
|
843
1430
|
throw new Error(
|
|
844
1431
|
`llama-cpp generation hit the token limit before completing the JSON${this.maxTokens != null ? ` (maxTokens: ${this.maxTokens})` : ""} \u2014 raise llamaCpp.maxTokens, or shorten the prompt if the context is full.`
|
|
@@ -849,6 +1436,39 @@ var LlamaCppProvider = class {
|
|
|
849
1436
|
await session.dispose().catch(() => void 0);
|
|
850
1437
|
}
|
|
851
1438
|
}
|
|
1439
|
+
/**
|
|
1440
|
+
* The context this call needs: both prompts in the model's own tokens, the
|
|
1441
|
+
* chat template's overhead, and room for the response. A prompt that does not
|
|
1442
|
+
* fit is refused here, before anything is created. llama.cpp would otherwise
|
|
1443
|
+
* shift the overflow out of the context and answer a prompt nobody sent.
|
|
1444
|
+
*/
|
|
1445
|
+
async contextFor(model, systemPrompt, user) {
|
|
1446
|
+
const ceiling = model.trainContextSize;
|
|
1447
|
+
const fallback = this.contextSize ?? (ceiling != null ? Math.min(DEFAULT_CONTEXT_SIZE, ceiling) : DEFAULT_CONTEXT_SIZE);
|
|
1448
|
+
if (!model.countTokens) return { contextSize: fallback };
|
|
1449
|
+
const [system, prompt] = await Promise.all([
|
|
1450
|
+
model.countTokens(systemPrompt),
|
|
1451
|
+
model.countTokens(user)
|
|
1452
|
+
]);
|
|
1453
|
+
const response = (this.maxTokens ?? DEFAULT_RESPONSE_RESERVE_TOKENS) + this.thoughtTokens;
|
|
1454
|
+
const promptTokens = system + prompt + CHAT_TEMPLATE_OVERHEAD_TOKENS;
|
|
1455
|
+
const needed = promptTokens + response;
|
|
1456
|
+
const counted = `Counted: system ${system} + user ${prompt} + ${CHAT_TEMPLATE_OVERHEAD_TOKENS} chat-template overhead + ${response} for the response.`;
|
|
1457
|
+
if (this.contextSize != null) {
|
|
1458
|
+
if (needed > this.contextSize) {
|
|
1459
|
+
throw new InferenceError(
|
|
1460
|
+
`llama-cpp prompt needs ${needed} tokens of context, more than llamaCpp.contextSize (${this.contextSize}). ${counted} Raise llamaCpp.contextSize, ${this.maxTokens != null ? "lower" : "set"} llamaCpp.maxTokens, or leave contextSize unset so the context is sized to the prompt.`
|
|
1461
|
+
);
|
|
1462
|
+
}
|
|
1463
|
+
return { contextSize: this.contextSize, promptTokens };
|
|
1464
|
+
}
|
|
1465
|
+
if (ceiling != null && needed > ceiling) {
|
|
1466
|
+
throw new InferenceError(
|
|
1467
|
+
`llama-cpp prompt needs ${needed} tokens of context, more than this model's training context of ${ceiling} tokens. ${counted} Shorten the prompt${this.maxTokens != null ? ", or lower llamaCpp.maxTokens" : ""}.`
|
|
1468
|
+
);
|
|
1469
|
+
}
|
|
1470
|
+
return { contextSize: Math.max(fallback, needed), promptTokens };
|
|
1471
|
+
}
|
|
852
1472
|
load() {
|
|
853
1473
|
const existing = loadedModels.get(this.cacheKey);
|
|
854
1474
|
if (existing) return existing;
|
|
@@ -893,95 +1513,8 @@ function restoreOpenBrace(text) {
|
|
|
893
1513
|
return text;
|
|
894
1514
|
}
|
|
895
1515
|
}
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
const real = () => (
|
|
899
|
-
// Drop a failed init so the next call retries. A GPU that failed to
|
|
900
|
-
// initialise, or a binary still being extracted by a concurrent install,
|
|
901
|
-
// must not poison the runtime for the rest of the process — the same rule
|
|
902
|
-
// `load()` applies to weights.
|
|
903
|
-
runtimePromise ??= loadNodeLlamaCpp().catch((e) => {
|
|
904
|
-
runtimePromise = void 0;
|
|
905
|
-
throw e;
|
|
906
|
-
})
|
|
907
|
-
);
|
|
908
|
-
return {
|
|
909
|
-
resolveModelFile: (uri, directory) => real().then((r) => r.resolveModelFile(uri, directory)),
|
|
910
|
-
loadModel: (path) => real().then((r) => r.loadModel(path)),
|
|
911
|
-
getMemoryBudgetBytes: () => real().then((r) => r.getMemoryBudgetBytes())
|
|
912
|
-
};
|
|
913
|
-
}
|
|
914
|
-
async function loadNodeLlamaCpp() {
|
|
915
|
-
let mod;
|
|
916
|
-
try {
|
|
917
|
-
mod = await import("node-llama-cpp");
|
|
918
|
-
} catch (e) {
|
|
919
|
-
if (!isModuleNotFound(e)) {
|
|
920
|
-
throw new InferenceError(
|
|
921
|
-
`node-llama-cpp is installed but failed to load (${e instanceof Error ? e.message : String(e)}). This is the copy resolved from your own node_modules, so reinstalling it here will not help \u2014 check the Node version and the platform build.`
|
|
922
|
-
);
|
|
923
|
-
}
|
|
924
|
-
mod = await importNodeLlamaCpp();
|
|
925
|
-
}
|
|
926
|
-
const { getLlama, resolveModelFile, LlamaChatSession, TokenMeter } = mod;
|
|
927
|
-
const llama = await getLlama();
|
|
928
|
-
return {
|
|
929
|
-
// `directory` is this library's own, not node-llama-cpp's global default —
|
|
930
|
-
// owning it is what makes `clearLlamaModels` safe.
|
|
931
|
-
resolveModelFile: (uri, directory) => resolveModelFile(uri, { directory }),
|
|
932
|
-
async loadModel(path) {
|
|
933
|
-
const model = await llama.loadModel({ modelPath: path });
|
|
934
|
-
return {
|
|
935
|
-
async createSession(systemPrompt) {
|
|
936
|
-
const context = await model.createContext();
|
|
937
|
-
const sequence = context.getSequence();
|
|
938
|
-
const session = new LlamaChatSession({
|
|
939
|
-
contextSequence: sequence,
|
|
940
|
-
systemPrompt
|
|
941
|
-
});
|
|
942
|
-
return {
|
|
943
|
-
async prompt(text, options) {
|
|
944
|
-
const grammar = await llama.createGrammarForJsonSchema(
|
|
945
|
-
options.schema
|
|
946
|
-
);
|
|
947
|
-
const before = sequence.tokenMeter.getState();
|
|
948
|
-
const result = await session.promptWithMeta(text, {
|
|
949
|
-
grammar,
|
|
950
|
-
temperature: options.temperature,
|
|
951
|
-
budgets: { thoughtTokens: options.thoughtTokens },
|
|
952
|
-
...options.maxTokens != null ? { maxTokens: options.maxTokens } : {}
|
|
953
|
-
});
|
|
954
|
-
const diff = TokenMeter.diff(sequence.tokenMeter, before);
|
|
955
|
-
return {
|
|
956
|
-
text: result.responseText,
|
|
957
|
-
stopReason: result.stopReason,
|
|
958
|
-
usage: {
|
|
959
|
-
inputTokens: diff.usedInputTokens,
|
|
960
|
-
outputTokens: diff.usedOutputTokens
|
|
961
|
-
}
|
|
962
|
-
};
|
|
963
|
-
},
|
|
964
|
-
async dispose() {
|
|
965
|
-
await context.dispose();
|
|
966
|
-
}
|
|
967
|
-
};
|
|
968
|
-
},
|
|
969
|
-
async dispose() {
|
|
970
|
-
await model.dispose();
|
|
971
|
-
}
|
|
972
|
-
};
|
|
973
|
-
},
|
|
974
|
-
async getMemoryBudgetBytes() {
|
|
975
|
-
const { totalmem } = await import("os");
|
|
976
|
-
const ramBudget = totalmem() / 2;
|
|
977
|
-
try {
|
|
978
|
-
const vram = await llama.getVramState();
|
|
979
|
-
return Math.max(vram.free, ramBudget);
|
|
980
|
-
} catch {
|
|
981
|
-
return ramBudget;
|
|
982
|
-
}
|
|
983
|
-
}
|
|
984
|
-
};
|
|
1516
|
+
function defaultLlamaRuntime(options = {}) {
|
|
1517
|
+
return createWorkerRuntime(options);
|
|
985
1518
|
}
|
|
986
1519
|
|
|
987
1520
|
// src/providers/detect.ts
|
|
@@ -1141,7 +1674,7 @@ async function resolveProviderIdentityAsync(spec) {
|
|
|
1141
1674
|
if (provider !== "llama-cpp" || !isLlamaSelector(model)) {
|
|
1142
1675
|
return resolveProviderIdentity(resolved);
|
|
1143
1676
|
}
|
|
1144
|
-
const tier = model === "auto" ? await probeTier(llamaRuntimeFor(spec)) : model;
|
|
1677
|
+
const tier = model === "auto" ? await probeTier(llamaRuntimeFor(spec), spec.llamaCpp?.gpu) : model;
|
|
1145
1678
|
return { provider, model: aliasForTier(tier) };
|
|
1146
1679
|
}
|
|
1147
1680
|
function warnIfDownloadPending(spec, model) {
|
|
@@ -1155,8 +1688,8 @@ function warnIfDownloadPending(spec, model) {
|
|
|
1155
1688
|
function llamaRuntimeFor(spec) {
|
|
1156
1689
|
return spec.llamaRuntime ?? spec.llamaCpp?.runtime;
|
|
1157
1690
|
}
|
|
1158
|
-
async function probeTier(runtime) {
|
|
1159
|
-
const source = runtime ?? defaultLlamaRuntime();
|
|
1691
|
+
async function probeTier(runtime, gpu) {
|
|
1692
|
+
const source = runtime ?? defaultLlamaRuntime(gpu !== void 0 ? { gpu } : {});
|
|
1160
1693
|
return tierForBudget(await source.getMemoryBudgetBytes());
|
|
1161
1694
|
}
|
|
1162
1695
|
function makeProvider(spec) {
|