@hawkeyexl/inference 0.3.2 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/chunk-UFHBIS5K.js +185 -0
- package/dist/chunk-UFHBIS5K.js.map +1 -0
- package/dist/index.d.ts +3 -190
- package/dist/index.js +579 -102
- package/dist/index.js.map +1 -1
- package/dist/llama-cpp-CdVKP63Z.d.ts +308 -0
- package/dist/llama-worker.d.ts +1 -0
- package/dist/llama-worker.js +9 -0
- package/dist/llama-worker.js.map +1 -0
- package/package.json +2 -2
package/dist/index.js
CHANGED
|
@@ -1,3 +1,8 @@
|
|
|
1
|
+
import {
|
|
2
|
+
WORKER_ENV_FLAG,
|
|
3
|
+
openBackend
|
|
4
|
+
} from "./chunk-UFHBIS5K.js";
|
|
5
|
+
|
|
1
6
|
// src/types.ts
|
|
2
7
|
var InferenceError = class extends Error {
|
|
3
8
|
constructor(message) {
|
|
@@ -596,6 +601,11 @@ function isModelPathOrUri(model) {
|
|
|
596
601
|
return /^(hf|huggingface):/i.test(model) || /^https?:\/\//i.test(model) || /^(hf|huggingface)\.co\//i.test(model) || model.endsWith(".gguf");
|
|
597
602
|
}
|
|
598
603
|
|
|
604
|
+
// src/providers/llama-host.ts
|
|
605
|
+
import { fork } from "child_process";
|
|
606
|
+
import { existsSync as existsSync3 } from "fs";
|
|
607
|
+
import { fileURLToPath } from "url";
|
|
608
|
+
|
|
599
609
|
// src/providers/llama-install.ts
|
|
600
610
|
import {
|
|
601
611
|
existsSync as existsSync2,
|
|
@@ -616,6 +626,9 @@ var LOCK_STALE_MS = INSTALL_TIMEOUT_MS + 6e4;
|
|
|
616
626
|
function defaultLlamaRuntimeDirectory(env = process.env) {
|
|
617
627
|
return env["INFERENCE_RUNTIME_DIR"] || join3(homedir2(), ".hawkeyexl-inference", "runtime");
|
|
618
628
|
}
|
|
629
|
+
function nodeLlamaCppShimUrl(directory = defaultLlamaRuntimeDirectory()) {
|
|
630
|
+
return pathToFileURL(join3(directory, SHIM)).href;
|
|
631
|
+
}
|
|
619
632
|
function isModuleNotFound(e) {
|
|
620
633
|
const code = e?.code;
|
|
621
634
|
return code === "ERR_MODULE_NOT_FOUND" || code === "MODULE_NOT_FOUND";
|
|
@@ -786,6 +799,554 @@ function delay(ms) {
|
|
|
786
799
|
return new Promise((resolve) => setTimeout(resolve, ms));
|
|
787
800
|
}
|
|
788
801
|
|
|
802
|
+
// src/providers/llama-host.ts
|
|
803
|
+
var LLAMA_GPUS = ["auto", "cuda", "vulkan", "metal", false];
|
|
804
|
+
function isLlamaGpu(value) {
|
|
805
|
+
return LLAMA_GPUS.includes(value);
|
|
806
|
+
}
|
|
807
|
+
var STDERR_TAIL_LINES = 20;
|
|
808
|
+
var SHUTDOWN_GRACE_MS = 5e3;
|
|
809
|
+
var STDERR_DRAIN_MS = 250;
|
|
810
|
+
var GPU_OFF_VALUES = ["false", "off", "none", "disable", "disabled"];
|
|
811
|
+
var LlamaWorkerCrash = class extends Error {
|
|
812
|
+
constructor(host, code, signal, blameless = false) {
|
|
813
|
+
super("The local-model worker exited.");
|
|
814
|
+
this.host = host;
|
|
815
|
+
this.code = code;
|
|
816
|
+
this.signal = signal;
|
|
817
|
+
this.blameless = blameless;
|
|
818
|
+
this.name = "LlamaWorkerCrash";
|
|
819
|
+
}
|
|
820
|
+
host;
|
|
821
|
+
code;
|
|
822
|
+
signal;
|
|
823
|
+
blameless;
|
|
824
|
+
};
|
|
825
|
+
var WorkerHost = class {
|
|
826
|
+
child;
|
|
827
|
+
/** The backend that initialised, once it has. */
|
|
828
|
+
gpu;
|
|
829
|
+
/** The backend initialising now, so a crash during `init` is attributable. */
|
|
830
|
+
trying;
|
|
831
|
+
dead = false;
|
|
832
|
+
/** How a crash of this host was handled, decided once for every caller. */
|
|
833
|
+
verdict;
|
|
834
|
+
exited;
|
|
835
|
+
closing = false;
|
|
836
|
+
nextId = 1;
|
|
837
|
+
pending = /* @__PURE__ */ new Map();
|
|
838
|
+
stderrTail = [];
|
|
839
|
+
partialLine = "";
|
|
840
|
+
constructor(entry) {
|
|
841
|
+
this.child = fork(entry, [], {
|
|
842
|
+
// stderr is piped rather than inherited so a crash can name its last
|
|
843
|
+
// line; it is still forwarded live, so llama.cpp's output is not lost.
|
|
844
|
+
stdio: ["ignore", "inherit", "pipe", "ipc"],
|
|
845
|
+
serialization: "json",
|
|
846
|
+
// Never the parent's flags (an inspector port, a test runner's loader).
|
|
847
|
+
// From `src/` Node strips the worker's types itself; say nothing about it.
|
|
848
|
+
execArgv: entry.endsWith(".ts") ? ["--disable-warning=ExperimentalWarning"] : [],
|
|
849
|
+
env: { ...process.env, [WORKER_ENV_FLAG]: "1" }
|
|
850
|
+
});
|
|
851
|
+
const stderr = this.child.stderr;
|
|
852
|
+
stderr.setEncoding("utf8");
|
|
853
|
+
stderr.on("data", (chunk) => {
|
|
854
|
+
process.stderr.write(chunk);
|
|
855
|
+
this.collect(chunk);
|
|
856
|
+
});
|
|
857
|
+
this.child.on("message", (message) => this.onMessage(message));
|
|
858
|
+
this.exited = new Promise((resolve) => {
|
|
859
|
+
this.child.once("error", (e) => {
|
|
860
|
+
this.dead = true;
|
|
861
|
+
this.rejectAll(
|
|
862
|
+
new InferenceError(`Could not start the local-model worker (${e.message}).`)
|
|
863
|
+
);
|
|
864
|
+
resolve();
|
|
865
|
+
});
|
|
866
|
+
this.child.once("exit", (code, signal) => {
|
|
867
|
+
this.dead = true;
|
|
868
|
+
void this.drainStderr().then(() => {
|
|
869
|
+
this.onExit(code, signal);
|
|
870
|
+
resolve();
|
|
871
|
+
});
|
|
872
|
+
});
|
|
873
|
+
});
|
|
874
|
+
this.idle();
|
|
875
|
+
}
|
|
876
|
+
get pid() {
|
|
877
|
+
return this.dead ? void 0 : this.child.pid;
|
|
878
|
+
}
|
|
879
|
+
get stderr() {
|
|
880
|
+
return this.partialLine ? [...this.stderrTail, this.partialLine] : this.stderrTail;
|
|
881
|
+
}
|
|
882
|
+
request(request) {
|
|
883
|
+
if (this.dead) {
|
|
884
|
+
return Promise.reject(new LlamaWorkerCrash(this, null, null, true));
|
|
885
|
+
}
|
|
886
|
+
const id = this.nextId++;
|
|
887
|
+
return new Promise((resolve, reject) => {
|
|
888
|
+
this.pending.set(id, {
|
|
889
|
+
resolve,
|
|
890
|
+
reject
|
|
891
|
+
});
|
|
892
|
+
if (this.pending.size === 1) this.busy();
|
|
893
|
+
this.child.send({ ...request, id }, (e) => {
|
|
894
|
+
if (e && this.pending.delete(id)) {
|
|
895
|
+
if (this.pending.size === 0) this.idle();
|
|
896
|
+
reject(new LlamaWorkerCrash(this, null, null, true));
|
|
897
|
+
}
|
|
898
|
+
});
|
|
899
|
+
});
|
|
900
|
+
}
|
|
901
|
+
async shutdown() {
|
|
902
|
+
if (this.dead) return;
|
|
903
|
+
this.closing = true;
|
|
904
|
+
this.busy();
|
|
905
|
+
try {
|
|
906
|
+
this.child.send({ id: this.nextId++, op: "shutdown" });
|
|
907
|
+
} catch {
|
|
908
|
+
}
|
|
909
|
+
const timer = setTimeout(() => this.child.kill(), SHUTDOWN_GRACE_MS);
|
|
910
|
+
await this.exited;
|
|
911
|
+
clearTimeout(timer);
|
|
912
|
+
}
|
|
913
|
+
onMessage(message) {
|
|
914
|
+
if ("event" in message) {
|
|
915
|
+
this.trying = message.gpu;
|
|
916
|
+
return;
|
|
917
|
+
}
|
|
918
|
+
const pending = this.pending.get(message.id);
|
|
919
|
+
if (!pending) return;
|
|
920
|
+
this.pending.delete(message.id);
|
|
921
|
+
if (this.pending.size === 0) this.idle();
|
|
922
|
+
if (message.ok) {
|
|
923
|
+
pending.resolve(message.value);
|
|
924
|
+
} else {
|
|
925
|
+
pending.reject(rebuildError(message.error));
|
|
926
|
+
}
|
|
927
|
+
}
|
|
928
|
+
onExit(code, signal) {
|
|
929
|
+
if (this.pending.size === 0) return;
|
|
930
|
+
this.rejectAll(
|
|
931
|
+
this.closing ? new InferenceError(
|
|
932
|
+
"The local-model worker was shut down by disposeLlamaModels while a call was still running."
|
|
933
|
+
) : new LlamaWorkerCrash(this, code, signal)
|
|
934
|
+
);
|
|
935
|
+
}
|
|
936
|
+
rejectAll(reason) {
|
|
937
|
+
const pending = [...this.pending.values()];
|
|
938
|
+
this.pending.clear();
|
|
939
|
+
this.idle();
|
|
940
|
+
for (const p of pending) p.reject(reason);
|
|
941
|
+
}
|
|
942
|
+
collect(chunk) {
|
|
943
|
+
const lines = (this.partialLine + chunk).split(/\r?\n/);
|
|
944
|
+
this.partialLine = lines.pop() ?? "";
|
|
945
|
+
this.stderrTail.push(...lines);
|
|
946
|
+
this.stderrTail.splice(0, Math.max(0, this.stderrTail.length - STDERR_TAIL_LINES));
|
|
947
|
+
}
|
|
948
|
+
/** `exit` can arrive before the last of stderr; give it a moment. */
|
|
949
|
+
drainStderr() {
|
|
950
|
+
const stderr = this.child.stderr;
|
|
951
|
+
if (stderr.readableEnded || stderr.destroyed) return Promise.resolve();
|
|
952
|
+
return new Promise((resolve) => {
|
|
953
|
+
const timer = setTimeout(resolve, STDERR_DRAIN_MS);
|
|
954
|
+
timer.unref();
|
|
955
|
+
const done = () => {
|
|
956
|
+
clearTimeout(timer);
|
|
957
|
+
resolve();
|
|
958
|
+
};
|
|
959
|
+
stderr.once("end", done);
|
|
960
|
+
stderr.once("close", done);
|
|
961
|
+
});
|
|
962
|
+
}
|
|
963
|
+
/**
|
|
964
|
+
* Hold the parent open only while it is waiting on the worker. An idle
|
|
965
|
+
* worker must not keep a consumer's process alive after its work is done —
|
|
966
|
+
* and when that process exits, the worker sees `disconnect` and exits too.
|
|
967
|
+
*/
|
|
968
|
+
busy() {
|
|
969
|
+
this.child.ref();
|
|
970
|
+
this.child.channel?.ref();
|
|
971
|
+
this.child.stderr?.ref?.();
|
|
972
|
+
}
|
|
973
|
+
idle() {
|
|
974
|
+
this.child.unref();
|
|
975
|
+
this.child.channel?.unref();
|
|
976
|
+
this.child.stderr?.unref?.();
|
|
977
|
+
}
|
|
978
|
+
};
|
|
979
|
+
function rebuildError(error) {
|
|
980
|
+
if (error.name === "InferenceError") return new InferenceError(error.message);
|
|
981
|
+
const rebuilt = new Error(error.message);
|
|
982
|
+
rebuilt.name = error.name;
|
|
983
|
+
return rebuilt;
|
|
984
|
+
}
|
|
985
|
+
function backendName(gpu) {
|
|
986
|
+
switch (gpu) {
|
|
987
|
+
case "cuda":
|
|
988
|
+
return "CUDA";
|
|
989
|
+
case "vulkan":
|
|
990
|
+
return "Vulkan";
|
|
991
|
+
case "metal":
|
|
992
|
+
return "Metal";
|
|
993
|
+
default:
|
|
994
|
+
return "CPU";
|
|
995
|
+
}
|
|
996
|
+
}
|
|
997
|
+
var GGML_ABORT_LINE = /\.(?:cu|cpp|cc|c|h|m|mm):\d+: \S/;
|
|
998
|
+
function describeExit(crash) {
|
|
999
|
+
const how = crash.signal ? `signal ${crash.signal}` : `exit code ${String(crash.code)}`;
|
|
1000
|
+
const lines = crash.host.stderr.map((l) => l.trim()).filter((l) => l !== "");
|
|
1001
|
+
const line = [...lines].reverse().find((l) => GGML_ABORT_LINE.test(l)) ?? lines.find((l) => /error|abort|assert|fatal/i.test(l)) ?? lines[lines.length - 1];
|
|
1002
|
+
return line ? `${how}: ${line}` : how;
|
|
1003
|
+
}
|
|
1004
|
+
var failedBackends = /* @__PURE__ */ new Map();
|
|
1005
|
+
var Slot = class {
|
|
1006
|
+
constructor(gpu, source, entry) {
|
|
1007
|
+
this.gpu = gpu;
|
|
1008
|
+
this.source = source;
|
|
1009
|
+
this.entry = entry;
|
|
1010
|
+
}
|
|
1011
|
+
gpu;
|
|
1012
|
+
source;
|
|
1013
|
+
entry;
|
|
1014
|
+
host;
|
|
1015
|
+
starting;
|
|
1016
|
+
/** The crash a new worker is replacing, so its start can say what changed. */
|
|
1017
|
+
switchedFrom;
|
|
1018
|
+
get pid() {
|
|
1019
|
+
return this.host?.pid;
|
|
1020
|
+
}
|
|
1021
|
+
/**
|
|
1022
|
+
* Run `fn` against a live worker, retrying on the next backend when the
|
|
1023
|
+
* worker crashes under it. Ordinary errors pass through untouched.
|
|
1024
|
+
*/
|
|
1025
|
+
async run(fn) {
|
|
1026
|
+
for (; ; ) {
|
|
1027
|
+
try {
|
|
1028
|
+
return await fn(await this.acquire());
|
|
1029
|
+
} catch (e) {
|
|
1030
|
+
if (!(e instanceof LlamaWorkerCrash)) throw e;
|
|
1031
|
+
const verdict = e.host.verdict ??= this.decide(e);
|
|
1032
|
+
if (verdict instanceof Error) throw verdict;
|
|
1033
|
+
}
|
|
1034
|
+
}
|
|
1035
|
+
}
|
|
1036
|
+
async shutdown() {
|
|
1037
|
+
const host = this.host;
|
|
1038
|
+
this.host = void 0;
|
|
1039
|
+
this.starting = void 0;
|
|
1040
|
+
await host?.shutdown();
|
|
1041
|
+
}
|
|
1042
|
+
acquire() {
|
|
1043
|
+
if (this.starting && !this.host?.dead) return this.starting;
|
|
1044
|
+
const starting = (async () => {
|
|
1045
|
+
const source = await backendSource();
|
|
1046
|
+
const host = new WorkerHost(this.entry);
|
|
1047
|
+
this.host = host;
|
|
1048
|
+
const { gpu } = await host.request({
|
|
1049
|
+
op: "init",
|
|
1050
|
+
moduleUrl: source.moduleUrl,
|
|
1051
|
+
options: this.options()
|
|
1052
|
+
});
|
|
1053
|
+
host.gpu = gpu;
|
|
1054
|
+
if (this.switchedFrom) {
|
|
1055
|
+
warnSwitch(this.switchedFrom, gpu);
|
|
1056
|
+
this.switchedFrom = void 0;
|
|
1057
|
+
}
|
|
1058
|
+
return host;
|
|
1059
|
+
})();
|
|
1060
|
+
this.starting = starting;
|
|
1061
|
+
starting.catch((e) => {
|
|
1062
|
+
if (e instanceof LlamaWorkerCrash || this.starting !== starting) return;
|
|
1063
|
+
const host = this.host;
|
|
1064
|
+
this.starting = void 0;
|
|
1065
|
+
this.host = void 0;
|
|
1066
|
+
void host?.shutdown();
|
|
1067
|
+
});
|
|
1068
|
+
return starting;
|
|
1069
|
+
}
|
|
1070
|
+
options() {
|
|
1071
|
+
if (this.gpu !== "auto") return { gpu: this.gpu };
|
|
1072
|
+
const exclude = [...failedBackends.keys()];
|
|
1073
|
+
return exclude.length > 0 ? { gpu: { type: "auto", exclude }, build: "never" } : { gpu: "auto" };
|
|
1074
|
+
}
|
|
1075
|
+
decide(crash) {
|
|
1076
|
+
if (this.host === crash.host) {
|
|
1077
|
+
this.host = void 0;
|
|
1078
|
+
this.starting = void 0;
|
|
1079
|
+
}
|
|
1080
|
+
if (crash.blameless) return "retry";
|
|
1081
|
+
const gpu = crash.host.gpu ?? crash.host.trying;
|
|
1082
|
+
const exit = describeExit(crash);
|
|
1083
|
+
if (gpu === void 0) {
|
|
1084
|
+
return new InferenceError(
|
|
1085
|
+
`llama.cpp's local-model worker crashed before it chose a backend (${exit}). The request was not answered.`
|
|
1086
|
+
);
|
|
1087
|
+
}
|
|
1088
|
+
const name = backendName(gpu);
|
|
1089
|
+
if (this.gpu !== "auto") {
|
|
1090
|
+
return new InferenceError(
|
|
1091
|
+
`llama.cpp's ${name} backend crashed the local-model worker (${exit}). ${name} was chosen explicitly (${this.source ?? "llamaCpp.gpu"}), so the library did not switch backends. Choose another \u2014 ${alternativesTo(gpu)} \u2014 or unset it to let the library fall back on its own.`
|
|
1092
|
+
);
|
|
1093
|
+
}
|
|
1094
|
+
if (gpu === false) {
|
|
1095
|
+
const all = [...failedBackends].map(([g, e]) => `${backendName(g)} (${e})`);
|
|
1096
|
+
return new InferenceError(
|
|
1097
|
+
`llama.cpp crashed the local-model worker on every backend this machine offers \u2014 ${[...all, `CPU (${exit})`].join(", ")}. The request was not answered.`
|
|
1098
|
+
);
|
|
1099
|
+
}
|
|
1100
|
+
failedBackends.set(gpu, exit);
|
|
1101
|
+
this.switchedFrom = { gpu, exit };
|
|
1102
|
+
return "retry";
|
|
1103
|
+
}
|
|
1104
|
+
};
|
|
1105
|
+
function alternativesTo(gpu) {
|
|
1106
|
+
switch (gpu) {
|
|
1107
|
+
case "cuda":
|
|
1108
|
+
return "NODE_LLAMA_CPP_GPU=vulkan, or false for the CPU";
|
|
1109
|
+
case "vulkan":
|
|
1110
|
+
return "NODE_LLAMA_CPP_GPU=cuda, or false for the CPU";
|
|
1111
|
+
case "metal":
|
|
1112
|
+
return "NODE_LLAMA_CPP_GPU=false for the CPU";
|
|
1113
|
+
default:
|
|
1114
|
+
return "NODE_LLAMA_CPP_GPU=auto for a GPU backend";
|
|
1115
|
+
}
|
|
1116
|
+
}
|
|
1117
|
+
function warnSwitch(from, to) {
|
|
1118
|
+
const crashed = `inference: llama.cpp's ${backendName(from.gpu)} backend crashed the local-model worker (${from.exit}).`;
|
|
1119
|
+
if (to === false) {
|
|
1120
|
+
console.warn(
|
|
1121
|
+
`${crashed} Retrying on the CPU, which is much slower \u2014 expect minutes per call \u2014 and staying there for the rest of this process. Pin a backend with NODE_LLAMA_CPP_GPU or llamaCpp.gpu to fail fast instead.`
|
|
1122
|
+
);
|
|
1123
|
+
return;
|
|
1124
|
+
}
|
|
1125
|
+
const name = backendName(to);
|
|
1126
|
+
console.warn(
|
|
1127
|
+
`${crashed} Retrying on ${name}, and staying on ${name} for the rest of this process. Set NODE_LLAMA_CPP_GPU=${to} (or llamaCpp.gpu: "${to}") to start there.`
|
|
1128
|
+
);
|
|
1129
|
+
}
|
|
1130
|
+
function requestedGpu(option, env = process.env) {
|
|
1131
|
+
if (option !== void 0) {
|
|
1132
|
+
return option === "auto" ? { gpu: "auto" } : { gpu: option, source: `llamaCpp.gpu: ${JSON.stringify(option)}` };
|
|
1133
|
+
}
|
|
1134
|
+
const raw = env["NODE_LLAMA_CPP_GPU"];
|
|
1135
|
+
if (raw == null || raw === "" || raw === "auto") return { gpu: "auto" };
|
|
1136
|
+
if (GPU_OFF_VALUES.includes(raw)) {
|
|
1137
|
+
return { gpu: false, source: `NODE_LLAMA_CPP_GPU=${raw}` };
|
|
1138
|
+
}
|
|
1139
|
+
if (raw === "cuda" || raw === "vulkan" || raw === "metal") {
|
|
1140
|
+
return { gpu: raw, source: `NODE_LLAMA_CPP_GPU=${raw}` };
|
|
1141
|
+
}
|
|
1142
|
+
return { gpu: "auto" };
|
|
1143
|
+
}
|
|
1144
|
+
var slots = /* @__PURE__ */ new Map();
|
|
1145
|
+
function slotFor(gpu, source, entry) {
|
|
1146
|
+
const key = JSON.stringify([gpu, source ?? null]);
|
|
1147
|
+
let slot = slots.get(key);
|
|
1148
|
+
if (!slot) {
|
|
1149
|
+
slot = new Slot(gpu, source, entry);
|
|
1150
|
+
slots.set(key, slot);
|
|
1151
|
+
}
|
|
1152
|
+
return slot;
|
|
1153
|
+
}
|
|
1154
|
+
var sourceOverride;
|
|
1155
|
+
var defaultSource;
|
|
1156
|
+
function backendSource() {
|
|
1157
|
+
if (sourceOverride) return Promise.resolve(sourceOverride);
|
|
1158
|
+
return defaultSource ??= nodeLlamaCppSource().catch((e) => {
|
|
1159
|
+
defaultSource = void 0;
|
|
1160
|
+
throw e;
|
|
1161
|
+
});
|
|
1162
|
+
}
|
|
1163
|
+
async function nodeLlamaCppSource() {
|
|
1164
|
+
let mod;
|
|
1165
|
+
let moduleUrl = "node-llama-cpp";
|
|
1166
|
+
try {
|
|
1167
|
+
mod = await import("node-llama-cpp");
|
|
1168
|
+
} catch (e) {
|
|
1169
|
+
if (!isModuleNotFound(e)) {
|
|
1170
|
+
throw new InferenceError(
|
|
1171
|
+
`node-llama-cpp is installed but failed to load (${e instanceof Error ? e.message : String(e)}). This is the copy resolved from your own node_modules, so reinstalling it here will not help \u2014 check the Node version and the platform build.`
|
|
1172
|
+
);
|
|
1173
|
+
}
|
|
1174
|
+
mod = await importNodeLlamaCpp();
|
|
1175
|
+
moduleUrl = nodeLlamaCppShimUrl();
|
|
1176
|
+
}
|
|
1177
|
+
return {
|
|
1178
|
+
moduleUrl,
|
|
1179
|
+
// `directory` is this library's own, not node-llama-cpp's global default —
|
|
1180
|
+
// owning it is what makes `clearLlamaModels` safe.
|
|
1181
|
+
resolveModelFile: (uri, directory) => mod.resolveModelFile(uri, { directory })
|
|
1182
|
+
};
|
|
1183
|
+
}
|
|
1184
|
+
function workerEntry() {
|
|
1185
|
+
for (const name of ["llama-worker.js", "llama-worker.ts"]) {
|
|
1186
|
+
const path = fileURLToPath(new URL(`./${name}`, import.meta.url));
|
|
1187
|
+
if (existsSync3(path)) return path;
|
|
1188
|
+
}
|
|
1189
|
+
return void 0;
|
|
1190
|
+
}
|
|
1191
|
+
var WorkerModelProxy = class {
|
|
1192
|
+
constructor(slot, path) {
|
|
1193
|
+
this.slot = slot;
|
|
1194
|
+
this.path = path;
|
|
1195
|
+
}
|
|
1196
|
+
slot;
|
|
1197
|
+
path;
|
|
1198
|
+
trainContextSize;
|
|
1199
|
+
ids = /* @__PURE__ */ new Map();
|
|
1200
|
+
async open() {
|
|
1201
|
+
const loaded = await this.slot.run((host) => this.on(host));
|
|
1202
|
+
this.trainContextSize = loaded.trainContextSize;
|
|
1203
|
+
}
|
|
1204
|
+
/** This model's id in `host`, loading it there first if a fallback moved us. */
|
|
1205
|
+
on(host) {
|
|
1206
|
+
for (const known of this.ids.keys()) if (known.dead) this.ids.delete(known);
|
|
1207
|
+
let loaded = this.ids.get(host);
|
|
1208
|
+
if (!loaded) {
|
|
1209
|
+
loaded = host.request({ op: "loadModel", path: this.path });
|
|
1210
|
+
this.ids.set(host, loaded);
|
|
1211
|
+
const settled = loaded;
|
|
1212
|
+
settled.catch(() => {
|
|
1213
|
+
if (this.ids.get(host) === settled) this.ids.delete(host);
|
|
1214
|
+
});
|
|
1215
|
+
}
|
|
1216
|
+
return loaded;
|
|
1217
|
+
}
|
|
1218
|
+
countTokens(text) {
|
|
1219
|
+
return this.slot.run(
|
|
1220
|
+
async (host) => host.request({
|
|
1221
|
+
op: "countTokens",
|
|
1222
|
+
modelId: (await this.on(host)).modelId,
|
|
1223
|
+
text
|
|
1224
|
+
})
|
|
1225
|
+
);
|
|
1226
|
+
}
|
|
1227
|
+
async createSession(systemPrompt, contextSize) {
|
|
1228
|
+
const session = new WorkerSessionProxy(this.slot, this, systemPrompt, contextSize);
|
|
1229
|
+
await session.open();
|
|
1230
|
+
return {
|
|
1231
|
+
contextSize: session.contextSize,
|
|
1232
|
+
prompt: (text, options) => session.prompt(text, options),
|
|
1233
|
+
dispose: () => session.dispose()
|
|
1234
|
+
};
|
|
1235
|
+
}
|
|
1236
|
+
async dispose() {
|
|
1237
|
+
await Promise.all(
|
|
1238
|
+
[...this.ids].map(async ([host, loaded]) => {
|
|
1239
|
+
if (host.dead) return;
|
|
1240
|
+
const { modelId } = await loaded;
|
|
1241
|
+
await host.request({ op: "disposeModel", modelId });
|
|
1242
|
+
}).map((p) => p.catch(() => void 0))
|
|
1243
|
+
);
|
|
1244
|
+
this.ids.clear();
|
|
1245
|
+
}
|
|
1246
|
+
};
|
|
1247
|
+
var WorkerSessionProxy = class {
|
|
1248
|
+
constructor(slot, model, systemPrompt, requestedSize) {
|
|
1249
|
+
this.slot = slot;
|
|
1250
|
+
this.model = model;
|
|
1251
|
+
this.systemPrompt = systemPrompt;
|
|
1252
|
+
this.requestedSize = requestedSize;
|
|
1253
|
+
}
|
|
1254
|
+
slot;
|
|
1255
|
+
model;
|
|
1256
|
+
systemPrompt;
|
|
1257
|
+
requestedSize;
|
|
1258
|
+
contextSize;
|
|
1259
|
+
ids = /* @__PURE__ */ new Map();
|
|
1260
|
+
async open() {
|
|
1261
|
+
await this.slot.run((host) => this.on(host));
|
|
1262
|
+
}
|
|
1263
|
+
/** This session's id in `host`. A retried prompt gets a fresh one there. */
|
|
1264
|
+
on(host) {
|
|
1265
|
+
let id = this.ids.get(host);
|
|
1266
|
+
if (!id) {
|
|
1267
|
+
id = (async () => {
|
|
1268
|
+
const opened = await host.request({
|
|
1269
|
+
op: "createSession",
|
|
1270
|
+
modelId: (await this.model.on(host)).modelId,
|
|
1271
|
+
systemPrompt: this.systemPrompt,
|
|
1272
|
+
...this.requestedSize != null ? { contextSize: this.requestedSize } : {}
|
|
1273
|
+
});
|
|
1274
|
+
this.contextSize = opened.contextSize;
|
|
1275
|
+
return opened.sessionId;
|
|
1276
|
+
})();
|
|
1277
|
+
this.ids.set(host, id);
|
|
1278
|
+
const settled = id;
|
|
1279
|
+
settled.catch(() => {
|
|
1280
|
+
if (this.ids.get(host) === settled) this.ids.delete(host);
|
|
1281
|
+
});
|
|
1282
|
+
}
|
|
1283
|
+
return id;
|
|
1284
|
+
}
|
|
1285
|
+
prompt(text, options) {
|
|
1286
|
+
return this.slot.run(
|
|
1287
|
+
async (host) => host.request({
|
|
1288
|
+
op: "prompt",
|
|
1289
|
+
sessionId: await this.on(host),
|
|
1290
|
+
text,
|
|
1291
|
+
options
|
|
1292
|
+
})
|
|
1293
|
+
);
|
|
1294
|
+
}
|
|
1295
|
+
async dispose() {
|
|
1296
|
+
await Promise.all(
|
|
1297
|
+
[...this.ids].map(async ([host, id]) => {
|
|
1298
|
+
if (host.dead) return;
|
|
1299
|
+
await host.request({ op: "disposeSession", sessionId: await id });
|
|
1300
|
+
}).map((p) => p.catch(() => void 0))
|
|
1301
|
+
);
|
|
1302
|
+
this.ids.clear();
|
|
1303
|
+
}
|
|
1304
|
+
};
|
|
1305
|
+
var warnedInProcess = false;
|
|
1306
|
+
function inProcessRuntime(gpu) {
|
|
1307
|
+
if (!warnedInProcess) {
|
|
1308
|
+
warnedInProcess = true;
|
|
1309
|
+
console.warn(
|
|
1310
|
+
`inference: the local-model worker (llama-worker.js) is missing beside this library \u2014 was it bundled? Running llama.cpp in-process instead, so a native crash in llama.cpp will end this process.`
|
|
1311
|
+
);
|
|
1312
|
+
}
|
|
1313
|
+
let opened;
|
|
1314
|
+
const backend = () => opened ??= backendSource().then(
|
|
1315
|
+
async (source) => openBackend(await import(source.moduleUrl), { gpu }, { trying: () => void 0 })
|
|
1316
|
+
).catch((e) => {
|
|
1317
|
+
opened = void 0;
|
|
1318
|
+
throw e;
|
|
1319
|
+
});
|
|
1320
|
+
return {
|
|
1321
|
+
resolveModelFile: (uri, directory) => backendSource().then((source) => source.resolveModelFile(uri, directory)),
|
|
1322
|
+
loadModel: (path) => backend().then((b) => b.loadModel(path)),
|
|
1323
|
+
getMemoryBudgetBytes: () => backend().then((b) => b.memoryBudget())
|
|
1324
|
+
};
|
|
1325
|
+
}
|
|
1326
|
+
function createWorkerRuntime(options = {}) {
|
|
1327
|
+
const { gpu, source } = requestedGpu(options.gpu);
|
|
1328
|
+
const entry = workerEntry();
|
|
1329
|
+
if (!entry) return inProcessRuntime(gpu);
|
|
1330
|
+
const slot = slotFor(gpu, source, entry);
|
|
1331
|
+
return {
|
|
1332
|
+
resolveModelFile: (uri, directory) => backendSource().then((s) => s.resolveModelFile(uri, directory)),
|
|
1333
|
+
async loadModel(path) {
|
|
1334
|
+
const model = new WorkerModelProxy(slot, path);
|
|
1335
|
+
await model.open();
|
|
1336
|
+
return {
|
|
1337
|
+
trainContextSize: model.trainContextSize,
|
|
1338
|
+
countTokens: (text) => model.countTokens(text),
|
|
1339
|
+
createSession: (systemPrompt, contextSize) => model.createSession(systemPrompt, contextSize),
|
|
1340
|
+
dispose: () => model.dispose()
|
|
1341
|
+
};
|
|
1342
|
+
},
|
|
1343
|
+
getMemoryBudgetBytes: () => slot.run((host) => host.request({ op: "memoryBudget" }))
|
|
1344
|
+
};
|
|
1345
|
+
}
|
|
1346
|
+
async function shutdownLlamaWorkers() {
|
|
1347
|
+
await Promise.all([...slots.values()].map((slot) => slot.shutdown()));
|
|
1348
|
+
}
|
|
1349
|
+
|
|
789
1350
|
// src/providers/llama-cpp.ts
|
|
790
1351
|
var loadedModels = /* @__PURE__ */ new Map();
|
|
791
1352
|
async function disposeLlamaModels() {
|
|
@@ -794,6 +1355,7 @@ async function disposeLlamaModels() {
|
|
|
794
1355
|
await Promise.all(
|
|
795
1356
|
pending.map((p) => p.then((m) => m.dispose()).catch(() => void 0))
|
|
796
1357
|
);
|
|
1358
|
+
await shutdownLlamaWorkers();
|
|
797
1359
|
}
|
|
798
1360
|
var DEFAULT_CONTEXT_SIZE = 8192;
|
|
799
1361
|
var CHAT_TEMPLATE_OVERHEAD_TOKENS = 512;
|
|
@@ -811,9 +1373,14 @@ var LlamaCppProvider = class {
|
|
|
811
1373
|
`llamaCpp.contextSize must be a positive integer number of tokens, got ${String(options.contextSize)}.`
|
|
812
1374
|
);
|
|
813
1375
|
}
|
|
1376
|
+
if (options.gpu !== void 0 && !isLlamaGpu(options.gpu)) {
|
|
1377
|
+
throw new InferenceError(
|
|
1378
|
+
`llamaCpp.gpu must be "auto", "cuda", "vulkan", "metal" or false, got ${JSON.stringify(options.gpu) ?? String(options.gpu)}.`
|
|
1379
|
+
);
|
|
1380
|
+
}
|
|
814
1381
|
this.contextSize = options.contextSize;
|
|
815
1382
|
this.uri = resolveLlamaModelRef(model);
|
|
816
|
-
this.runtime = options.runtime ?? defaultLlamaRuntime();
|
|
1383
|
+
this.runtime = options.runtime ?? defaultLlamaRuntime(options.gpu !== void 0 ? { gpu: options.gpu } : {});
|
|
817
1384
|
this.thoughtTokens = options.thoughtTokens ?? 0;
|
|
818
1385
|
this.maxTokens = options.maxTokens;
|
|
819
1386
|
this.modelsDirectory = options.modelsDirectory ?? defaultLlamaModelsDirectory();
|
|
@@ -842,7 +1409,7 @@ var LlamaCppProvider = class {
|
|
|
842
1409
|
async completeJSON(req) {
|
|
843
1410
|
const model = await this.load();
|
|
844
1411
|
const systemPrompt = systemPromptFor(req);
|
|
845
|
-
const plan = this.contextFor(model, systemPrompt, req.user);
|
|
1412
|
+
const plan = await this.contextFor(model, systemPrompt, req.user);
|
|
846
1413
|
const session = await model.createSession(systemPrompt, plan.contextSize);
|
|
847
1414
|
const contextSize = session.contextSize ?? plan.contextSize;
|
|
848
1415
|
const implicitMaxTokens = this.maxTokens == null && plan.promptTokens != null ? contextSize - plan.promptTokens : void 0;
|
|
@@ -875,12 +1442,14 @@ var LlamaCppProvider = class {
|
|
|
875
1442
|
* fit is refused here, before anything is created. llama.cpp would otherwise
|
|
876
1443
|
* shift the overflow out of the context and answer a prompt nobody sent.
|
|
877
1444
|
*/
|
|
878
|
-
contextFor(model, systemPrompt, user) {
|
|
1445
|
+
async contextFor(model, systemPrompt, user) {
|
|
879
1446
|
const ceiling = model.trainContextSize;
|
|
880
1447
|
const fallback = this.contextSize ?? (ceiling != null ? Math.min(DEFAULT_CONTEXT_SIZE, ceiling) : DEFAULT_CONTEXT_SIZE);
|
|
881
1448
|
if (!model.countTokens) return { contextSize: fallback };
|
|
882
|
-
const system =
|
|
883
|
-
|
|
1449
|
+
const [system, prompt] = await Promise.all([
|
|
1450
|
+
model.countTokens(systemPrompt),
|
|
1451
|
+
model.countTokens(user)
|
|
1452
|
+
]);
|
|
884
1453
|
const response = (this.maxTokens ?? DEFAULT_RESPONSE_RESERVE_TOKENS) + this.thoughtTokens;
|
|
885
1454
|
const promptTokens = system + prompt + CHAT_TEMPLATE_OVERHEAD_TOKENS;
|
|
886
1455
|
const needed = promptTokens + response;
|
|
@@ -944,100 +1513,8 @@ function restoreOpenBrace(text) {
|
|
|
944
1513
|
return text;
|
|
945
1514
|
}
|
|
946
1515
|
}
|
|
947
|
-
|
|
948
|
-
|
|
949
|
-
const real = () => (
|
|
950
|
-
// Drop a failed init so the next call retries. A GPU that failed to
|
|
951
|
-
// initialise, or a binary still being extracted by a concurrent install,
|
|
952
|
-
// must not poison the runtime for the rest of the process — the same rule
|
|
953
|
-
// `load()` applies to weights.
|
|
954
|
-
runtimePromise ??= loadNodeLlamaCpp().catch((e) => {
|
|
955
|
-
runtimePromise = void 0;
|
|
956
|
-
throw e;
|
|
957
|
-
})
|
|
958
|
-
);
|
|
959
|
-
return {
|
|
960
|
-
resolveModelFile: (uri, directory) => real().then((r) => r.resolveModelFile(uri, directory)),
|
|
961
|
-
loadModel: (path) => real().then((r) => r.loadModel(path)),
|
|
962
|
-
getMemoryBudgetBytes: () => real().then((r) => r.getMemoryBudgetBytes())
|
|
963
|
-
};
|
|
964
|
-
}
|
|
965
|
-
async function loadNodeLlamaCpp() {
|
|
966
|
-
let mod;
|
|
967
|
-
try {
|
|
968
|
-
mod = await import("node-llama-cpp");
|
|
969
|
-
} catch (e) {
|
|
970
|
-
if (!isModuleNotFound(e)) {
|
|
971
|
-
throw new InferenceError(
|
|
972
|
-
`node-llama-cpp is installed but failed to load (${e instanceof Error ? e.message : String(e)}). This is the copy resolved from your own node_modules, so reinstalling it here will not help \u2014 check the Node version and the platform build.`
|
|
973
|
-
);
|
|
974
|
-
}
|
|
975
|
-
mod = await importNodeLlamaCpp();
|
|
976
|
-
}
|
|
977
|
-
const { getLlama, resolveModelFile, LlamaChatSession, TokenMeter } = mod;
|
|
978
|
-
const llama = await getLlama();
|
|
979
|
-
return {
|
|
980
|
-
// `directory` is this library's own, not node-llama-cpp's global default —
|
|
981
|
-
// owning it is what makes `clearLlamaModels` safe.
|
|
982
|
-
resolveModelFile: (uri, directory) => resolveModelFile(uri, { directory }),
|
|
983
|
-
async loadModel(path) {
|
|
984
|
-
const model = await llama.loadModel({ modelPath: path });
|
|
985
|
-
return {
|
|
986
|
-
trainContextSize: model.trainContextSize,
|
|
987
|
-
countTokens: (text) => model.tokenize(text).length,
|
|
988
|
-
async createSession(systemPrompt, contextSize) {
|
|
989
|
-
const context = await model.createContext({
|
|
990
|
-
contextSize: contextSize ?? DEFAULT_CONTEXT_SIZE
|
|
991
|
-
});
|
|
992
|
-
const sequence = context.getSequence();
|
|
993
|
-
const session = new LlamaChatSession({
|
|
994
|
-
contextSequence: sequence,
|
|
995
|
-
systemPrompt
|
|
996
|
-
});
|
|
997
|
-
return {
|
|
998
|
-
contextSize: context.contextSize,
|
|
999
|
-
async prompt(text, options) {
|
|
1000
|
-
const grammar = await llama.createGrammarForJsonSchema(
|
|
1001
|
-
options.schema
|
|
1002
|
-
);
|
|
1003
|
-
const before = sequence.tokenMeter.getState();
|
|
1004
|
-
const result = await session.promptWithMeta(text, {
|
|
1005
|
-
grammar,
|
|
1006
|
-
temperature: options.temperature,
|
|
1007
|
-
budgets: { thoughtTokens: options.thoughtTokens },
|
|
1008
|
-
...options.maxTokens != null ? { maxTokens: options.maxTokens } : {}
|
|
1009
|
-
});
|
|
1010
|
-
const diff = TokenMeter.diff(sequence.tokenMeter, before);
|
|
1011
|
-
return {
|
|
1012
|
-
text: result.responseText,
|
|
1013
|
-
stopReason: result.stopReason,
|
|
1014
|
-
usage: {
|
|
1015
|
-
inputTokens: diff.usedInputTokens,
|
|
1016
|
-
outputTokens: diff.usedOutputTokens
|
|
1017
|
-
}
|
|
1018
|
-
};
|
|
1019
|
-
},
|
|
1020
|
-
async dispose() {
|
|
1021
|
-
await context.dispose();
|
|
1022
|
-
}
|
|
1023
|
-
};
|
|
1024
|
-
},
|
|
1025
|
-
async dispose() {
|
|
1026
|
-
await model.dispose();
|
|
1027
|
-
}
|
|
1028
|
-
};
|
|
1029
|
-
},
|
|
1030
|
-
async getMemoryBudgetBytes() {
|
|
1031
|
-
const { totalmem } = await import("os");
|
|
1032
|
-
const ramBudget = totalmem() / 2;
|
|
1033
|
-
try {
|
|
1034
|
-
const vram = await llama.getVramState();
|
|
1035
|
-
return Math.max(vram.free, ramBudget);
|
|
1036
|
-
} catch {
|
|
1037
|
-
return ramBudget;
|
|
1038
|
-
}
|
|
1039
|
-
}
|
|
1040
|
-
};
|
|
1516
|
+
function defaultLlamaRuntime(options = {}) {
|
|
1517
|
+
return createWorkerRuntime(options);
|
|
1041
1518
|
}
|
|
1042
1519
|
|
|
1043
1520
|
// src/providers/detect.ts
|
|
@@ -1197,7 +1674,7 @@ async function resolveProviderIdentityAsync(spec) {
|
|
|
1197
1674
|
if (provider !== "llama-cpp" || !isLlamaSelector(model)) {
|
|
1198
1675
|
return resolveProviderIdentity(resolved);
|
|
1199
1676
|
}
|
|
1200
|
-
const tier = model === "auto" ? await probeTier(llamaRuntimeFor(spec)) : model;
|
|
1677
|
+
const tier = model === "auto" ? await probeTier(llamaRuntimeFor(spec), spec.llamaCpp?.gpu) : model;
|
|
1201
1678
|
return { provider, model: aliasForTier(tier) };
|
|
1202
1679
|
}
|
|
1203
1680
|
function warnIfDownloadPending(spec, model) {
|
|
@@ -1211,8 +1688,8 @@ function warnIfDownloadPending(spec, model) {
|
|
|
1211
1688
|
function llamaRuntimeFor(spec) {
|
|
1212
1689
|
return spec.llamaRuntime ?? spec.llamaCpp?.runtime;
|
|
1213
1690
|
}
|
|
1214
|
-
async function probeTier(runtime) {
|
|
1215
|
-
const source = runtime ?? defaultLlamaRuntime();
|
|
1691
|
+
async function probeTier(runtime, gpu) {
|
|
1692
|
+
const source = runtime ?? defaultLlamaRuntime(gpu !== void 0 ? { gpu } : {});
|
|
1216
1693
|
return tierForBudget(await source.getMemoryBudgetBytes());
|
|
1217
1694
|
}
|
|
1218
1695
|
function makeProvider(spec) {
|