@hawkeyexl/inference 0.3.2 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,3 +1,8 @@
1
+ import {
2
+ WORKER_ENV_FLAG,
3
+ openBackend
4
+ } from "./chunk-UFHBIS5K.js";
5
+
1
6
  // src/types.ts
2
7
  var InferenceError = class extends Error {
3
8
  constructor(message) {
@@ -596,6 +601,11 @@ function isModelPathOrUri(model) {
596
601
  return /^(hf|huggingface):/i.test(model) || /^https?:\/\//i.test(model) || /^(hf|huggingface)\.co\//i.test(model) || model.endsWith(".gguf");
597
602
  }
598
603
 
604
+ // src/providers/llama-host.ts
605
+ import { fork } from "child_process";
606
+ import { existsSync as existsSync3 } from "fs";
607
+ import { fileURLToPath } from "url";
608
+
599
609
  // src/providers/llama-install.ts
600
610
  import {
601
611
  existsSync as existsSync2,
@@ -616,6 +626,9 @@ var LOCK_STALE_MS = INSTALL_TIMEOUT_MS + 6e4;
616
626
  function defaultLlamaRuntimeDirectory(env = process.env) {
617
627
  return env["INFERENCE_RUNTIME_DIR"] || join3(homedir2(), ".hawkeyexl-inference", "runtime");
618
628
  }
629
+ function nodeLlamaCppShimUrl(directory = defaultLlamaRuntimeDirectory()) {
630
+ return pathToFileURL(join3(directory, SHIM)).href;
631
+ }
619
632
  function isModuleNotFound(e) {
620
633
  const code = e?.code;
621
634
  return code === "ERR_MODULE_NOT_FOUND" || code === "MODULE_NOT_FOUND";
@@ -786,6 +799,554 @@ function delay(ms) {
786
799
  return new Promise((resolve) => setTimeout(resolve, ms));
787
800
  }
788
801
 
802
+ // src/providers/llama-host.ts
803
+ var LLAMA_GPUS = ["auto", "cuda", "vulkan", "metal", false];
804
+ function isLlamaGpu(value) {
805
+ return LLAMA_GPUS.includes(value);
806
+ }
807
+ var STDERR_TAIL_LINES = 20;
808
+ var SHUTDOWN_GRACE_MS = 5e3;
809
+ var STDERR_DRAIN_MS = 250;
810
+ var GPU_OFF_VALUES = ["false", "off", "none", "disable", "disabled"];
811
+ var LlamaWorkerCrash = class extends Error {
812
+ constructor(host, code, signal, blameless = false) {
813
+ super("The local-model worker exited.");
814
+ this.host = host;
815
+ this.code = code;
816
+ this.signal = signal;
817
+ this.blameless = blameless;
818
+ this.name = "LlamaWorkerCrash";
819
+ }
820
+ host;
821
+ code;
822
+ signal;
823
+ blameless;
824
+ };
825
+ var WorkerHost = class {
826
+ child;
827
+ /** The backend that initialised, once it has. */
828
+ gpu;
829
+ /** The backend initialising now, so a crash during `init` is attributable. */
830
+ trying;
831
+ dead = false;
832
+ /** How a crash of this host was handled, decided once for every caller. */
833
+ verdict;
834
+ exited;
835
+ closing = false;
836
+ nextId = 1;
837
+ pending = /* @__PURE__ */ new Map();
838
+ stderrTail = [];
839
+ partialLine = "";
840
+ constructor(entry) {
841
+ this.child = fork(entry, [], {
842
+ // stderr is piped rather than inherited so a crash can name its last
843
+ // line; it is still forwarded live, so llama.cpp's output is not lost.
844
+ stdio: ["ignore", "inherit", "pipe", "ipc"],
845
+ serialization: "json",
846
+ // Never the parent's flags (an inspector port, a test runner's loader).
847
+ // From `src/` Node strips the worker's types itself; say nothing about it.
848
+ execArgv: entry.endsWith(".ts") ? ["--disable-warning=ExperimentalWarning"] : [],
849
+ env: { ...process.env, [WORKER_ENV_FLAG]: "1" }
850
+ });
851
+ const stderr = this.child.stderr;
852
+ stderr.setEncoding("utf8");
853
+ stderr.on("data", (chunk) => {
854
+ process.stderr.write(chunk);
855
+ this.collect(chunk);
856
+ });
857
+ this.child.on("message", (message) => this.onMessage(message));
858
+ this.exited = new Promise((resolve) => {
859
+ this.child.once("error", (e) => {
860
+ this.dead = true;
861
+ this.rejectAll(
862
+ new InferenceError(`Could not start the local-model worker (${e.message}).`)
863
+ );
864
+ resolve();
865
+ });
866
+ this.child.once("exit", (code, signal) => {
867
+ this.dead = true;
868
+ void this.drainStderr().then(() => {
869
+ this.onExit(code, signal);
870
+ resolve();
871
+ });
872
+ });
873
+ });
874
+ this.idle();
875
+ }
876
+ get pid() {
877
+ return this.dead ? void 0 : this.child.pid;
878
+ }
879
+ get stderr() {
880
+ return this.partialLine ? [...this.stderrTail, this.partialLine] : this.stderrTail;
881
+ }
882
+ request(request) {
883
+ if (this.dead) {
884
+ return Promise.reject(new LlamaWorkerCrash(this, null, null, true));
885
+ }
886
+ const id = this.nextId++;
887
+ return new Promise((resolve, reject) => {
888
+ this.pending.set(id, {
889
+ resolve,
890
+ reject
891
+ });
892
+ if (this.pending.size === 1) this.busy();
893
+ this.child.send({ ...request, id }, (e) => {
894
+ if (e && this.pending.delete(id)) {
895
+ if (this.pending.size === 0) this.idle();
896
+ reject(new LlamaWorkerCrash(this, null, null, true));
897
+ }
898
+ });
899
+ });
900
+ }
901
+ async shutdown() {
902
+ if (this.dead) return;
903
+ this.closing = true;
904
+ this.busy();
905
+ try {
906
+ this.child.send({ id: this.nextId++, op: "shutdown" });
907
+ } catch {
908
+ }
909
+ const timer = setTimeout(() => this.child.kill(), SHUTDOWN_GRACE_MS);
910
+ await this.exited;
911
+ clearTimeout(timer);
912
+ }
913
+ onMessage(message) {
914
+ if ("event" in message) {
915
+ this.trying = message.gpu;
916
+ return;
917
+ }
918
+ const pending = this.pending.get(message.id);
919
+ if (!pending) return;
920
+ this.pending.delete(message.id);
921
+ if (this.pending.size === 0) this.idle();
922
+ if (message.ok) {
923
+ pending.resolve(message.value);
924
+ } else {
925
+ pending.reject(rebuildError(message.error));
926
+ }
927
+ }
928
+ onExit(code, signal) {
929
+ if (this.pending.size === 0) return;
930
+ this.rejectAll(
931
+ this.closing ? new InferenceError(
932
+ "The local-model worker was shut down by disposeLlamaModels while a call was still running."
933
+ ) : new LlamaWorkerCrash(this, code, signal)
934
+ );
935
+ }
936
+ rejectAll(reason) {
937
+ const pending = [...this.pending.values()];
938
+ this.pending.clear();
939
+ this.idle();
940
+ for (const p of pending) p.reject(reason);
941
+ }
942
+ collect(chunk) {
943
+ const lines = (this.partialLine + chunk).split(/\r?\n/);
944
+ this.partialLine = lines.pop() ?? "";
945
+ this.stderrTail.push(...lines);
946
+ this.stderrTail.splice(0, Math.max(0, this.stderrTail.length - STDERR_TAIL_LINES));
947
+ }
948
+ /** `exit` can arrive before the last of stderr; give it a moment. */
949
+ drainStderr() {
950
+ const stderr = this.child.stderr;
951
+ if (stderr.readableEnded || stderr.destroyed) return Promise.resolve();
952
+ return new Promise((resolve) => {
953
+ const timer = setTimeout(resolve, STDERR_DRAIN_MS);
954
+ timer.unref();
955
+ const done = () => {
956
+ clearTimeout(timer);
957
+ resolve();
958
+ };
959
+ stderr.once("end", done);
960
+ stderr.once("close", done);
961
+ });
962
+ }
963
+ /**
964
+ * Hold the parent open only while it is waiting on the worker. An idle
965
+ * worker must not keep a consumer's process alive after its work is done —
966
+ * and when that process exits, the worker sees `disconnect` and exits too.
967
+ */
968
+ busy() {
969
+ this.child.ref();
970
+ this.child.channel?.ref();
971
+ this.child.stderr?.ref?.();
972
+ }
973
+ idle() {
974
+ this.child.unref();
975
+ this.child.channel?.unref();
976
+ this.child.stderr?.unref?.();
977
+ }
978
+ };
979
+ function rebuildError(error) {
980
+ if (error.name === "InferenceError") return new InferenceError(error.message);
981
+ const rebuilt = new Error(error.message);
982
+ rebuilt.name = error.name;
983
+ return rebuilt;
984
+ }
985
+ function backendName(gpu) {
986
+ switch (gpu) {
987
+ case "cuda":
988
+ return "CUDA";
989
+ case "vulkan":
990
+ return "Vulkan";
991
+ case "metal":
992
+ return "Metal";
993
+ default:
994
+ return "CPU";
995
+ }
996
+ }
997
+ var GGML_ABORT_LINE = /\.(?:cu|cpp|cc|c|h|m|mm):\d+: \S/;
998
+ function describeExit(crash) {
999
+ const how = crash.signal ? `signal ${crash.signal}` : `exit code ${String(crash.code)}`;
1000
+ const lines = crash.host.stderr.map((l) => l.trim()).filter((l) => l !== "");
1001
+ const line = [...lines].reverse().find((l) => GGML_ABORT_LINE.test(l)) ?? lines.find((l) => /error|abort|assert|fatal/i.test(l)) ?? lines[lines.length - 1];
1002
+ return line ? `${how}: ${line}` : how;
1003
+ }
1004
+ var failedBackends = /* @__PURE__ */ new Map();
1005
+ var Slot = class {
1006
+ constructor(gpu, source, entry) {
1007
+ this.gpu = gpu;
1008
+ this.source = source;
1009
+ this.entry = entry;
1010
+ }
1011
+ gpu;
1012
+ source;
1013
+ entry;
1014
+ host;
1015
+ starting;
1016
+ /** The crash a new worker is replacing, so its start can say what changed. */
1017
+ switchedFrom;
1018
+ get pid() {
1019
+ return this.host?.pid;
1020
+ }
1021
+ /**
1022
+ * Run `fn` against a live worker, retrying on the next backend when the
1023
+ * worker crashes under it. Ordinary errors pass through untouched.
1024
+ */
1025
+ async run(fn) {
1026
+ for (; ; ) {
1027
+ try {
1028
+ return await fn(await this.acquire());
1029
+ } catch (e) {
1030
+ if (!(e instanceof LlamaWorkerCrash)) throw e;
1031
+ const verdict = e.host.verdict ??= this.decide(e);
1032
+ if (verdict instanceof Error) throw verdict;
1033
+ }
1034
+ }
1035
+ }
1036
+ async shutdown() {
1037
+ const host = this.host;
1038
+ this.host = void 0;
1039
+ this.starting = void 0;
1040
+ await host?.shutdown();
1041
+ }
1042
+ acquire() {
1043
+ if (this.starting && !this.host?.dead) return this.starting;
1044
+ const starting = (async () => {
1045
+ const source = await backendSource();
1046
+ const host = new WorkerHost(this.entry);
1047
+ this.host = host;
1048
+ const { gpu } = await host.request({
1049
+ op: "init",
1050
+ moduleUrl: source.moduleUrl,
1051
+ options: this.options()
1052
+ });
1053
+ host.gpu = gpu;
1054
+ if (this.switchedFrom) {
1055
+ warnSwitch(this.switchedFrom, gpu);
1056
+ this.switchedFrom = void 0;
1057
+ }
1058
+ return host;
1059
+ })();
1060
+ this.starting = starting;
1061
+ starting.catch((e) => {
1062
+ if (e instanceof LlamaWorkerCrash || this.starting !== starting) return;
1063
+ const host = this.host;
1064
+ this.starting = void 0;
1065
+ this.host = void 0;
1066
+ void host?.shutdown();
1067
+ });
1068
+ return starting;
1069
+ }
1070
+ options() {
1071
+ if (this.gpu !== "auto") return { gpu: this.gpu };
1072
+ const exclude = [...failedBackends.keys()];
1073
+ return exclude.length > 0 ? { gpu: { type: "auto", exclude }, build: "never" } : { gpu: "auto" };
1074
+ }
1075
+ decide(crash) {
1076
+ if (this.host === crash.host) {
1077
+ this.host = void 0;
1078
+ this.starting = void 0;
1079
+ }
1080
+ if (crash.blameless) return "retry";
1081
+ const gpu = crash.host.gpu ?? crash.host.trying;
1082
+ const exit = describeExit(crash);
1083
+ if (gpu === void 0) {
1084
+ return new InferenceError(
1085
+ `llama.cpp's local-model worker crashed before it chose a backend (${exit}). The request was not answered.`
1086
+ );
1087
+ }
1088
+ const name = backendName(gpu);
1089
+ if (this.gpu !== "auto") {
1090
+ return new InferenceError(
1091
+ `llama.cpp's ${name} backend crashed the local-model worker (${exit}). ${name} was chosen explicitly (${this.source ?? "llamaCpp.gpu"}), so the library did not switch backends. Choose another \u2014 ${alternativesTo(gpu)} \u2014 or unset it to let the library fall back on its own.`
1092
+ );
1093
+ }
1094
+ if (gpu === false) {
1095
+ const all = [...failedBackends].map(([g, e]) => `${backendName(g)} (${e})`);
1096
+ return new InferenceError(
1097
+ `llama.cpp crashed the local-model worker on every backend this machine offers \u2014 ${[...all, `CPU (${exit})`].join(", ")}. The request was not answered.`
1098
+ );
1099
+ }
1100
+ failedBackends.set(gpu, exit);
1101
+ this.switchedFrom = { gpu, exit };
1102
+ return "retry";
1103
+ }
1104
+ };
1105
+ function alternativesTo(gpu) {
1106
+ switch (gpu) {
1107
+ case "cuda":
1108
+ return "NODE_LLAMA_CPP_GPU=vulkan, or false for the CPU";
1109
+ case "vulkan":
1110
+ return "NODE_LLAMA_CPP_GPU=cuda, or false for the CPU";
1111
+ case "metal":
1112
+ return "NODE_LLAMA_CPP_GPU=false for the CPU";
1113
+ default:
1114
+ return "NODE_LLAMA_CPP_GPU=auto for a GPU backend";
1115
+ }
1116
+ }
1117
+ function warnSwitch(from, to) {
1118
+ const crashed = `inference: llama.cpp's ${backendName(from.gpu)} backend crashed the local-model worker (${from.exit}).`;
1119
+ if (to === false) {
1120
+ console.warn(
1121
+ `${crashed} Retrying on the CPU, which is much slower \u2014 expect minutes per call \u2014 and staying there for the rest of this process. Pin a backend with NODE_LLAMA_CPP_GPU or llamaCpp.gpu to fail fast instead.`
1122
+ );
1123
+ return;
1124
+ }
1125
+ const name = backendName(to);
1126
+ console.warn(
1127
+ `${crashed} Retrying on ${name}, and staying on ${name} for the rest of this process. Set NODE_LLAMA_CPP_GPU=${to} (or llamaCpp.gpu: "${to}") to start there.`
1128
+ );
1129
+ }
1130
+ function requestedGpu(option, env = process.env) {
1131
+ if (option !== void 0) {
1132
+ return option === "auto" ? { gpu: "auto" } : { gpu: option, source: `llamaCpp.gpu: ${JSON.stringify(option)}` };
1133
+ }
1134
+ const raw = env["NODE_LLAMA_CPP_GPU"];
1135
+ if (raw == null || raw === "" || raw === "auto") return { gpu: "auto" };
1136
+ if (GPU_OFF_VALUES.includes(raw)) {
1137
+ return { gpu: false, source: `NODE_LLAMA_CPP_GPU=${raw}` };
1138
+ }
1139
+ if (raw === "cuda" || raw === "vulkan" || raw === "metal") {
1140
+ return { gpu: raw, source: `NODE_LLAMA_CPP_GPU=${raw}` };
1141
+ }
1142
+ return { gpu: "auto" };
1143
+ }
1144
+ var slots = /* @__PURE__ */ new Map();
1145
+ function slotFor(gpu, source, entry) {
1146
+ const key = JSON.stringify([gpu, source ?? null]);
1147
+ let slot = slots.get(key);
1148
+ if (!slot) {
1149
+ slot = new Slot(gpu, source, entry);
1150
+ slots.set(key, slot);
1151
+ }
1152
+ return slot;
1153
+ }
1154
+ var sourceOverride;
1155
+ var defaultSource;
1156
+ function backendSource() {
1157
+ if (sourceOverride) return Promise.resolve(sourceOverride);
1158
+ return defaultSource ??= nodeLlamaCppSource().catch((e) => {
1159
+ defaultSource = void 0;
1160
+ throw e;
1161
+ });
1162
+ }
1163
+ async function nodeLlamaCppSource() {
1164
+ let mod;
1165
+ let moduleUrl = "node-llama-cpp";
1166
+ try {
1167
+ mod = await import("node-llama-cpp");
1168
+ } catch (e) {
1169
+ if (!isModuleNotFound(e)) {
1170
+ throw new InferenceError(
1171
+ `node-llama-cpp is installed but failed to load (${e instanceof Error ? e.message : String(e)}). This is the copy resolved from your own node_modules, so reinstalling it here will not help \u2014 check the Node version and the platform build.`
1172
+ );
1173
+ }
1174
+ mod = await importNodeLlamaCpp();
1175
+ moduleUrl = nodeLlamaCppShimUrl();
1176
+ }
1177
+ return {
1178
+ moduleUrl,
1179
+ // `directory` is this library's own, not node-llama-cpp's global default —
1180
+ // owning it is what makes `clearLlamaModels` safe.
1181
+ resolveModelFile: (uri, directory) => mod.resolveModelFile(uri, { directory })
1182
+ };
1183
+ }
1184
+ function workerEntry() {
1185
+ for (const name of ["llama-worker.js", "llama-worker.ts"]) {
1186
+ const path = fileURLToPath(new URL(`./${name}`, import.meta.url));
1187
+ if (existsSync3(path)) return path;
1188
+ }
1189
+ return void 0;
1190
+ }
1191
+ var WorkerModelProxy = class {
1192
+ constructor(slot, path) {
1193
+ this.slot = slot;
1194
+ this.path = path;
1195
+ }
1196
+ slot;
1197
+ path;
1198
+ trainContextSize;
1199
+ ids = /* @__PURE__ */ new Map();
1200
+ async open() {
1201
+ const loaded = await this.slot.run((host) => this.on(host));
1202
+ this.trainContextSize = loaded.trainContextSize;
1203
+ }
1204
+ /** This model's id in `host`, loading it there first if a fallback moved us. */
1205
+ on(host) {
1206
+ for (const known of this.ids.keys()) if (known.dead) this.ids.delete(known);
1207
+ let loaded = this.ids.get(host);
1208
+ if (!loaded) {
1209
+ loaded = host.request({ op: "loadModel", path: this.path });
1210
+ this.ids.set(host, loaded);
1211
+ const settled = loaded;
1212
+ settled.catch(() => {
1213
+ if (this.ids.get(host) === settled) this.ids.delete(host);
1214
+ });
1215
+ }
1216
+ return loaded;
1217
+ }
1218
+ countTokens(text) {
1219
+ return this.slot.run(
1220
+ async (host) => host.request({
1221
+ op: "countTokens",
1222
+ modelId: (await this.on(host)).modelId,
1223
+ text
1224
+ })
1225
+ );
1226
+ }
1227
+ async createSession(systemPrompt, contextSize) {
1228
+ const session = new WorkerSessionProxy(this.slot, this, systemPrompt, contextSize);
1229
+ await session.open();
1230
+ return {
1231
+ contextSize: session.contextSize,
1232
+ prompt: (text, options) => session.prompt(text, options),
1233
+ dispose: () => session.dispose()
1234
+ };
1235
+ }
1236
+ async dispose() {
1237
+ await Promise.all(
1238
+ [...this.ids].map(async ([host, loaded]) => {
1239
+ if (host.dead) return;
1240
+ const { modelId } = await loaded;
1241
+ await host.request({ op: "disposeModel", modelId });
1242
+ }).map((p) => p.catch(() => void 0))
1243
+ );
1244
+ this.ids.clear();
1245
+ }
1246
+ };
1247
+ var WorkerSessionProxy = class {
1248
+ constructor(slot, model, systemPrompt, requestedSize) {
1249
+ this.slot = slot;
1250
+ this.model = model;
1251
+ this.systemPrompt = systemPrompt;
1252
+ this.requestedSize = requestedSize;
1253
+ }
1254
+ slot;
1255
+ model;
1256
+ systemPrompt;
1257
+ requestedSize;
1258
+ contextSize;
1259
+ ids = /* @__PURE__ */ new Map();
1260
+ async open() {
1261
+ await this.slot.run((host) => this.on(host));
1262
+ }
1263
+ /** This session's id in `host`. A retried prompt gets a fresh one there. */
1264
+ on(host) {
1265
+ let id = this.ids.get(host);
1266
+ if (!id) {
1267
+ id = (async () => {
1268
+ const opened = await host.request({
1269
+ op: "createSession",
1270
+ modelId: (await this.model.on(host)).modelId,
1271
+ systemPrompt: this.systemPrompt,
1272
+ ...this.requestedSize != null ? { contextSize: this.requestedSize } : {}
1273
+ });
1274
+ this.contextSize = opened.contextSize;
1275
+ return opened.sessionId;
1276
+ })();
1277
+ this.ids.set(host, id);
1278
+ const settled = id;
1279
+ settled.catch(() => {
1280
+ if (this.ids.get(host) === settled) this.ids.delete(host);
1281
+ });
1282
+ }
1283
+ return id;
1284
+ }
1285
+ prompt(text, options) {
1286
+ return this.slot.run(
1287
+ async (host) => host.request({
1288
+ op: "prompt",
1289
+ sessionId: await this.on(host),
1290
+ text,
1291
+ options
1292
+ })
1293
+ );
1294
+ }
1295
+ async dispose() {
1296
+ await Promise.all(
1297
+ [...this.ids].map(async ([host, id]) => {
1298
+ if (host.dead) return;
1299
+ await host.request({ op: "disposeSession", sessionId: await id });
1300
+ }).map((p) => p.catch(() => void 0))
1301
+ );
1302
+ this.ids.clear();
1303
+ }
1304
+ };
1305
+ var warnedInProcess = false;
1306
+ function inProcessRuntime(gpu) {
1307
+ if (!warnedInProcess) {
1308
+ warnedInProcess = true;
1309
+ console.warn(
1310
+ `inference: the local-model worker (llama-worker.js) is missing beside this library \u2014 was it bundled? Running llama.cpp in-process instead, so a native crash in llama.cpp will end this process.`
1311
+ );
1312
+ }
1313
+ let opened;
1314
+ const backend = () => opened ??= backendSource().then(
1315
+ async (source) => openBackend(await import(source.moduleUrl), { gpu }, { trying: () => void 0 })
1316
+ ).catch((e) => {
1317
+ opened = void 0;
1318
+ throw e;
1319
+ });
1320
+ return {
1321
+ resolveModelFile: (uri, directory) => backendSource().then((source) => source.resolveModelFile(uri, directory)),
1322
+ loadModel: (path) => backend().then((b) => b.loadModel(path)),
1323
+ getMemoryBudgetBytes: () => backend().then((b) => b.memoryBudget())
1324
+ };
1325
+ }
1326
+ function createWorkerRuntime(options = {}) {
1327
+ const { gpu, source } = requestedGpu(options.gpu);
1328
+ const entry = workerEntry();
1329
+ if (!entry) return inProcessRuntime(gpu);
1330
+ const slot = slotFor(gpu, source, entry);
1331
+ return {
1332
+ resolveModelFile: (uri, directory) => backendSource().then((s) => s.resolveModelFile(uri, directory)),
1333
+ async loadModel(path) {
1334
+ const model = new WorkerModelProxy(slot, path);
1335
+ await model.open();
1336
+ return {
1337
+ trainContextSize: model.trainContextSize,
1338
+ countTokens: (text) => model.countTokens(text),
1339
+ createSession: (systemPrompt, contextSize) => model.createSession(systemPrompt, contextSize),
1340
+ dispose: () => model.dispose()
1341
+ };
1342
+ },
1343
+ getMemoryBudgetBytes: () => slot.run((host) => host.request({ op: "memoryBudget" }))
1344
+ };
1345
+ }
1346
+ async function shutdownLlamaWorkers() {
1347
+ await Promise.all([...slots.values()].map((slot) => slot.shutdown()));
1348
+ }
1349
+
789
1350
  // src/providers/llama-cpp.ts
790
1351
  var loadedModels = /* @__PURE__ */ new Map();
791
1352
  async function disposeLlamaModels() {
@@ -794,6 +1355,7 @@ async function disposeLlamaModels() {
794
1355
  await Promise.all(
795
1356
  pending.map((p) => p.then((m) => m.dispose()).catch(() => void 0))
796
1357
  );
1358
+ await shutdownLlamaWorkers();
797
1359
  }
798
1360
  var DEFAULT_CONTEXT_SIZE = 8192;
799
1361
  var CHAT_TEMPLATE_OVERHEAD_TOKENS = 512;
@@ -811,9 +1373,14 @@ var LlamaCppProvider = class {
811
1373
  `llamaCpp.contextSize must be a positive integer number of tokens, got ${String(options.contextSize)}.`
812
1374
  );
813
1375
  }
1376
+ if (options.gpu !== void 0 && !isLlamaGpu(options.gpu)) {
1377
+ throw new InferenceError(
1378
+ `llamaCpp.gpu must be "auto", "cuda", "vulkan", "metal" or false, got ${JSON.stringify(options.gpu) ?? String(options.gpu)}.`
1379
+ );
1380
+ }
814
1381
  this.contextSize = options.contextSize;
815
1382
  this.uri = resolveLlamaModelRef(model);
816
- this.runtime = options.runtime ?? defaultLlamaRuntime();
1383
+ this.runtime = options.runtime ?? defaultLlamaRuntime(options.gpu !== void 0 ? { gpu: options.gpu } : {});
817
1384
  this.thoughtTokens = options.thoughtTokens ?? 0;
818
1385
  this.maxTokens = options.maxTokens;
819
1386
  this.modelsDirectory = options.modelsDirectory ?? defaultLlamaModelsDirectory();
@@ -842,7 +1409,7 @@ var LlamaCppProvider = class {
842
1409
  async completeJSON(req) {
843
1410
  const model = await this.load();
844
1411
  const systemPrompt = systemPromptFor(req);
845
- const plan = this.contextFor(model, systemPrompt, req.user);
1412
+ const plan = await this.contextFor(model, systemPrompt, req.user);
846
1413
  const session = await model.createSession(systemPrompt, plan.contextSize);
847
1414
  const contextSize = session.contextSize ?? plan.contextSize;
848
1415
  const implicitMaxTokens = this.maxTokens == null && plan.promptTokens != null ? contextSize - plan.promptTokens : void 0;
@@ -875,12 +1442,14 @@ var LlamaCppProvider = class {
875
1442
  * fit is refused here, before anything is created. llama.cpp would otherwise
876
1443
  * shift the overflow out of the context and answer a prompt nobody sent.
877
1444
  */
878
- contextFor(model, systemPrompt, user) {
1445
+ async contextFor(model, systemPrompt, user) {
879
1446
  const ceiling = model.trainContextSize;
880
1447
  const fallback = this.contextSize ?? (ceiling != null ? Math.min(DEFAULT_CONTEXT_SIZE, ceiling) : DEFAULT_CONTEXT_SIZE);
881
1448
  if (!model.countTokens) return { contextSize: fallback };
882
- const system = model.countTokens(systemPrompt);
883
- const prompt = model.countTokens(user);
1449
+ const [system, prompt] = await Promise.all([
1450
+ model.countTokens(systemPrompt),
1451
+ model.countTokens(user)
1452
+ ]);
884
1453
  const response = (this.maxTokens ?? DEFAULT_RESPONSE_RESERVE_TOKENS) + this.thoughtTokens;
885
1454
  const promptTokens = system + prompt + CHAT_TEMPLATE_OVERHEAD_TOKENS;
886
1455
  const needed = promptTokens + response;
@@ -944,100 +1513,8 @@ function restoreOpenBrace(text) {
944
1513
  return text;
945
1514
  }
946
1515
  }
947
- var runtimePromise;
948
- function defaultLlamaRuntime() {
949
- const real = () => (
950
- // Drop a failed init so the next call retries. A GPU that failed to
951
- // initialise, or a binary still being extracted by a concurrent install,
952
- // must not poison the runtime for the rest of the process — the same rule
953
- // `load()` applies to weights.
954
- runtimePromise ??= loadNodeLlamaCpp().catch((e) => {
955
- runtimePromise = void 0;
956
- throw e;
957
- })
958
- );
959
- return {
960
- resolveModelFile: (uri, directory) => real().then((r) => r.resolveModelFile(uri, directory)),
961
- loadModel: (path) => real().then((r) => r.loadModel(path)),
962
- getMemoryBudgetBytes: () => real().then((r) => r.getMemoryBudgetBytes())
963
- };
964
- }
965
- async function loadNodeLlamaCpp() {
966
- let mod;
967
- try {
968
- mod = await import("node-llama-cpp");
969
- } catch (e) {
970
- if (!isModuleNotFound(e)) {
971
- throw new InferenceError(
972
- `node-llama-cpp is installed but failed to load (${e instanceof Error ? e.message : String(e)}). This is the copy resolved from your own node_modules, so reinstalling it here will not help \u2014 check the Node version and the platform build.`
973
- );
974
- }
975
- mod = await importNodeLlamaCpp();
976
- }
977
- const { getLlama, resolveModelFile, LlamaChatSession, TokenMeter } = mod;
978
- const llama = await getLlama();
979
- return {
980
- // `directory` is this library's own, not node-llama-cpp's global default —
981
- // owning it is what makes `clearLlamaModels` safe.
982
- resolveModelFile: (uri, directory) => resolveModelFile(uri, { directory }),
983
- async loadModel(path) {
984
- const model = await llama.loadModel({ modelPath: path });
985
- return {
986
- trainContextSize: model.trainContextSize,
987
- countTokens: (text) => model.tokenize(text).length,
988
- async createSession(systemPrompt, contextSize) {
989
- const context = await model.createContext({
990
- contextSize: contextSize ?? DEFAULT_CONTEXT_SIZE
991
- });
992
- const sequence = context.getSequence();
993
- const session = new LlamaChatSession({
994
- contextSequence: sequence,
995
- systemPrompt
996
- });
997
- return {
998
- contextSize: context.contextSize,
999
- async prompt(text, options) {
1000
- const grammar = await llama.createGrammarForJsonSchema(
1001
- options.schema
1002
- );
1003
- const before = sequence.tokenMeter.getState();
1004
- const result = await session.promptWithMeta(text, {
1005
- grammar,
1006
- temperature: options.temperature,
1007
- budgets: { thoughtTokens: options.thoughtTokens },
1008
- ...options.maxTokens != null ? { maxTokens: options.maxTokens } : {}
1009
- });
1010
- const diff = TokenMeter.diff(sequence.tokenMeter, before);
1011
- return {
1012
- text: result.responseText,
1013
- stopReason: result.stopReason,
1014
- usage: {
1015
- inputTokens: diff.usedInputTokens,
1016
- outputTokens: diff.usedOutputTokens
1017
- }
1018
- };
1019
- },
1020
- async dispose() {
1021
- await context.dispose();
1022
- }
1023
- };
1024
- },
1025
- async dispose() {
1026
- await model.dispose();
1027
- }
1028
- };
1029
- },
1030
- async getMemoryBudgetBytes() {
1031
- const { totalmem } = await import("os");
1032
- const ramBudget = totalmem() / 2;
1033
- try {
1034
- const vram = await llama.getVramState();
1035
- return Math.max(vram.free, ramBudget);
1036
- } catch {
1037
- return ramBudget;
1038
- }
1039
- }
1040
- };
1516
+ function defaultLlamaRuntime(options = {}) {
1517
+ return createWorkerRuntime(options);
1041
1518
  }
1042
1519
 
1043
1520
  // src/providers/detect.ts
@@ -1197,7 +1674,7 @@ async function resolveProviderIdentityAsync(spec) {
1197
1674
  if (provider !== "llama-cpp" || !isLlamaSelector(model)) {
1198
1675
  return resolveProviderIdentity(resolved);
1199
1676
  }
1200
- const tier = model === "auto" ? await probeTier(llamaRuntimeFor(spec)) : model;
1677
+ const tier = model === "auto" ? await probeTier(llamaRuntimeFor(spec), spec.llamaCpp?.gpu) : model;
1201
1678
  return { provider, model: aliasForTier(tier) };
1202
1679
  }
1203
1680
  function warnIfDownloadPending(spec, model) {
@@ -1211,8 +1688,8 @@ function warnIfDownloadPending(spec, model) {
1211
1688
  function llamaRuntimeFor(spec) {
1212
1689
  return spec.llamaRuntime ?? spec.llamaCpp?.runtime;
1213
1690
  }
1214
- async function probeTier(runtime) {
1215
- const source = runtime ?? defaultLlamaRuntime();
1691
+ async function probeTier(runtime, gpu) {
1692
+ const source = runtime ?? defaultLlamaRuntime(gpu !== void 0 ? { gpu } : {});
1216
1693
  return tierForBudget(await source.getMemoryBudgetBytes());
1217
1694
  }
1218
1695
  function makeProvider(spec) {