@hawkeyexl/inference 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +64 -138
- package/dist/index.d.ts +338 -7
- package/dist/index.js +830 -56
- package/dist/index.js.map +1 -1
- package/package.json +15 -3
package/dist/index.js
CHANGED
|
@@ -6,6 +6,22 @@ var InferenceError = class extends Error {
|
|
|
6
6
|
}
|
|
7
7
|
};
|
|
8
8
|
|
|
9
|
+
// src/runtime.ts
|
|
10
|
+
var MINIMUM_NODE_MAJOR = 24;
|
|
11
|
+
var warnedNodeVersion = false;
|
|
12
|
+
function warnIfUnsupportedNode(version = process.versions.node) {
|
|
13
|
+
if (warnedNodeVersion) return;
|
|
14
|
+
const major = Number.parseInt(version, 10);
|
|
15
|
+
if (!Number.isInteger(major) || major >= MINIMUM_NODE_MAJOR) return;
|
|
16
|
+
warnedNodeVersion = true;
|
|
17
|
+
console.warn(
|
|
18
|
+
`inference: running on Node ${version}, older than the Node ${MINIMUM_NODE_MAJOR} this package requires. npm only warns about that at install time (EBADENGINE), so nothing has stopped you yet \u2014 upgrade Node, or expect failures this library cannot explain.`
|
|
19
|
+
);
|
|
20
|
+
}
|
|
21
|
+
function resetNodeVersionWarning() {
|
|
22
|
+
warnedNodeVersion = false;
|
|
23
|
+
}
|
|
24
|
+
|
|
9
25
|
// src/providers/anthropic.ts
|
|
10
26
|
import Anthropic from "@anthropic-ai/sdk";
|
|
11
27
|
var DEFAULT_TOOL_NAME = "record_result";
|
|
@@ -331,7 +347,15 @@ var ClaudeCliProvider = class {
|
|
|
331
347
|
`Claude CLI exited ${result.code}: ${result.stderr.trim().slice(-300)}`
|
|
332
348
|
);
|
|
333
349
|
}
|
|
334
|
-
|
|
350
|
+
let wrapper;
|
|
351
|
+
try {
|
|
352
|
+
wrapper = JSON.parse(result.stdout);
|
|
353
|
+
} catch {
|
|
354
|
+
const excerpt = result.stdout.trim().replace(/\s+/g, " ").slice(0, 200);
|
|
355
|
+
throw new Error(
|
|
356
|
+
`Claude CLI printed non-JSON output (is it logged in?): ${excerpt || "(no output)"}`
|
|
357
|
+
);
|
|
358
|
+
}
|
|
335
359
|
if (typeof wrapper.result !== "string") {
|
|
336
360
|
throw new Error("Claude CLI returned no result field");
|
|
337
361
|
}
|
|
@@ -384,12 +408,691 @@ function mockVerdict(match, confidence, overrides = {}) {
|
|
|
384
408
|
};
|
|
385
409
|
}
|
|
386
410
|
|
|
411
|
+
// src/cache.ts
|
|
412
|
+
import { createHash } from "crypto";
|
|
413
|
+
import { existsSync, mkdirSync, readFileSync, writeFileSync } from "fs";
|
|
414
|
+
import { join } from "path";
|
|
415
|
+
function sha256(text) {
|
|
416
|
+
return createHash("sha256").update(text, "utf8").digest("hex");
|
|
417
|
+
}
|
|
418
|
+
function buildCacheKey(parts) {
|
|
419
|
+
return sha256(parts.map((p) => `${p.length}:${p}`).join("|"));
|
|
420
|
+
}
|
|
421
|
+
var JsonCache = class {
|
|
422
|
+
constructor(dir, enabled = true, label = "inference") {
|
|
423
|
+
this.dir = dir;
|
|
424
|
+
this.enabled = enabled;
|
|
425
|
+
this.label = label;
|
|
426
|
+
}
|
|
427
|
+
dir;
|
|
428
|
+
enabled;
|
|
429
|
+
label;
|
|
430
|
+
/** Cache-write failures warn once per process, not once per entry. */
|
|
431
|
+
warned = false;
|
|
432
|
+
get(key) {
|
|
433
|
+
if (!this.enabled) return void 0;
|
|
434
|
+
const path = join(this.dir, `${key}.json`);
|
|
435
|
+
if (!existsSync(path)) return void 0;
|
|
436
|
+
try {
|
|
437
|
+
return JSON.parse(readFileSync(path, "utf8"));
|
|
438
|
+
} catch {
|
|
439
|
+
return void 0;
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
set(key, value) {
|
|
443
|
+
if (!this.enabled) return;
|
|
444
|
+
try {
|
|
445
|
+
mkdirSync(this.dir, { recursive: true });
|
|
446
|
+
writeFileSync(
|
|
447
|
+
join(this.dir, `${key}.json`),
|
|
448
|
+
JSON.stringify(value, null, 2)
|
|
449
|
+
);
|
|
450
|
+
} catch (e) {
|
|
451
|
+
if (!this.warned) {
|
|
452
|
+
this.warned = true;
|
|
453
|
+
console.warn(
|
|
454
|
+
`${this.label}: could not write the cache at ${this.dir} (${e instanceof Error ? e.message : String(e)}). Continuing without caching.`
|
|
455
|
+
);
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
}
|
|
459
|
+
};
|
|
460
|
+
|
|
461
|
+
// src/providers/llama-models.ts
|
|
462
|
+
import { readdirSync } from "fs";
|
|
463
|
+
import { homedir } from "os";
|
|
464
|
+
import { join as join2 } from "path";
|
|
465
|
+
function defaultLlamaModelsDirectory() {
|
|
466
|
+
return process.env["INFERENCE_MODELS_DIR"] || join2(homedir(), ".hawkeyexl-inference", "models");
|
|
467
|
+
}
|
|
468
|
+
var LLAMA_TIERS = ["fast", "balanced", "quality"];
|
|
469
|
+
var LLAMA_SELECTORS = ["auto", ...LLAMA_TIERS];
|
|
470
|
+
var LLAMA_MODELS = deepFreezeEntries({
|
|
471
|
+
"granite-4.1-3b-q2": {
|
|
472
|
+
uri: "hf:unsloth/granite-4.1-3b-GGUF/granite-4.1-3b-UD-Q2_K_XL.gguf",
|
|
473
|
+
sizeBytes: 1414548800,
|
|
474
|
+
license: "Apache-2.0",
|
|
475
|
+
tier: "fast",
|
|
476
|
+
notes: "Smallest tier and the quickest measured (4.8s/page). Scores level with models three times its size on schema-constrained extraction."
|
|
477
|
+
},
|
|
478
|
+
"qwen3.5-4b": {
|
|
479
|
+
uri: "hf:unsloth/Qwen3.5-4B-GGUF/Qwen3.5-4B-UD-Q4_K_XL.gguf",
|
|
480
|
+
sizeBytes: 2912109728,
|
|
481
|
+
license: "Apache-2.0",
|
|
482
|
+
notes: "The default for most machines. Smaller and faster than the Gemma 4 E4B it replaces, at indistinguishable measured quality.",
|
|
483
|
+
tier: "balanced"
|
|
484
|
+
},
|
|
485
|
+
"qwen3.5-9b": {
|
|
486
|
+
uri: "hf:unsloth/Qwen3.5-9B-GGUF/Qwen3.5-9B-UD-Q4_K_XL.gguf",
|
|
487
|
+
sizeBytes: 5966095584,
|
|
488
|
+
license: "Apache-2.0",
|
|
489
|
+
tier: "quality",
|
|
490
|
+
notes: "Best measured of everything tried, and still smaller and faster than the Gemma 4 12B it replaces. Wants a GPU or plenty of RAM."
|
|
491
|
+
},
|
|
492
|
+
// --- Superseded, kept resolvable by name. Untiered: nothing selects these
|
|
493
|
+
// unless a caller asks for one outright.
|
|
494
|
+
"gemma-4-e2b": {
|
|
495
|
+
uri: "hf:unsloth/gemma-4-E2B-it-qat-GGUF/gemma-4-E2B-it-qat-UD-Q4_K_XL.gguf",
|
|
496
|
+
sizeBytes: 2620370976,
|
|
497
|
+
license: "Apache-2.0",
|
|
498
|
+
notes: "IFEval 94.6. Former `fast` tier; sound, but larger than Granite."
|
|
499
|
+
},
|
|
500
|
+
"gemma-4-e4b": {
|
|
501
|
+
uri: "hf:unsloth/gemma-4-E4B-it-qat-GGUF/gemma-4-E4B-it-qat-UD-Q4_K_XL.gguf",
|
|
502
|
+
sizeBytes: 4215695776,
|
|
503
|
+
license: "Apache-2.0",
|
|
504
|
+
notes: "IFEval 96.7. Former `balanced` tier; sound, but larger."
|
|
505
|
+
},
|
|
506
|
+
"gemma-4-12b": {
|
|
507
|
+
uri: "hf:unsloth/gemma-4-12B-it-qat-GGUF/gemma-4-12B-it-qat-UD-Q4_K_XL.gguf",
|
|
508
|
+
sizeBytes: 6716356800,
|
|
509
|
+
license: "Apache-2.0",
|
|
510
|
+
notes: "IFEval 97.2. Former `quality` tier. Dense 12B; wants a GPU."
|
|
511
|
+
},
|
|
512
|
+
"gemma-4-26b-a4b": {
|
|
513
|
+
uri: "hf:unsloth/gemma-4-26B-A4B-it-qat-GGUF/gemma-4-26B-A4B-it-qat-UD-Q4_K_XL.gguf",
|
|
514
|
+
sizeBytes: 14249047104,
|
|
515
|
+
license: "Apache-2.0",
|
|
516
|
+
notes: "MoE: 25.2B total, 3.8B active \u2014 infers near E4B speed if it fits in memory."
|
|
517
|
+
},
|
|
518
|
+
"gemma-4-e2b-q2": {
|
|
519
|
+
uri: "hf:unsloth/gemma-4-E2B-it-qat-GGUF/gemma-4-E2B-it-qat-UD-Q2_K_XL.gguf",
|
|
520
|
+
sizeBytes: 2186186784,
|
|
521
|
+
license: "Apache-2.0",
|
|
522
|
+
notes: "AVOID. Smallest download, but it does not reliably terminate: 6 of 12 pages unfinished at 120s, one still running at 400s, and the pages that did finish proposed identifiers as prose. Kept only so existing pins still resolve."
|
|
523
|
+
}
|
|
524
|
+
});
|
|
525
|
+
function deepFreezeEntries(catalog) {
|
|
526
|
+
for (const entry of Object.values(catalog)) Object.freeze(entry);
|
|
527
|
+
return Object.freeze(catalog);
|
|
528
|
+
}
|
|
529
|
+
var TIER_ALIAS = {
|
|
530
|
+
fast: "granite-4.1-3b-q2",
|
|
531
|
+
balanced: "qwen3.5-4b",
|
|
532
|
+
quality: "qwen3.5-9b"
|
|
533
|
+
};
|
|
534
|
+
function isLlamaSelector(model) {
|
|
535
|
+
return LLAMA_SELECTORS.includes(model);
|
|
536
|
+
}
|
|
537
|
+
var MEMORY_HEADROOM = 3.5;
|
|
538
|
+
function tierForBudget(budgetBytes) {
|
|
539
|
+
let chosen = "fast";
|
|
540
|
+
for (const tier of LLAMA_TIERS) {
|
|
541
|
+
const entry = LLAMA_MODELS[TIER_ALIAS[tier]];
|
|
542
|
+
if (entry.sizeBytes * MEMORY_HEADROOM <= budgetBytes) chosen = tier;
|
|
543
|
+
}
|
|
544
|
+
return chosen;
|
|
545
|
+
}
|
|
546
|
+
function aliasForTier(tier) {
|
|
547
|
+
return TIER_ALIAS[tier];
|
|
548
|
+
}
|
|
549
|
+
function uriForTier(tier) {
|
|
550
|
+
return LLAMA_MODELS[TIER_ALIAS[tier]].uri;
|
|
551
|
+
}
|
|
552
|
+
function resolveLlamaModelRef(model) {
|
|
553
|
+
if (isLlamaSelector(model)) {
|
|
554
|
+
throw new InferenceError(
|
|
555
|
+
`llama-cpp model "${model}" is a selector and needs a hardware probe to resolve. Use resolveProviderIdentityAsync/makeProviderAsync, or name a concrete model (e.g. "${TIER_ALIAS.balanced}").`
|
|
556
|
+
);
|
|
557
|
+
}
|
|
558
|
+
const entry = LLAMA_MODELS[model];
|
|
559
|
+
if (entry) return entry.uri;
|
|
560
|
+
if (isModelPathOrUri(model)) return model;
|
|
561
|
+
throw new InferenceError(
|
|
562
|
+
`Unknown llama-cpp model "${model}". Use a selector (${LLAMA_SELECTORS.join(
|
|
563
|
+
", "
|
|
564
|
+
)}), a curated alias (${Object.keys(LLAMA_MODELS).join(
|
|
565
|
+
", "
|
|
566
|
+
)}), an hf: URI, or a path to a .gguf file.`
|
|
567
|
+
);
|
|
568
|
+
}
|
|
569
|
+
function blobNameFor(model) {
|
|
570
|
+
const ref = resolveLlamaModelRef(model).split("#")[0];
|
|
571
|
+
return ref.split(/[/\\]/).pop();
|
|
572
|
+
}
|
|
573
|
+
function matchesModelBlob(entry, blobName) {
|
|
574
|
+
const base = entry.replace(/\.ipull$/, "");
|
|
575
|
+
if (base.endsWith(blobName)) return true;
|
|
576
|
+
const stem = blobName.replace(/\.gguf$/, "");
|
|
577
|
+
return new RegExp(`${escapeRegExp(stem)}-\\d{5}-of-\\d{5}\\.gguf$`).test(base);
|
|
578
|
+
}
|
|
579
|
+
function escapeRegExp(text) {
|
|
580
|
+
return text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
581
|
+
}
|
|
582
|
+
function isModelDownloaded(model, directory) {
|
|
583
|
+
const blobName = blobNameFor(model);
|
|
584
|
+
return listModelDirectory(directory).some(
|
|
585
|
+
(entry) => !entry.endsWith(".ipull") && matchesModelBlob(entry, blobName)
|
|
586
|
+
);
|
|
587
|
+
}
|
|
588
|
+
function listModelDirectory(directory) {
|
|
589
|
+
try {
|
|
590
|
+
return readdirSync(directory, { withFileTypes: true }).filter((entry) => entry.isFile()).map((entry) => entry.name);
|
|
591
|
+
} catch {
|
|
592
|
+
return [];
|
|
593
|
+
}
|
|
594
|
+
}
|
|
595
|
+
function isModelPathOrUri(model) {
|
|
596
|
+
return /^(hf|huggingface):/i.test(model) || /^https?:\/\//i.test(model) || /^(hf|huggingface)\.co\//i.test(model) || model.endsWith(".gguf");
|
|
597
|
+
}
|
|
598
|
+
|
|
599
|
+
// src/providers/llama-install.ts
|
|
600
|
+
import {
|
|
601
|
+
existsSync as existsSync2,
|
|
602
|
+
mkdirSync as mkdirSync2,
|
|
603
|
+
rmSync,
|
|
604
|
+
statSync,
|
|
605
|
+
writeFileSync as writeFileSync2
|
|
606
|
+
} from "fs";
|
|
607
|
+
import { homedir as homedir2 } from "os";
|
|
608
|
+
import { join as join3 } from "path";
|
|
609
|
+
import { pathToFileURL } from "url";
|
|
610
|
+
var PACKAGE_SPEC = "node-llama-cpp@^3.19.0";
|
|
611
|
+
var SHIM = "loader.mjs";
|
|
612
|
+
var LOCK = ".install.lock";
|
|
613
|
+
var INSTALL_TIMEOUT_MS = 9e5;
|
|
614
|
+
var LOCK_WAIT_MS = INSTALL_TIMEOUT_MS;
|
|
615
|
+
var LOCK_STALE_MS = INSTALL_TIMEOUT_MS + 6e4;
|
|
616
|
+
function defaultLlamaRuntimeDirectory(env = process.env) {
|
|
617
|
+
return env["INFERENCE_RUNTIME_DIR"] || join3(homedir2(), ".hawkeyexl-inference", "runtime");
|
|
618
|
+
}
|
|
619
|
+
function isModuleNotFound(e) {
|
|
620
|
+
const code = e?.code;
|
|
621
|
+
return code === "ERR_MODULE_NOT_FOUND" || code === "MODULE_NOT_FOUND";
|
|
622
|
+
}
|
|
623
|
+
function describe(e) {
|
|
624
|
+
return e instanceof Error ? e.message : String(e);
|
|
625
|
+
}
|
|
626
|
+
async function nodeLlamaCppStatus(options = {}) {
|
|
627
|
+
const env = options.env ?? process.env;
|
|
628
|
+
const directory = options.directory ?? defaultLlamaRuntimeDirectory(env);
|
|
629
|
+
const probeImport = options.probeImport ?? (() => import("node-llama-cpp"));
|
|
630
|
+
try {
|
|
631
|
+
await probeImport();
|
|
632
|
+
return { state: "present" };
|
|
633
|
+
} catch (e) {
|
|
634
|
+
if (!isModuleNotFound(e)) {
|
|
635
|
+
return {
|
|
636
|
+
state: "refused",
|
|
637
|
+
reason: `node-llama-cpp is installed but failed to load (${describe(e)})`
|
|
638
|
+
};
|
|
639
|
+
}
|
|
640
|
+
}
|
|
641
|
+
if (existsSync2(join3(directory, SHIM))) return { state: "present" };
|
|
642
|
+
if ((env["INFERENCE_NO_AUTO_INSTALL"] ?? "") !== "") {
|
|
643
|
+
return {
|
|
644
|
+
state: "refused",
|
|
645
|
+
reason: `node-llama-cpp is not installed and INFERENCE_NO_AUTO_INSTALL is set`
|
|
646
|
+
};
|
|
647
|
+
}
|
|
648
|
+
return { state: "installable", directory };
|
|
649
|
+
}
|
|
650
|
+
var installs = /* @__PURE__ */ new Map();
|
|
651
|
+
var warnedInstall = false;
|
|
652
|
+
function resetRuntimeInstall() {
|
|
653
|
+
installs.clear();
|
|
654
|
+
warnedInstall = false;
|
|
655
|
+
}
|
|
656
|
+
function importNodeLlamaCpp(options = {}) {
|
|
657
|
+
const env = options.env ?? process.env;
|
|
658
|
+
const directory = options.directory ?? defaultLlamaRuntimeDirectory(env);
|
|
659
|
+
const existing = installs.get(directory);
|
|
660
|
+
if (existing) return existing;
|
|
661
|
+
const pending = fromPrefix(directory, env, options);
|
|
662
|
+
const guarded = pending.catch((e) => {
|
|
663
|
+
if (installs.get(directory) === guarded) installs.delete(directory);
|
|
664
|
+
throw e;
|
|
665
|
+
});
|
|
666
|
+
installs.set(directory, guarded);
|
|
667
|
+
return guarded;
|
|
668
|
+
}
|
|
669
|
+
async function fromPrefix(directory, env, options) {
|
|
670
|
+
const importShim = options.importShim ?? ((url) => import(url));
|
|
671
|
+
const shim = join3(directory, SHIM);
|
|
672
|
+
if (existsSync2(shim)) return importShim(pathToFileURL(shim).href);
|
|
673
|
+
if ((env["INFERENCE_NO_AUTO_INSTALL"] ?? "") !== "") {
|
|
674
|
+
throw new InferenceError(
|
|
675
|
+
`The llama-cpp provider needs node-llama-cpp, and INFERENCE_NO_AUTO_INSTALL is set. Install it yourself (npm i ${PACKAGE_SPEC}), unset INFERENCE_NO_AUTO_INSTALL to allow installing into ${directory}, or name a different provider.`
|
|
676
|
+
);
|
|
677
|
+
}
|
|
678
|
+
mkdirSync2(directory, { recursive: true });
|
|
679
|
+
await withLock(directory, async () => {
|
|
680
|
+
if (existsSync2(shim)) return;
|
|
681
|
+
warnInstalling(directory);
|
|
682
|
+
await runInstall(directory, env, options);
|
|
683
|
+
writeFileSync2(shim, `export * from "node-llama-cpp";
|
|
684
|
+
`, "utf8");
|
|
685
|
+
});
|
|
686
|
+
return importShim(pathToFileURL(shim).href);
|
|
687
|
+
}
|
|
688
|
+
async function runInstall(directory, env, options) {
|
|
689
|
+
const manifest = join3(directory, "package.json");
|
|
690
|
+
if (!existsSync2(manifest)) {
|
|
691
|
+
writeFileSync2(
|
|
692
|
+
manifest,
|
|
693
|
+
`${JSON.stringify(
|
|
694
|
+
{
|
|
695
|
+
name: "hawkeyexl-inference-runtime",
|
|
696
|
+
version: "0.0.0",
|
|
697
|
+
private: true,
|
|
698
|
+
description: "Auto-installed runtime for @hawkeyexl/inference. Safe to delete."
|
|
699
|
+
},
|
|
700
|
+
null,
|
|
701
|
+
2
|
|
702
|
+
)}
|
|
703
|
+
`,
|
|
704
|
+
"utf8"
|
|
705
|
+
);
|
|
706
|
+
}
|
|
707
|
+
const exec = options.exec ?? realExec;
|
|
708
|
+
const result = await exec(
|
|
709
|
+
[
|
|
710
|
+
"npm",
|
|
711
|
+
"install",
|
|
712
|
+
"--prefix",
|
|
713
|
+
directory,
|
|
714
|
+
PACKAGE_SPEC,
|
|
715
|
+
"--no-audit",
|
|
716
|
+
"--no-fund"
|
|
717
|
+
],
|
|
718
|
+
{ timeoutMs: options.timeoutMs ?? INSTALL_TIMEOUT_MS, env }
|
|
719
|
+
);
|
|
720
|
+
if (result.spawnError != null) {
|
|
721
|
+
throw new InferenceError(
|
|
722
|
+
`Could not run npm to install node-llama-cpp (${result.spawnError}). Install it yourself with: npm i ${PACKAGE_SPEC}, or name a provider that does not need it.`
|
|
723
|
+
);
|
|
724
|
+
}
|
|
725
|
+
if (result.timedOut) {
|
|
726
|
+
throw new InferenceError(
|
|
727
|
+
`Installing node-llama-cpp into ${directory} timed out. A source build can take a while \u2014 retry, raise the timeout, or install it yourself with: npm i ${PACKAGE_SPEC}.`
|
|
728
|
+
);
|
|
729
|
+
}
|
|
730
|
+
if (result.code !== 0) {
|
|
731
|
+
throw new InferenceError(
|
|
732
|
+
`Installing node-llama-cpp into ${directory} failed (exit ${String(
|
|
733
|
+
result.code
|
|
734
|
+
)}).
|
|
735
|
+
${tail(result.stderr || result.stdout)}
|
|
736
|
+
Install it yourself with: npm i ${PACKAGE_SPEC}, or name a provider that does not need it.`
|
|
737
|
+
);
|
|
738
|
+
}
|
|
739
|
+
}
|
|
740
|
+
function tail(output, lines = 12) {
|
|
741
|
+
return output.trimEnd().split(/\r?\n/).slice(-lines).join("\n");
|
|
742
|
+
}
|
|
743
|
+
function warnInstalling(directory) {
|
|
744
|
+
if (warnedInstall) return;
|
|
745
|
+
warnedInstall = true;
|
|
746
|
+
console.warn(
|
|
747
|
+
`inference: node-llama-cpp is not installed \u2014 fetching it into ${directory} so the local model can run. This is a one-time native install; set INFERENCE_NO_AUTO_INSTALL=1 to refuse it, or name a provider that does not need it.`
|
|
748
|
+
);
|
|
749
|
+
}
|
|
750
|
+
async function withLock(directory, fn) {
|
|
751
|
+
const lock = join3(directory, LOCK);
|
|
752
|
+
const deadline = Date.now() + LOCK_WAIT_MS;
|
|
753
|
+
for (; ; ) {
|
|
754
|
+
try {
|
|
755
|
+
writeFileSync2(lock, String(process.pid), { flag: "wx" });
|
|
756
|
+
break;
|
|
757
|
+
} catch (e) {
|
|
758
|
+
if (e.code !== "EEXIST") throw e;
|
|
759
|
+
if (ageOf(lock) > LOCK_STALE_MS) {
|
|
760
|
+
rmSync(lock, { force: true });
|
|
761
|
+
continue;
|
|
762
|
+
}
|
|
763
|
+
if (existsSync2(join3(directory, SHIM))) return;
|
|
764
|
+
if (Date.now() > deadline) {
|
|
765
|
+
throw new InferenceError(
|
|
766
|
+
`Timed out waiting for another process to install node-llama-cpp into ${directory}. If nothing else is running, remove ${lock} and retry.`
|
|
767
|
+
);
|
|
768
|
+
}
|
|
769
|
+
await delay(250);
|
|
770
|
+
}
|
|
771
|
+
}
|
|
772
|
+
try {
|
|
773
|
+
await fn();
|
|
774
|
+
} finally {
|
|
775
|
+
rmSync(lock, { force: true });
|
|
776
|
+
}
|
|
777
|
+
}
|
|
778
|
+
function ageOf(path) {
|
|
779
|
+
try {
|
|
780
|
+
return Date.now() - statSync(path).mtimeMs;
|
|
781
|
+
} catch {
|
|
782
|
+
return Number.POSITIVE_INFINITY;
|
|
783
|
+
}
|
|
784
|
+
}
|
|
785
|
+
function delay(ms) {
|
|
786
|
+
return new Promise((resolve) => setTimeout(resolve, ms));
|
|
787
|
+
}
|
|
788
|
+
|
|
789
|
+
// src/providers/llama-cpp.ts
|
|
790
|
+
var loadedModels = /* @__PURE__ */ new Map();
|
|
791
|
+
async function disposeLlamaModels() {
|
|
792
|
+
const pending = [...loadedModels.values()];
|
|
793
|
+
loadedModels.clear();
|
|
794
|
+
await Promise.all(
|
|
795
|
+
pending.map((p) => p.then((m) => m.dispose()).catch(() => void 0))
|
|
796
|
+
);
|
|
797
|
+
}
|
|
798
|
+
var LlamaCppProvider = class {
|
|
799
|
+
constructor(model, options = {}) {
|
|
800
|
+
this.model = model;
|
|
801
|
+
if (isLlamaSelector(model)) {
|
|
802
|
+
throw new InferenceError(
|
|
803
|
+
`llama-cpp model "${model}" is a selector. Constructing a provider directly needs a concrete model (e.g. "${aliasForTier("balanced")}") \u2014 use makeProviderAsync to resolve a selector against this machine.`
|
|
804
|
+
);
|
|
805
|
+
}
|
|
806
|
+
this.uri = resolveLlamaModelRef(model);
|
|
807
|
+
this.runtime = options.runtime ?? defaultLlamaRuntime();
|
|
808
|
+
this.thoughtTokens = options.thoughtTokens ?? 0;
|
|
809
|
+
this.maxTokens = options.maxTokens;
|
|
810
|
+
this.modelsDirectory = options.modelsDirectory ?? defaultLlamaModelsDirectory();
|
|
811
|
+
this.cacheKey = buildCacheKey([this.modelsDirectory, this.uri]);
|
|
812
|
+
}
|
|
813
|
+
model;
|
|
814
|
+
uri;
|
|
815
|
+
runtime;
|
|
816
|
+
thoughtTokens;
|
|
817
|
+
maxTokens;
|
|
818
|
+
modelsDirectory;
|
|
819
|
+
/**
|
|
820
|
+
* Loaded-model key: the same URI in two directories is two different files.
|
|
821
|
+
* Built with `buildCacheKey` so its parts are length-prefixed — a plain join
|
|
822
|
+
* would let two different (directory, uri) pairs collide and hand a provider
|
|
823
|
+
* back the wrong weights.
|
|
824
|
+
*/
|
|
825
|
+
cacheKey;
|
|
826
|
+
provider() {
|
|
827
|
+
return "llama-cpp";
|
|
828
|
+
}
|
|
829
|
+
modelName() {
|
|
830
|
+
return this.model;
|
|
831
|
+
}
|
|
832
|
+
async completeJSON(req) {
|
|
833
|
+
const model = await this.load();
|
|
834
|
+
const session = await model.createSession(systemPromptFor(req));
|
|
835
|
+
try {
|
|
836
|
+
const result = await session.prompt(req.user, {
|
|
837
|
+
schema: req.schema,
|
|
838
|
+
temperature: req.temperature,
|
|
839
|
+
thoughtTokens: this.thoughtTokens,
|
|
840
|
+
...this.maxTokens != null ? { maxTokens: this.maxTokens } : {}
|
|
841
|
+
});
|
|
842
|
+
if (result.stopReason === "maxTokens") {
|
|
843
|
+
throw new Error(
|
|
844
|
+
`llama-cpp generation hit the token limit before completing the JSON${this.maxTokens != null ? ` (maxTokens: ${this.maxTokens})` : ""} \u2014 raise llamaCpp.maxTokens, or shorten the prompt if the context is full.`
|
|
845
|
+
);
|
|
846
|
+
}
|
|
847
|
+
return { json: extractJson(result.text), usage: result.usage };
|
|
848
|
+
} finally {
|
|
849
|
+
await session.dispose().catch(() => void 0);
|
|
850
|
+
}
|
|
851
|
+
}
|
|
852
|
+
load() {
|
|
853
|
+
const existing = loadedModels.get(this.cacheKey);
|
|
854
|
+
if (existing) return existing;
|
|
855
|
+
const pending = (async () => {
|
|
856
|
+
const path = await this.runtime.resolveModelFile(
|
|
857
|
+
this.uri,
|
|
858
|
+
this.modelsDirectory
|
|
859
|
+
);
|
|
860
|
+
return this.runtime.loadModel(path);
|
|
861
|
+
})();
|
|
862
|
+
const guarded = pending.catch((e) => {
|
|
863
|
+
if (loadedModels.get(this.cacheKey) === guarded) {
|
|
864
|
+
loadedModels.delete(this.cacheKey);
|
|
865
|
+
}
|
|
866
|
+
throw e;
|
|
867
|
+
});
|
|
868
|
+
loadedModels.set(this.cacheKey, guarded);
|
|
869
|
+
return guarded;
|
|
870
|
+
}
|
|
871
|
+
};
|
|
872
|
+
function systemPromptFor(req) {
|
|
873
|
+
return `${req.system}
|
|
874
|
+
|
|
875
|
+
Respond with ONLY a JSON object conforming to this JSON Schema:
|
|
876
|
+
${JSON.stringify(
|
|
877
|
+
req.schema
|
|
878
|
+
)}`;
|
|
879
|
+
}
|
|
880
|
+
var runtimePromise;
|
|
881
|
+
function defaultLlamaRuntime() {
|
|
882
|
+
const real = () => (
|
|
883
|
+
// Drop a failed init so the next call retries. A GPU that failed to
|
|
884
|
+
// initialise, or a binary still being extracted by a concurrent install,
|
|
885
|
+
// must not poison the runtime for the rest of the process — the same rule
|
|
886
|
+
// `load()` applies to weights.
|
|
887
|
+
runtimePromise ??= loadNodeLlamaCpp().catch((e) => {
|
|
888
|
+
runtimePromise = void 0;
|
|
889
|
+
throw e;
|
|
890
|
+
})
|
|
891
|
+
);
|
|
892
|
+
return {
|
|
893
|
+
resolveModelFile: (uri, directory) => real().then((r) => r.resolveModelFile(uri, directory)),
|
|
894
|
+
loadModel: (path) => real().then((r) => r.loadModel(path)),
|
|
895
|
+
getMemoryBudgetBytes: () => real().then((r) => r.getMemoryBudgetBytes())
|
|
896
|
+
};
|
|
897
|
+
}
|
|
898
|
+
async function loadNodeLlamaCpp() {
|
|
899
|
+
let mod;
|
|
900
|
+
try {
|
|
901
|
+
mod = await import("node-llama-cpp");
|
|
902
|
+
} catch (e) {
|
|
903
|
+
if (!isModuleNotFound(e)) {
|
|
904
|
+
throw new InferenceError(
|
|
905
|
+
`node-llama-cpp is installed but failed to load (${e instanceof Error ? e.message : String(e)}). This is the copy resolved from your own node_modules, so reinstalling it here will not help \u2014 check the Node version and the platform build.`
|
|
906
|
+
);
|
|
907
|
+
}
|
|
908
|
+
mod = await importNodeLlamaCpp();
|
|
909
|
+
}
|
|
910
|
+
const { getLlama, resolveModelFile, LlamaChatSession, TokenMeter } = mod;
|
|
911
|
+
const llama = await getLlama();
|
|
912
|
+
return {
|
|
913
|
+
// `directory` is this library's own, not node-llama-cpp's global default —
|
|
914
|
+
// owning it is what makes `clearLlamaModels` safe.
|
|
915
|
+
resolveModelFile: (uri, directory) => resolveModelFile(uri, { directory }),
|
|
916
|
+
async loadModel(path) {
|
|
917
|
+
const model = await llama.loadModel({ modelPath: path });
|
|
918
|
+
return {
|
|
919
|
+
async createSession(systemPrompt) {
|
|
920
|
+
const context = await model.createContext();
|
|
921
|
+
const sequence = context.getSequence();
|
|
922
|
+
const session = new LlamaChatSession({
|
|
923
|
+
contextSequence: sequence,
|
|
924
|
+
systemPrompt
|
|
925
|
+
});
|
|
926
|
+
return {
|
|
927
|
+
async prompt(text, options) {
|
|
928
|
+
const grammar = await llama.createGrammarForJsonSchema(
|
|
929
|
+
options.schema
|
|
930
|
+
);
|
|
931
|
+
const before = sequence.tokenMeter.getState();
|
|
932
|
+
const result = await session.promptWithMeta(text, {
|
|
933
|
+
grammar,
|
|
934
|
+
temperature: options.temperature,
|
|
935
|
+
budgets: { thoughtTokens: options.thoughtTokens },
|
|
936
|
+
...options.maxTokens != null ? { maxTokens: options.maxTokens } : {}
|
|
937
|
+
});
|
|
938
|
+
const diff = TokenMeter.diff(sequence.tokenMeter, before);
|
|
939
|
+
return {
|
|
940
|
+
text: result.responseText,
|
|
941
|
+
stopReason: result.stopReason,
|
|
942
|
+
usage: {
|
|
943
|
+
inputTokens: diff.usedInputTokens,
|
|
944
|
+
outputTokens: diff.usedOutputTokens
|
|
945
|
+
}
|
|
946
|
+
};
|
|
947
|
+
},
|
|
948
|
+
async dispose() {
|
|
949
|
+
await context.dispose();
|
|
950
|
+
}
|
|
951
|
+
};
|
|
952
|
+
},
|
|
953
|
+
async dispose() {
|
|
954
|
+
await model.dispose();
|
|
955
|
+
}
|
|
956
|
+
};
|
|
957
|
+
},
|
|
958
|
+
async getMemoryBudgetBytes() {
|
|
959
|
+
const { totalmem } = await import("os");
|
|
960
|
+
const ramBudget = totalmem() / 2;
|
|
961
|
+
try {
|
|
962
|
+
const vram = await llama.getVramState();
|
|
963
|
+
return Math.max(vram.free, ramBudget);
|
|
964
|
+
} catch {
|
|
965
|
+
return ramBudget;
|
|
966
|
+
}
|
|
967
|
+
}
|
|
968
|
+
};
|
|
969
|
+
}
|
|
970
|
+
|
|
971
|
+
// src/providers/detect.ts
|
|
972
|
+
var DETECTION_ORDER = [
|
|
973
|
+
"anthropic",
|
|
974
|
+
"openai",
|
|
975
|
+
"claude-cli",
|
|
976
|
+
"llama-cpp"
|
|
977
|
+
];
|
|
978
|
+
var DEFAULT_KEY_ENV = {
|
|
979
|
+
anthropic: "ANTHROPIC_API_KEY",
|
|
980
|
+
openai: "OPENAI_API_KEY"
|
|
981
|
+
};
|
|
982
|
+
function hasKey(provider) {
|
|
983
|
+
return (process.env[DEFAULT_KEY_ENV[provider]] ?? "") !== "";
|
|
984
|
+
}
|
|
985
|
+
var cliProbes = /* @__PURE__ */ new Map();
|
|
986
|
+
function resetClaudeCliProbe() {
|
|
987
|
+
cliProbes.clear();
|
|
988
|
+
}
|
|
989
|
+
function probeClaudeCli(spec) {
|
|
990
|
+
const exec = spec.exec ?? realExec;
|
|
991
|
+
const command = spec.command ?? "claude";
|
|
992
|
+
const cached = cliProbes.get(command);
|
|
993
|
+
if (cached) return cached;
|
|
994
|
+
const probing = exec([command, "--version"], { timeoutMs: 1e4 }).then((r) => r.code === 0 && !r.timedOut && r.spawnError == null).catch(() => false);
|
|
995
|
+
cliProbes.set(command, probing);
|
|
996
|
+
return probing;
|
|
997
|
+
}
|
|
998
|
+
async function probeLlamaCpp(spec) {
|
|
999
|
+
const injected = spec.llamaRuntime ?? spec.llamaCpp?.runtime;
|
|
1000
|
+
if (injected) return budgetProbe(injected);
|
|
1001
|
+
const status = await nodeLlamaCppStatus();
|
|
1002
|
+
if (status.state === "refused") {
|
|
1003
|
+
return { available: false, reason: status.reason };
|
|
1004
|
+
}
|
|
1005
|
+
if (status.state === "installable") {
|
|
1006
|
+
return { available: true };
|
|
1007
|
+
}
|
|
1008
|
+
return budgetProbe(defaultLlamaRuntime());
|
|
1009
|
+
}
|
|
1010
|
+
function budgetProbe(runtime) {
|
|
1011
|
+
return runtime.getMemoryBudgetBytes().then(
|
|
1012
|
+
() => ({ available: true }),
|
|
1013
|
+
(e) => ({
|
|
1014
|
+
available: false,
|
|
1015
|
+
reason: e instanceof Error && /node-llama-cpp/.test(e.message) ? "node-llama-cpp is not installed (npm i node-llama-cpp)" : `node-llama-cpp could not start (${e instanceof Error ? e.message : String(e)})`
|
|
1016
|
+
})
|
|
1017
|
+
);
|
|
1018
|
+
}
|
|
1019
|
+
async function probe(provider, spec) {
|
|
1020
|
+
switch (provider) {
|
|
1021
|
+
case "anthropic":
|
|
1022
|
+
return hasKey("anthropic") ? { available: true } : {
|
|
1023
|
+
available: false,
|
|
1024
|
+
reason: `ANTHROPIC_API_KEY is not set`
|
|
1025
|
+
};
|
|
1026
|
+
case "openai":
|
|
1027
|
+
return hasKey("openai") || spec.baseUrl ? { available: true } : {
|
|
1028
|
+
available: false,
|
|
1029
|
+
reason: `OPENAI_API_KEY is not set and no baseUrl was given`
|
|
1030
|
+
};
|
|
1031
|
+
case "claude-cli":
|
|
1032
|
+
return await probeClaudeCli(spec) ? { available: true } : {
|
|
1033
|
+
available: false,
|
|
1034
|
+
reason: `could not run \`${spec.command ?? "claude"}\` (is the Claude CLI installed?)`
|
|
1035
|
+
};
|
|
1036
|
+
case "llama-cpp":
|
|
1037
|
+
return probeLlamaCpp(spec);
|
|
1038
|
+
default:
|
|
1039
|
+
return { available: false, reason: "not auto-selectable" };
|
|
1040
|
+
}
|
|
1041
|
+
}
|
|
1042
|
+
async function availableProviders(spec = {}) {
|
|
1043
|
+
const probes = await Promise.all(
|
|
1044
|
+
DETECTION_ORDER.map((name) => probe(name, spec))
|
|
1045
|
+
);
|
|
1046
|
+
return DETECTION_ORDER.filter((_, i) => probes[i].available);
|
|
1047
|
+
}
|
|
1048
|
+
async function detectProvider(spec = {}) {
|
|
1049
|
+
const reasons = [];
|
|
1050
|
+
for (const name of DETECTION_ORDER) {
|
|
1051
|
+
const result = await probe(name, spec);
|
|
1052
|
+
if (result.available) {
|
|
1053
|
+
warnSelected(name, spec.provider === "auto");
|
|
1054
|
+
return name;
|
|
1055
|
+
}
|
|
1056
|
+
reasons.push(` ${name.padEnd(10)} \u2014 ${result.reason}`);
|
|
1057
|
+
}
|
|
1058
|
+
throw new InferenceError(
|
|
1059
|
+
`No inference provider is available. Tried:
|
|
1060
|
+
${reasons.join("\n")}
|
|
1061
|
+
Pass an explicit \`provider\`, set one of the keys above, or install node-llama-cpp.`
|
|
1062
|
+
);
|
|
1063
|
+
}
|
|
1064
|
+
var warnedSelection = false;
|
|
1065
|
+
var warnedDownload = false;
|
|
1066
|
+
function resetProviderDetectionWarning() {
|
|
1067
|
+
warnedSelection = false;
|
|
1068
|
+
warnedDownload = false;
|
|
1069
|
+
}
|
|
1070
|
+
function warnSelected(provider, wasExplicitAuto) {
|
|
1071
|
+
if (warnedSelection) return;
|
|
1072
|
+
warnedSelection = true;
|
|
1073
|
+
const because = wasExplicitAuto ? `provider "auto"` : "no provider specified";
|
|
1074
|
+
console.warn(
|
|
1075
|
+
`inference: ${because} \u2014 auto-selected "${provider}". Pass an explicit \`provider\` to pin it.`
|
|
1076
|
+
);
|
|
1077
|
+
}
|
|
1078
|
+
function warnPendingDownload(model, sizeBytes) {
|
|
1079
|
+
if (warnedDownload) return;
|
|
1080
|
+
warnedDownload = true;
|
|
1081
|
+
console.warn(
|
|
1082
|
+
`inference: "${model}" is not downloaded yet \u2014 the first run will fetch ~${(sizeBytes / 1e9).toFixed(2)} GB. Pre-fetch it, or pass an explicit \`provider\` to avoid the local model entirely.`
|
|
1083
|
+
);
|
|
1084
|
+
}
|
|
1085
|
+
|
|
387
1086
|
// src/providers/index.ts
|
|
388
1087
|
var DEFAULT_MODELS = {
|
|
389
1088
|
anthropic: "claude-sonnet-4-5",
|
|
390
1089
|
openai: "gpt-4o-mini",
|
|
391
1090
|
"claude-cli": "claude-sonnet-4-5",
|
|
392
|
-
mock: "mock-model"
|
|
1091
|
+
mock: "mock-model",
|
|
1092
|
+
// A selector, not a pinned model: which weights a tier points at is then a
|
|
1093
|
+
// catalog change rather than an API change. Resolving it needs the async
|
|
1094
|
+
// factory — see `resolveProviderIdentityAsync`.
|
|
1095
|
+
"llama-cpp": "auto"
|
|
393
1096
|
};
|
|
394
1097
|
var DEFAULT_API_KEY_ENV = {
|
|
395
1098
|
anthropic: "ANTHROPIC_API_KEY",
|
|
@@ -397,12 +1100,51 @@ var DEFAULT_API_KEY_ENV = {
|
|
|
397
1100
|
};
|
|
398
1101
|
var DEFAULT_OPENAI_BASE_URL = "https://api.openai.com/v1";
|
|
399
1102
|
function resolveProviderIdentity(spec) {
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
1103
|
+
if (spec.provider == null || spec.provider === "auto") {
|
|
1104
|
+
throw new InferenceError(
|
|
1105
|
+
`No provider specified. Detecting one probes the environment, the Claude CLI and the local model runtime, which cannot be done synchronously \u2014 use resolveProviderIdentityAsync/makeProviderAsync, or name a provider (${Object.keys(DEFAULT_MODELS).join(", ")}).`
|
|
1106
|
+
);
|
|
1107
|
+
}
|
|
1108
|
+
const model = spec.model ?? DEFAULT_MODELS[spec.provider] ?? "unknown";
|
|
1109
|
+
if (spec.provider === "llama-cpp" && isLlamaSelector(model)) {
|
|
1110
|
+
throw new InferenceError(
|
|
1111
|
+
`llama-cpp model "${model}" is a selector and cannot be resolved synchronously \u2014 picking a tier probes GPU memory. Use resolveProviderIdentityAsync/makeProviderAsync, or name a concrete model (e.g. "${aliasForTier("balanced")}").`
|
|
1112
|
+
);
|
|
1113
|
+
}
|
|
1114
|
+
return { provider: spec.provider, model };
|
|
1115
|
+
}
|
|
1116
|
+
async function resolveProviderIdentityAsync(spec) {
|
|
1117
|
+
if ((spec.provider == null || spec.provider === "auto") && spec.model != null) {
|
|
1118
|
+
throw new InferenceError(
|
|
1119
|
+
`Model "${spec.model}" was given without a provider, and a model name does not say which provider owns it. Name the provider too (${Object.keys(DEFAULT_MODELS).join(", ")}), or drop the model to take the detected provider's default.`
|
|
1120
|
+
);
|
|
1121
|
+
}
|
|
1122
|
+
const provider = spec.provider == null || spec.provider === "auto" ? await detectProvider(spec) : spec.provider;
|
|
1123
|
+
const resolved = { ...spec, provider };
|
|
1124
|
+
const model = spec.model ?? DEFAULT_MODELS[provider] ?? "unknown";
|
|
1125
|
+
if (provider !== "llama-cpp" || !isLlamaSelector(model)) {
|
|
1126
|
+
return resolveProviderIdentity(resolved);
|
|
1127
|
+
}
|
|
1128
|
+
const tier = model === "auto" ? await probeTier(llamaRuntimeFor(spec)) : model;
|
|
1129
|
+
return { provider, model: aliasForTier(tier) };
|
|
1130
|
+
}
|
|
1131
|
+
function warnIfDownloadPending(spec, model) {
|
|
1132
|
+
const entry = LLAMA_MODELS[model];
|
|
1133
|
+
if (!entry) return;
|
|
1134
|
+
const directory = spec.llamaCpp?.modelsDirectory ?? defaultLlamaModelsDirectory();
|
|
1135
|
+
if (!isModelDownloaded(model, directory)) {
|
|
1136
|
+
warnPendingDownload(model, entry.sizeBytes);
|
|
1137
|
+
}
|
|
1138
|
+
}
|
|
1139
|
+
function llamaRuntimeFor(spec) {
|
|
1140
|
+
return spec.llamaRuntime ?? spec.llamaCpp?.runtime;
|
|
1141
|
+
}
|
|
1142
|
+
async function probeTier(runtime) {
|
|
1143
|
+
const source = runtime ?? defaultLlamaRuntime();
|
|
1144
|
+
return tierForBudget(await source.getMemoryBudgetBytes());
|
|
404
1145
|
}
|
|
405
1146
|
function makeProvider(spec) {
|
|
1147
|
+
warnIfUnsupportedNode();
|
|
406
1148
|
const { model } = resolveProviderIdentity(spec);
|
|
407
1149
|
switch (spec.provider) {
|
|
408
1150
|
case "anthropic":
|
|
@@ -428,6 +1170,11 @@ function makeProvider(spec) {
|
|
|
428
1170
|
);
|
|
429
1171
|
case "mock":
|
|
430
1172
|
return new MockProvider(spec.mockResponses ?? [{ json: {} }], model);
|
|
1173
|
+
case "llama-cpp":
|
|
1174
|
+
return new LlamaCppProvider(model, {
|
|
1175
|
+
...spec.llamaCpp ?? {},
|
|
1176
|
+
...spec.llamaRuntime ? { runtime: spec.llamaRuntime } : {}
|
|
1177
|
+
});
|
|
431
1178
|
default:
|
|
432
1179
|
throw new InferenceError(
|
|
433
1180
|
`Unknown provider "${String(spec.provider)}". Available: ${Object.keys(
|
|
@@ -436,6 +1183,55 @@ function makeProvider(spec) {
|
|
|
436
1183
|
);
|
|
437
1184
|
}
|
|
438
1185
|
}
|
|
1186
|
+
async function makeProviderAsync(spec) {
|
|
1187
|
+
const { provider, model } = await resolveProviderIdentityAsync(spec);
|
|
1188
|
+
if (provider === "llama-cpp") warnIfDownloadPending(spec, model);
|
|
1189
|
+
return makeProvider({ ...spec, provider, model });
|
|
1190
|
+
}
|
|
1191
|
+
|
|
1192
|
+
// src/providers/llama-clean.ts
|
|
1193
|
+
import { rmSync as rmSync2, statSync as statSync2 } from "fs";
|
|
1194
|
+
import { join as join4 } from "path";
|
|
1195
|
+
async function clearLlamaModels(options = {}) {
|
|
1196
|
+
const directory = options.directory ?? defaultLlamaModelsDirectory();
|
|
1197
|
+
const dryRun = options.dryRun ?? false;
|
|
1198
|
+
const wanted = options.models?.map(blobNameFor);
|
|
1199
|
+
if (!dryRun) await disposeLlamaModels();
|
|
1200
|
+
const files = [];
|
|
1201
|
+
for (const entry of listModelDirectory(directory)) {
|
|
1202
|
+
if (!isModelBlob(entry)) continue;
|
|
1203
|
+
if (wanted && !wanted.some((name) => matchesModelBlob(entry, name))) {
|
|
1204
|
+
continue;
|
|
1205
|
+
}
|
|
1206
|
+
const path = join4(directory, entry);
|
|
1207
|
+
const sizeBytes = sizeOf(path);
|
|
1208
|
+
if (sizeBytes === void 0) continue;
|
|
1209
|
+
if (!dryRun) {
|
|
1210
|
+
try {
|
|
1211
|
+
rmSync2(path);
|
|
1212
|
+
} catch {
|
|
1213
|
+
continue;
|
|
1214
|
+
}
|
|
1215
|
+
}
|
|
1216
|
+
files.push({ path, sizeBytes });
|
|
1217
|
+
}
|
|
1218
|
+
return {
|
|
1219
|
+
files,
|
|
1220
|
+
freedBytes: files.reduce((total, file) => total + file.sizeBytes, 0),
|
|
1221
|
+
directory,
|
|
1222
|
+
dryRun
|
|
1223
|
+
};
|
|
1224
|
+
}
|
|
1225
|
+
function isModelBlob(entry) {
|
|
1226
|
+
return entry.endsWith(".gguf") || entry.endsWith(".gguf.ipull");
|
|
1227
|
+
}
|
|
1228
|
+
function sizeOf(path) {
|
|
1229
|
+
try {
|
|
1230
|
+
return statSync2(path).size;
|
|
1231
|
+
} catch {
|
|
1232
|
+
return void 0;
|
|
1233
|
+
}
|
|
1234
|
+
}
|
|
439
1235
|
|
|
440
1236
|
// src/complete.ts
|
|
441
1237
|
import { Ajv2020 } from "ajv/dist/2020.js";
|
|
@@ -448,6 +1244,7 @@ function validatorFor(schema) {
|
|
|
448
1244
|
return compiled;
|
|
449
1245
|
}
|
|
450
1246
|
async function completeValidatedJSON(options) {
|
|
1247
|
+
warnIfUnsupportedNode();
|
|
451
1248
|
const {
|
|
452
1249
|
provider,
|
|
453
1250
|
system,
|
|
@@ -488,56 +1285,6 @@ async function completeValidatedJSON(options) {
|
|
|
488
1285
|
return { ...base, error: lastError, durationMs: Date.now() - start };
|
|
489
1286
|
}
|
|
490
1287
|
|
|
491
|
-
// src/cache.ts
|
|
492
|
-
import { createHash } from "crypto";
|
|
493
|
-
import { existsSync, mkdirSync, readFileSync, writeFileSync } from "fs";
|
|
494
|
-
import { join } from "path";
|
|
495
|
-
function sha256(text) {
|
|
496
|
-
return createHash("sha256").update(text, "utf8").digest("hex");
|
|
497
|
-
}
|
|
498
|
-
function buildCacheKey(parts) {
|
|
499
|
-
return sha256(parts.map((p) => `${p.length}:${p}`).join("|"));
|
|
500
|
-
}
|
|
501
|
-
var JsonCache = class {
|
|
502
|
-
constructor(dir, enabled = true, label = "inference") {
|
|
503
|
-
this.dir = dir;
|
|
504
|
-
this.enabled = enabled;
|
|
505
|
-
this.label = label;
|
|
506
|
-
}
|
|
507
|
-
dir;
|
|
508
|
-
enabled;
|
|
509
|
-
label;
|
|
510
|
-
/** Cache-write failures warn once per process, not once per entry. */
|
|
511
|
-
warned = false;
|
|
512
|
-
get(key) {
|
|
513
|
-
if (!this.enabled) return void 0;
|
|
514
|
-
const path = join(this.dir, `${key}.json`);
|
|
515
|
-
if (!existsSync(path)) return void 0;
|
|
516
|
-
try {
|
|
517
|
-
return JSON.parse(readFileSync(path, "utf8"));
|
|
518
|
-
} catch {
|
|
519
|
-
return void 0;
|
|
520
|
-
}
|
|
521
|
-
}
|
|
522
|
-
set(key, value) {
|
|
523
|
-
if (!this.enabled) return;
|
|
524
|
-
try {
|
|
525
|
-
mkdirSync(this.dir, { recursive: true });
|
|
526
|
-
writeFileSync(
|
|
527
|
-
join(this.dir, `${key}.json`),
|
|
528
|
-
JSON.stringify(value, null, 2)
|
|
529
|
-
);
|
|
530
|
-
} catch (e) {
|
|
531
|
-
if (!this.warned) {
|
|
532
|
-
this.warned = true;
|
|
533
|
-
console.warn(
|
|
534
|
-
`${this.label}: could not write the cache at ${this.dir} (${e instanceof Error ? e.message : String(e)}). Continuing without caching.`
|
|
535
|
-
);
|
|
536
|
-
}
|
|
537
|
-
}
|
|
538
|
-
}
|
|
539
|
-
};
|
|
540
|
-
|
|
541
1288
|
// src/cost.ts
|
|
542
1289
|
var PRICE_TABLE = {
|
|
543
1290
|
"claude-sonnet-4-5": { inputPerMTok: 3, outputPerMTok: 15 },
|
|
@@ -693,29 +1440,56 @@ export {
|
|
|
693
1440
|
DEFAULT_MODELS,
|
|
694
1441
|
DEFAULT_OPENAI_BASE_URL,
|
|
695
1442
|
DEFAULT_ZONES,
|
|
1443
|
+
DETECTION_ORDER,
|
|
696
1444
|
InferenceError,
|
|
697
1445
|
JsonCache,
|
|
1446
|
+
LLAMA_MODELS,
|
|
1447
|
+
LLAMA_SELECTORS,
|
|
1448
|
+
LLAMA_TIERS,
|
|
1449
|
+
LlamaCppProvider,
|
|
698
1450
|
MockProvider,
|
|
699
1451
|
OpenAICompatProvider,
|
|
700
1452
|
PRICE_TABLE,
|
|
701
1453
|
VERDICT_SCHEMA,
|
|
1454
|
+
aliasForTier,
|
|
1455
|
+
availableProviders,
|
|
1456
|
+
blobNameFor,
|
|
702
1457
|
buildCacheKey,
|
|
1458
|
+
clearLlamaModels,
|
|
703
1459
|
completeValidatedJSON,
|
|
704
1460
|
computeConsensus,
|
|
705
1461
|
costOfRuns,
|
|
706
1462
|
costOfUsage,
|
|
1463
|
+
defaultLlamaModelsDirectory,
|
|
1464
|
+
defaultLlamaRuntime,
|
|
1465
|
+
defaultLlamaRuntimeDirectory,
|
|
1466
|
+
detectProvider,
|
|
1467
|
+
disposeLlamaModels,
|
|
707
1468
|
extractJson,
|
|
1469
|
+
importNodeLlamaCpp,
|
|
1470
|
+
isLlamaSelector,
|
|
1471
|
+
isModelDownloaded,
|
|
708
1472
|
judge,
|
|
709
1473
|
makeProvider,
|
|
1474
|
+
makeProviderAsync,
|
|
710
1475
|
mockVerdict,
|
|
1476
|
+
nodeLlamaCppStatus,
|
|
711
1477
|
pricingFor,
|
|
712
1478
|
realExec,
|
|
1479
|
+
resetClaudeCliProbe,
|
|
1480
|
+
resetNodeVersionWarning,
|
|
1481
|
+
resetProviderDetectionWarning,
|
|
1482
|
+
resetRuntimeInstall,
|
|
713
1483
|
resetTemperatureWarning,
|
|
1484
|
+
resolveLlamaModelRef,
|
|
714
1485
|
resolveProviderIdentity,
|
|
1486
|
+
resolveProviderIdentityAsync,
|
|
715
1487
|
runEnsemble,
|
|
716
1488
|
sha256,
|
|
717
1489
|
stripNulls,
|
|
1490
|
+
tierForBudget,
|
|
718
1491
|
toStrictSchema,
|
|
1492
|
+
uriForTier,
|
|
719
1493
|
validatorFor,
|
|
720
1494
|
zoneFor
|
|
721
1495
|
};
|