@hawkeyexl/inference 0.1.0 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -6,6 +6,22 @@ var InferenceError = class extends Error {
6
6
  }
7
7
  };
8
8
 
9
+ // src/runtime.ts
10
+ var MINIMUM_NODE_MAJOR = 24;
11
+ var warnedNodeVersion = false;
12
+ function warnIfUnsupportedNode(version = process.versions.node) {
13
+ if (warnedNodeVersion) return;
14
+ const major = Number.parseInt(version, 10);
15
+ if (!Number.isInteger(major) || major >= MINIMUM_NODE_MAJOR) return;
16
+ warnedNodeVersion = true;
17
+ console.warn(
18
+ `inference: running on Node ${version}, older than the Node ${MINIMUM_NODE_MAJOR} this package requires. npm only warns about that at install time (EBADENGINE), so nothing has stopped you yet \u2014 upgrade Node, or expect failures this library cannot explain.`
19
+ );
20
+ }
21
+ function resetNodeVersionWarning() {
22
+ warnedNodeVersion = false;
23
+ }
24
+
9
25
  // src/providers/anthropic.ts
10
26
  import Anthropic from "@anthropic-ai/sdk";
11
27
  var DEFAULT_TOOL_NAME = "record_result";
@@ -331,7 +347,15 @@ var ClaudeCliProvider = class {
331
347
  `Claude CLI exited ${result.code}: ${result.stderr.trim().slice(-300)}`
332
348
  );
333
349
  }
334
- const wrapper = JSON.parse(result.stdout);
350
+ let wrapper;
351
+ try {
352
+ wrapper = JSON.parse(result.stdout);
353
+ } catch {
354
+ const excerpt = result.stdout.trim().replace(/\s+/g, " ").slice(0, 200);
355
+ throw new Error(
356
+ `Claude CLI printed non-JSON output (is it logged in?): ${excerpt || "(no output)"}`
357
+ );
358
+ }
335
359
  if (typeof wrapper.result !== "string") {
336
360
  throw new Error("Claude CLI returned no result field");
337
361
  }
@@ -384,12 +408,691 @@ function mockVerdict(match, confidence, overrides = {}) {
384
408
  };
385
409
  }
386
410
 
411
+ // src/cache.ts
412
+ import { createHash } from "crypto";
413
+ import { existsSync, mkdirSync, readFileSync, writeFileSync } from "fs";
414
+ import { join } from "path";
415
+ function sha256(text) {
416
+ return createHash("sha256").update(text, "utf8").digest("hex");
417
+ }
418
+ function buildCacheKey(parts) {
419
+ return sha256(parts.map((p) => `${p.length}:${p}`).join("|"));
420
+ }
421
+ var JsonCache = class {
422
+ constructor(dir, enabled = true, label = "inference") {
423
+ this.dir = dir;
424
+ this.enabled = enabled;
425
+ this.label = label;
426
+ }
427
+ dir;
428
+ enabled;
429
+ label;
430
+ /** Cache-write failures warn once per process, not once per entry. */
431
+ warned = false;
432
+ get(key) {
433
+ if (!this.enabled) return void 0;
434
+ const path = join(this.dir, `${key}.json`);
435
+ if (!existsSync(path)) return void 0;
436
+ try {
437
+ return JSON.parse(readFileSync(path, "utf8"));
438
+ } catch {
439
+ return void 0;
440
+ }
441
+ }
442
+ set(key, value) {
443
+ if (!this.enabled) return;
444
+ try {
445
+ mkdirSync(this.dir, { recursive: true });
446
+ writeFileSync(
447
+ join(this.dir, `${key}.json`),
448
+ JSON.stringify(value, null, 2)
449
+ );
450
+ } catch (e) {
451
+ if (!this.warned) {
452
+ this.warned = true;
453
+ console.warn(
454
+ `${this.label}: could not write the cache at ${this.dir} (${e instanceof Error ? e.message : String(e)}). Continuing without caching.`
455
+ );
456
+ }
457
+ }
458
+ }
459
+ };
460
+
461
+ // src/providers/llama-models.ts
462
+ import { readdirSync } from "fs";
463
+ import { homedir } from "os";
464
+ import { join as join2 } from "path";
465
+ function defaultLlamaModelsDirectory() {
466
+ return process.env["INFERENCE_MODELS_DIR"] || join2(homedir(), ".hawkeyexl-inference", "models");
467
+ }
468
+ var LLAMA_TIERS = ["fast", "balanced", "quality"];
469
+ var LLAMA_SELECTORS = ["auto", ...LLAMA_TIERS];
470
+ var LLAMA_MODELS = deepFreezeEntries({
471
+ "granite-4.1-3b-q2": {
472
+ uri: "hf:unsloth/granite-4.1-3b-GGUF/granite-4.1-3b-UD-Q2_K_XL.gguf",
473
+ sizeBytes: 1414548800,
474
+ license: "Apache-2.0",
475
+ tier: "fast",
476
+ notes: "Smallest tier and the quickest measured (4.8s/page). Scores level with models three times its size on schema-constrained extraction."
477
+ },
478
+ "qwen3.5-4b": {
479
+ uri: "hf:unsloth/Qwen3.5-4B-GGUF/Qwen3.5-4B-UD-Q4_K_XL.gguf",
480
+ sizeBytes: 2912109728,
481
+ license: "Apache-2.0",
482
+ notes: "The default for most machines. Smaller and faster than the Gemma 4 E4B it replaces, at indistinguishable measured quality.",
483
+ tier: "balanced"
484
+ },
485
+ "qwen3.5-9b": {
486
+ uri: "hf:unsloth/Qwen3.5-9B-GGUF/Qwen3.5-9B-UD-Q4_K_XL.gguf",
487
+ sizeBytes: 5966095584,
488
+ license: "Apache-2.0",
489
+ tier: "quality",
490
+ notes: "Best measured of everything tried, and still smaller and faster than the Gemma 4 12B it replaces. Wants a GPU or plenty of RAM."
491
+ },
492
+ // --- Superseded, kept resolvable by name. Untiered: nothing selects these
493
+ // unless a caller asks for one outright.
494
+ "gemma-4-e2b": {
495
+ uri: "hf:unsloth/gemma-4-E2B-it-qat-GGUF/gemma-4-E2B-it-qat-UD-Q4_K_XL.gguf",
496
+ sizeBytes: 2620370976,
497
+ license: "Apache-2.0",
498
+ notes: "IFEval 94.6. Former `fast` tier; sound, but larger than Granite."
499
+ },
500
+ "gemma-4-e4b": {
501
+ uri: "hf:unsloth/gemma-4-E4B-it-qat-GGUF/gemma-4-E4B-it-qat-UD-Q4_K_XL.gguf",
502
+ sizeBytes: 4215695776,
503
+ license: "Apache-2.0",
504
+ notes: "IFEval 96.7. Former `balanced` tier; sound, but larger."
505
+ },
506
+ "gemma-4-12b": {
507
+ uri: "hf:unsloth/gemma-4-12B-it-qat-GGUF/gemma-4-12B-it-qat-UD-Q4_K_XL.gguf",
508
+ sizeBytes: 6716356800,
509
+ license: "Apache-2.0",
510
+ notes: "IFEval 97.2. Former `quality` tier. Dense 12B; wants a GPU."
511
+ },
512
+ "gemma-4-26b-a4b": {
513
+ uri: "hf:unsloth/gemma-4-26B-A4B-it-qat-GGUF/gemma-4-26B-A4B-it-qat-UD-Q4_K_XL.gguf",
514
+ sizeBytes: 14249047104,
515
+ license: "Apache-2.0",
516
+ notes: "MoE: 25.2B total, 3.8B active \u2014 infers near E4B speed if it fits in memory."
517
+ },
518
+ "gemma-4-e2b-q2": {
519
+ uri: "hf:unsloth/gemma-4-E2B-it-qat-GGUF/gemma-4-E2B-it-qat-UD-Q2_K_XL.gguf",
520
+ sizeBytes: 2186186784,
521
+ license: "Apache-2.0",
522
+ notes: "AVOID. Smallest download, but it does not reliably terminate: 6 of 12 pages unfinished at 120s, one still running at 400s, and the pages that did finish proposed identifiers as prose. Kept only so existing pins still resolve."
523
+ }
524
+ });
525
+ function deepFreezeEntries(catalog) {
526
+ for (const entry of Object.values(catalog)) Object.freeze(entry);
527
+ return Object.freeze(catalog);
528
+ }
529
+ var TIER_ALIAS = {
530
+ fast: "granite-4.1-3b-q2",
531
+ balanced: "qwen3.5-4b",
532
+ quality: "qwen3.5-9b"
533
+ };
534
+ function isLlamaSelector(model) {
535
+ return LLAMA_SELECTORS.includes(model);
536
+ }
537
+ var MEMORY_HEADROOM = 3.5;
538
+ function tierForBudget(budgetBytes) {
539
+ let chosen = "fast";
540
+ for (const tier of LLAMA_TIERS) {
541
+ const entry = LLAMA_MODELS[TIER_ALIAS[tier]];
542
+ if (entry.sizeBytes * MEMORY_HEADROOM <= budgetBytes) chosen = tier;
543
+ }
544
+ return chosen;
545
+ }
546
+ function aliasForTier(tier) {
547
+ return TIER_ALIAS[tier];
548
+ }
549
+ function uriForTier(tier) {
550
+ return LLAMA_MODELS[TIER_ALIAS[tier]].uri;
551
+ }
552
+ function resolveLlamaModelRef(model) {
553
+ if (isLlamaSelector(model)) {
554
+ throw new InferenceError(
555
+ `llama-cpp model "${model}" is a selector and needs a hardware probe to resolve. Use resolveProviderIdentityAsync/makeProviderAsync, or name a concrete model (e.g. "${TIER_ALIAS.balanced}").`
556
+ );
557
+ }
558
+ const entry = LLAMA_MODELS[model];
559
+ if (entry) return entry.uri;
560
+ if (isModelPathOrUri(model)) return model;
561
+ throw new InferenceError(
562
+ `Unknown llama-cpp model "${model}". Use a selector (${LLAMA_SELECTORS.join(
563
+ ", "
564
+ )}), a curated alias (${Object.keys(LLAMA_MODELS).join(
565
+ ", "
566
+ )}), an hf: URI, or a path to a .gguf file.`
567
+ );
568
+ }
569
+ function blobNameFor(model) {
570
+ const ref = resolveLlamaModelRef(model).split("#")[0];
571
+ return ref.split(/[/\\]/).pop();
572
+ }
573
+ function matchesModelBlob(entry, blobName) {
574
+ const base = entry.replace(/\.ipull$/, "");
575
+ if (base.endsWith(blobName)) return true;
576
+ const stem = blobName.replace(/\.gguf$/, "");
577
+ return new RegExp(`${escapeRegExp(stem)}-\\d{5}-of-\\d{5}\\.gguf$`).test(base);
578
+ }
579
+ function escapeRegExp(text) {
580
+ return text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
581
+ }
582
+ function isModelDownloaded(model, directory) {
583
+ const blobName = blobNameFor(model);
584
+ return listModelDirectory(directory).some(
585
+ (entry) => !entry.endsWith(".ipull") && matchesModelBlob(entry, blobName)
586
+ );
587
+ }
588
+ function listModelDirectory(directory) {
589
+ try {
590
+ return readdirSync(directory, { withFileTypes: true }).filter((entry) => entry.isFile()).map((entry) => entry.name);
591
+ } catch {
592
+ return [];
593
+ }
594
+ }
595
+ function isModelPathOrUri(model) {
596
+ return /^(hf|huggingface):/i.test(model) || /^https?:\/\//i.test(model) || /^(hf|huggingface)\.co\//i.test(model) || model.endsWith(".gguf");
597
+ }
598
+
599
+ // src/providers/llama-install.ts
600
+ import {
601
+ existsSync as existsSync2,
602
+ mkdirSync as mkdirSync2,
603
+ rmSync,
604
+ statSync,
605
+ writeFileSync as writeFileSync2
606
+ } from "fs";
607
+ import { homedir as homedir2 } from "os";
608
+ import { join as join3 } from "path";
609
+ import { pathToFileURL } from "url";
610
+ var PACKAGE_SPEC = "node-llama-cpp@^3.19.0";
611
+ var SHIM = "loader.mjs";
612
+ var LOCK = ".install.lock";
613
+ var INSTALL_TIMEOUT_MS = 9e5;
614
+ var LOCK_WAIT_MS = INSTALL_TIMEOUT_MS;
615
+ var LOCK_STALE_MS = INSTALL_TIMEOUT_MS + 6e4;
616
+ function defaultLlamaRuntimeDirectory(env = process.env) {
617
+ return env["INFERENCE_RUNTIME_DIR"] || join3(homedir2(), ".hawkeyexl-inference", "runtime");
618
+ }
619
+ function isModuleNotFound(e) {
620
+ const code = e?.code;
621
+ return code === "ERR_MODULE_NOT_FOUND" || code === "MODULE_NOT_FOUND";
622
+ }
623
+ function describe(e) {
624
+ return e instanceof Error ? e.message : String(e);
625
+ }
626
+ async function nodeLlamaCppStatus(options = {}) {
627
+ const env = options.env ?? process.env;
628
+ const directory = options.directory ?? defaultLlamaRuntimeDirectory(env);
629
+ const probeImport = options.probeImport ?? (() => import("node-llama-cpp"));
630
+ try {
631
+ await probeImport();
632
+ return { state: "present" };
633
+ } catch (e) {
634
+ if (!isModuleNotFound(e)) {
635
+ return {
636
+ state: "refused",
637
+ reason: `node-llama-cpp is installed but failed to load (${describe(e)})`
638
+ };
639
+ }
640
+ }
641
+ if (existsSync2(join3(directory, SHIM))) return { state: "present" };
642
+ if ((env["INFERENCE_NO_AUTO_INSTALL"] ?? "") !== "") {
643
+ return {
644
+ state: "refused",
645
+ reason: `node-llama-cpp is not installed and INFERENCE_NO_AUTO_INSTALL is set`
646
+ };
647
+ }
648
+ return { state: "installable", directory };
649
+ }
650
+ var installs = /* @__PURE__ */ new Map();
651
+ var warnedInstall = false;
652
+ function resetRuntimeInstall() {
653
+ installs.clear();
654
+ warnedInstall = false;
655
+ }
656
+ function importNodeLlamaCpp(options = {}) {
657
+ const env = options.env ?? process.env;
658
+ const directory = options.directory ?? defaultLlamaRuntimeDirectory(env);
659
+ const existing = installs.get(directory);
660
+ if (existing) return existing;
661
+ const pending = fromPrefix(directory, env, options);
662
+ const guarded = pending.catch((e) => {
663
+ if (installs.get(directory) === guarded) installs.delete(directory);
664
+ throw e;
665
+ });
666
+ installs.set(directory, guarded);
667
+ return guarded;
668
+ }
669
+ async function fromPrefix(directory, env, options) {
670
+ const importShim = options.importShim ?? ((url) => import(url));
671
+ const shim = join3(directory, SHIM);
672
+ if (existsSync2(shim)) return importShim(pathToFileURL(shim).href);
673
+ if ((env["INFERENCE_NO_AUTO_INSTALL"] ?? "") !== "") {
674
+ throw new InferenceError(
675
+ `The llama-cpp provider needs node-llama-cpp, and INFERENCE_NO_AUTO_INSTALL is set. Install it yourself (npm i ${PACKAGE_SPEC}), unset INFERENCE_NO_AUTO_INSTALL to allow installing into ${directory}, or name a different provider.`
676
+ );
677
+ }
678
+ mkdirSync2(directory, { recursive: true });
679
+ await withLock(directory, async () => {
680
+ if (existsSync2(shim)) return;
681
+ warnInstalling(directory);
682
+ await runInstall(directory, env, options);
683
+ writeFileSync2(shim, `export * from "node-llama-cpp";
684
+ `, "utf8");
685
+ });
686
+ return importShim(pathToFileURL(shim).href);
687
+ }
688
+ async function runInstall(directory, env, options) {
689
+ const manifest = join3(directory, "package.json");
690
+ if (!existsSync2(manifest)) {
691
+ writeFileSync2(
692
+ manifest,
693
+ `${JSON.stringify(
694
+ {
695
+ name: "hawkeyexl-inference-runtime",
696
+ version: "0.0.0",
697
+ private: true,
698
+ description: "Auto-installed runtime for @hawkeyexl/inference. Safe to delete."
699
+ },
700
+ null,
701
+ 2
702
+ )}
703
+ `,
704
+ "utf8"
705
+ );
706
+ }
707
+ const exec = options.exec ?? realExec;
708
+ const result = await exec(
709
+ [
710
+ "npm",
711
+ "install",
712
+ "--prefix",
713
+ directory,
714
+ PACKAGE_SPEC,
715
+ "--no-audit",
716
+ "--no-fund"
717
+ ],
718
+ { timeoutMs: options.timeoutMs ?? INSTALL_TIMEOUT_MS, env }
719
+ );
720
+ if (result.spawnError != null) {
721
+ throw new InferenceError(
722
+ `Could not run npm to install node-llama-cpp (${result.spawnError}). Install it yourself with: npm i ${PACKAGE_SPEC}, or name a provider that does not need it.`
723
+ );
724
+ }
725
+ if (result.timedOut) {
726
+ throw new InferenceError(
727
+ `Installing node-llama-cpp into ${directory} timed out. A source build can take a while \u2014 retry, raise the timeout, or install it yourself with: npm i ${PACKAGE_SPEC}.`
728
+ );
729
+ }
730
+ if (result.code !== 0) {
731
+ throw new InferenceError(
732
+ `Installing node-llama-cpp into ${directory} failed (exit ${String(
733
+ result.code
734
+ )}).
735
+ ${tail(result.stderr || result.stdout)}
736
+ Install it yourself with: npm i ${PACKAGE_SPEC}, or name a provider that does not need it.`
737
+ );
738
+ }
739
+ }
740
+ function tail(output, lines = 12) {
741
+ return output.trimEnd().split(/\r?\n/).slice(-lines).join("\n");
742
+ }
743
+ function warnInstalling(directory) {
744
+ if (warnedInstall) return;
745
+ warnedInstall = true;
746
+ console.warn(
747
+ `inference: node-llama-cpp is not installed \u2014 fetching it into ${directory} so the local model can run. This is a one-time native install; set INFERENCE_NO_AUTO_INSTALL=1 to refuse it, or name a provider that does not need it.`
748
+ );
749
+ }
750
+ async function withLock(directory, fn) {
751
+ const lock = join3(directory, LOCK);
752
+ const deadline = Date.now() + LOCK_WAIT_MS;
753
+ for (; ; ) {
754
+ try {
755
+ writeFileSync2(lock, String(process.pid), { flag: "wx" });
756
+ break;
757
+ } catch (e) {
758
+ if (e.code !== "EEXIST") throw e;
759
+ if (ageOf(lock) > LOCK_STALE_MS) {
760
+ rmSync(lock, { force: true });
761
+ continue;
762
+ }
763
+ if (existsSync2(join3(directory, SHIM))) return;
764
+ if (Date.now() > deadline) {
765
+ throw new InferenceError(
766
+ `Timed out waiting for another process to install node-llama-cpp into ${directory}. If nothing else is running, remove ${lock} and retry.`
767
+ );
768
+ }
769
+ await delay(250);
770
+ }
771
+ }
772
+ try {
773
+ await fn();
774
+ } finally {
775
+ rmSync(lock, { force: true });
776
+ }
777
+ }
778
+ function ageOf(path) {
779
+ try {
780
+ return Date.now() - statSync(path).mtimeMs;
781
+ } catch {
782
+ return Number.POSITIVE_INFINITY;
783
+ }
784
+ }
785
+ function delay(ms) {
786
+ return new Promise((resolve) => setTimeout(resolve, ms));
787
+ }
788
+
789
+ // src/providers/llama-cpp.ts
790
+ var loadedModels = /* @__PURE__ */ new Map();
791
+ async function disposeLlamaModels() {
792
+ const pending = [...loadedModels.values()];
793
+ loadedModels.clear();
794
+ await Promise.all(
795
+ pending.map((p) => p.then((m) => m.dispose()).catch(() => void 0))
796
+ );
797
+ }
798
+ var LlamaCppProvider = class {
799
+ constructor(model, options = {}) {
800
+ this.model = model;
801
+ if (isLlamaSelector(model)) {
802
+ throw new InferenceError(
803
+ `llama-cpp model "${model}" is a selector. Constructing a provider directly needs a concrete model (e.g. "${aliasForTier("balanced")}") \u2014 use makeProviderAsync to resolve a selector against this machine.`
804
+ );
805
+ }
806
+ this.uri = resolveLlamaModelRef(model);
807
+ this.runtime = options.runtime ?? defaultLlamaRuntime();
808
+ this.thoughtTokens = options.thoughtTokens ?? 0;
809
+ this.maxTokens = options.maxTokens;
810
+ this.modelsDirectory = options.modelsDirectory ?? defaultLlamaModelsDirectory();
811
+ this.cacheKey = buildCacheKey([this.modelsDirectory, this.uri]);
812
+ }
813
+ model;
814
+ uri;
815
+ runtime;
816
+ thoughtTokens;
817
+ maxTokens;
818
+ modelsDirectory;
819
+ /**
820
+ * Loaded-model key: the same URI in two directories is two different files.
821
+ * Built with `buildCacheKey` so its parts are length-prefixed — a plain join
822
+ * would let two different (directory, uri) pairs collide and hand a provider
823
+ * back the wrong weights.
824
+ */
825
+ cacheKey;
826
+ provider() {
827
+ return "llama-cpp";
828
+ }
829
+ modelName() {
830
+ return this.model;
831
+ }
832
+ async completeJSON(req) {
833
+ const model = await this.load();
834
+ const session = await model.createSession(systemPromptFor(req));
835
+ try {
836
+ const result = await session.prompt(req.user, {
837
+ schema: req.schema,
838
+ temperature: req.temperature,
839
+ thoughtTokens: this.thoughtTokens,
840
+ ...this.maxTokens != null ? { maxTokens: this.maxTokens } : {}
841
+ });
842
+ if (result.stopReason === "maxTokens") {
843
+ throw new Error(
844
+ `llama-cpp generation hit the token limit before completing the JSON${this.maxTokens != null ? ` (maxTokens: ${this.maxTokens})` : ""} \u2014 raise llamaCpp.maxTokens, or shorten the prompt if the context is full.`
845
+ );
846
+ }
847
+ return { json: extractJson(result.text), usage: result.usage };
848
+ } finally {
849
+ await session.dispose().catch(() => void 0);
850
+ }
851
+ }
852
+ load() {
853
+ const existing = loadedModels.get(this.cacheKey);
854
+ if (existing) return existing;
855
+ const pending = (async () => {
856
+ const path = await this.runtime.resolveModelFile(
857
+ this.uri,
858
+ this.modelsDirectory
859
+ );
860
+ return this.runtime.loadModel(path);
861
+ })();
862
+ const guarded = pending.catch((e) => {
863
+ if (loadedModels.get(this.cacheKey) === guarded) {
864
+ loadedModels.delete(this.cacheKey);
865
+ }
866
+ throw e;
867
+ });
868
+ loadedModels.set(this.cacheKey, guarded);
869
+ return guarded;
870
+ }
871
+ };
872
+ function systemPromptFor(req) {
873
+ return `${req.system}
874
+
875
+ Respond with ONLY a JSON object conforming to this JSON Schema:
876
+ ${JSON.stringify(
877
+ req.schema
878
+ )}`;
879
+ }
880
+ var runtimePromise;
881
+ function defaultLlamaRuntime() {
882
+ const real = () => (
883
+ // Drop a failed init so the next call retries. A GPU that failed to
884
+ // initialise, or a binary still being extracted by a concurrent install,
885
+ // must not poison the runtime for the rest of the process — the same rule
886
+ // `load()` applies to weights.
887
+ runtimePromise ??= loadNodeLlamaCpp().catch((e) => {
888
+ runtimePromise = void 0;
889
+ throw e;
890
+ })
891
+ );
892
+ return {
893
+ resolveModelFile: (uri, directory) => real().then((r) => r.resolveModelFile(uri, directory)),
894
+ loadModel: (path) => real().then((r) => r.loadModel(path)),
895
+ getMemoryBudgetBytes: () => real().then((r) => r.getMemoryBudgetBytes())
896
+ };
897
+ }
898
+ async function loadNodeLlamaCpp() {
899
+ let mod;
900
+ try {
901
+ mod = await import("node-llama-cpp");
902
+ } catch (e) {
903
+ if (!isModuleNotFound(e)) {
904
+ throw new InferenceError(
905
+ `node-llama-cpp is installed but failed to load (${e instanceof Error ? e.message : String(e)}). This is the copy resolved from your own node_modules, so reinstalling it here will not help \u2014 check the Node version and the platform build.`
906
+ );
907
+ }
908
+ mod = await importNodeLlamaCpp();
909
+ }
910
+ const { getLlama, resolveModelFile, LlamaChatSession, TokenMeter } = mod;
911
+ const llama = await getLlama();
912
+ return {
913
+ // `directory` is this library's own, not node-llama-cpp's global default —
914
+ // owning it is what makes `clearLlamaModels` safe.
915
+ resolveModelFile: (uri, directory) => resolveModelFile(uri, { directory }),
916
+ async loadModel(path) {
917
+ const model = await llama.loadModel({ modelPath: path });
918
+ return {
919
+ async createSession(systemPrompt) {
920
+ const context = await model.createContext();
921
+ const sequence = context.getSequence();
922
+ const session = new LlamaChatSession({
923
+ contextSequence: sequence,
924
+ systemPrompt
925
+ });
926
+ return {
927
+ async prompt(text, options) {
928
+ const grammar = await llama.createGrammarForJsonSchema(
929
+ options.schema
930
+ );
931
+ const before = sequence.tokenMeter.getState();
932
+ const result = await session.promptWithMeta(text, {
933
+ grammar,
934
+ temperature: options.temperature,
935
+ budgets: { thoughtTokens: options.thoughtTokens },
936
+ ...options.maxTokens != null ? { maxTokens: options.maxTokens } : {}
937
+ });
938
+ const diff = TokenMeter.diff(sequence.tokenMeter, before);
939
+ return {
940
+ text: result.responseText,
941
+ stopReason: result.stopReason,
942
+ usage: {
943
+ inputTokens: diff.usedInputTokens,
944
+ outputTokens: diff.usedOutputTokens
945
+ }
946
+ };
947
+ },
948
+ async dispose() {
949
+ await context.dispose();
950
+ }
951
+ };
952
+ },
953
+ async dispose() {
954
+ await model.dispose();
955
+ }
956
+ };
957
+ },
958
+ async getMemoryBudgetBytes() {
959
+ const { totalmem } = await import("os");
960
+ const ramBudget = totalmem() / 2;
961
+ try {
962
+ const vram = await llama.getVramState();
963
+ return Math.max(vram.free, ramBudget);
964
+ } catch {
965
+ return ramBudget;
966
+ }
967
+ }
968
+ };
969
+ }
970
+
971
+ // src/providers/detect.ts
972
+ var DETECTION_ORDER = [
973
+ "anthropic",
974
+ "openai",
975
+ "claude-cli",
976
+ "llama-cpp"
977
+ ];
978
+ var DEFAULT_KEY_ENV = {
979
+ anthropic: "ANTHROPIC_API_KEY",
980
+ openai: "OPENAI_API_KEY"
981
+ };
982
+ function hasKey(provider) {
983
+ return (process.env[DEFAULT_KEY_ENV[provider]] ?? "") !== "";
984
+ }
985
+ var cliProbes = /* @__PURE__ */ new Map();
986
+ function resetClaudeCliProbe() {
987
+ cliProbes.clear();
988
+ }
989
+ function probeClaudeCli(spec) {
990
+ const exec = spec.exec ?? realExec;
991
+ const command = spec.command ?? "claude";
992
+ const cached = cliProbes.get(command);
993
+ if (cached) return cached;
994
+ const probing = exec([command, "--version"], { timeoutMs: 1e4 }).then((r) => r.code === 0 && !r.timedOut && r.spawnError == null).catch(() => false);
995
+ cliProbes.set(command, probing);
996
+ return probing;
997
+ }
998
+ async function probeLlamaCpp(spec) {
999
+ const injected = spec.llamaRuntime ?? spec.llamaCpp?.runtime;
1000
+ if (injected) return budgetProbe(injected);
1001
+ const status = await nodeLlamaCppStatus();
1002
+ if (status.state === "refused") {
1003
+ return { available: false, reason: status.reason };
1004
+ }
1005
+ if (status.state === "installable") {
1006
+ return { available: true };
1007
+ }
1008
+ return budgetProbe(defaultLlamaRuntime());
1009
+ }
1010
+ function budgetProbe(runtime) {
1011
+ return runtime.getMemoryBudgetBytes().then(
1012
+ () => ({ available: true }),
1013
+ (e) => ({
1014
+ available: false,
1015
+ reason: e instanceof Error && /node-llama-cpp/.test(e.message) ? "node-llama-cpp is not installed (npm i node-llama-cpp)" : `node-llama-cpp could not start (${e instanceof Error ? e.message : String(e)})`
1016
+ })
1017
+ );
1018
+ }
1019
+ async function probe(provider, spec) {
1020
+ switch (provider) {
1021
+ case "anthropic":
1022
+ return hasKey("anthropic") ? { available: true } : {
1023
+ available: false,
1024
+ reason: `ANTHROPIC_API_KEY is not set`
1025
+ };
1026
+ case "openai":
1027
+ return hasKey("openai") || spec.baseUrl ? { available: true } : {
1028
+ available: false,
1029
+ reason: `OPENAI_API_KEY is not set and no baseUrl was given`
1030
+ };
1031
+ case "claude-cli":
1032
+ return await probeClaudeCli(spec) ? { available: true } : {
1033
+ available: false,
1034
+ reason: `could not run \`${spec.command ?? "claude"}\` (is the Claude CLI installed?)`
1035
+ };
1036
+ case "llama-cpp":
1037
+ return probeLlamaCpp(spec);
1038
+ default:
1039
+ return { available: false, reason: "not auto-selectable" };
1040
+ }
1041
+ }
1042
+ async function availableProviders(spec = {}) {
1043
+ const probes = await Promise.all(
1044
+ DETECTION_ORDER.map((name) => probe(name, spec))
1045
+ );
1046
+ return DETECTION_ORDER.filter((_, i) => probes[i].available);
1047
+ }
1048
+ async function detectProvider(spec = {}) {
1049
+ const reasons = [];
1050
+ for (const name of DETECTION_ORDER) {
1051
+ const result = await probe(name, spec);
1052
+ if (result.available) {
1053
+ warnSelected(name, spec.provider === "auto");
1054
+ return name;
1055
+ }
1056
+ reasons.push(` ${name.padEnd(10)} \u2014 ${result.reason}`);
1057
+ }
1058
+ throw new InferenceError(
1059
+ `No inference provider is available. Tried:
1060
+ ${reasons.join("\n")}
1061
+ Pass an explicit \`provider\`, set one of the keys above, or install node-llama-cpp.`
1062
+ );
1063
+ }
1064
+ var warnedSelection = false;
1065
+ var warnedDownload = false;
1066
+ function resetProviderDetectionWarning() {
1067
+ warnedSelection = false;
1068
+ warnedDownload = false;
1069
+ }
1070
+ function warnSelected(provider, wasExplicitAuto) {
1071
+ if (warnedSelection) return;
1072
+ warnedSelection = true;
1073
+ const because = wasExplicitAuto ? `provider "auto"` : "no provider specified";
1074
+ console.warn(
1075
+ `inference: ${because} \u2014 auto-selected "${provider}". Pass an explicit \`provider\` to pin it.`
1076
+ );
1077
+ }
1078
+ function warnPendingDownload(model, sizeBytes) {
1079
+ if (warnedDownload) return;
1080
+ warnedDownload = true;
1081
+ console.warn(
1082
+ `inference: "${model}" is not downloaded yet \u2014 the first run will fetch ~${(sizeBytes / 1e9).toFixed(2)} GB. Pre-fetch it, or pass an explicit \`provider\` to avoid the local model entirely.`
1083
+ );
1084
+ }
1085
+
387
1086
  // src/providers/index.ts
388
1087
  var DEFAULT_MODELS = {
389
1088
  anthropic: "claude-sonnet-4-5",
390
1089
  openai: "gpt-4o-mini",
391
1090
  "claude-cli": "claude-sonnet-4-5",
392
- mock: "mock-model"
1091
+ mock: "mock-model",
1092
+ // A selector, not a pinned model: which weights a tier points at is then a
1093
+ // catalog change rather than an API change. Resolving it needs the async
1094
+ // factory — see `resolveProviderIdentityAsync`.
1095
+ "llama-cpp": "auto"
393
1096
  };
394
1097
  var DEFAULT_API_KEY_ENV = {
395
1098
  anthropic: "ANTHROPIC_API_KEY",
@@ -397,12 +1100,51 @@ var DEFAULT_API_KEY_ENV = {
397
1100
  };
398
1101
  var DEFAULT_OPENAI_BASE_URL = "https://api.openai.com/v1";
399
1102
  function resolveProviderIdentity(spec) {
400
- return {
401
- provider: spec.provider,
402
- model: spec.model ?? DEFAULT_MODELS[spec.provider] ?? "unknown"
403
- };
1103
+ if (spec.provider == null || spec.provider === "auto") {
1104
+ throw new InferenceError(
1105
+ `No provider specified. Detecting one probes the environment, the Claude CLI and the local model runtime, which cannot be done synchronously \u2014 use resolveProviderIdentityAsync/makeProviderAsync, or name a provider (${Object.keys(DEFAULT_MODELS).join(", ")}).`
1106
+ );
1107
+ }
1108
+ const model = spec.model ?? DEFAULT_MODELS[spec.provider] ?? "unknown";
1109
+ if (spec.provider === "llama-cpp" && isLlamaSelector(model)) {
1110
+ throw new InferenceError(
1111
+ `llama-cpp model "${model}" is a selector and cannot be resolved synchronously \u2014 picking a tier probes GPU memory. Use resolveProviderIdentityAsync/makeProviderAsync, or name a concrete model (e.g. "${aliasForTier("balanced")}").`
1112
+ );
1113
+ }
1114
+ return { provider: spec.provider, model };
1115
+ }
1116
+ async function resolveProviderIdentityAsync(spec) {
1117
+ if ((spec.provider == null || spec.provider === "auto") && spec.model != null) {
1118
+ throw new InferenceError(
1119
+ `Model "${spec.model}" was given without a provider, and a model name does not say which provider owns it. Name the provider too (${Object.keys(DEFAULT_MODELS).join(", ")}), or drop the model to take the detected provider's default.`
1120
+ );
1121
+ }
1122
+ const provider = spec.provider == null || spec.provider === "auto" ? await detectProvider(spec) : spec.provider;
1123
+ const resolved = { ...spec, provider };
1124
+ const model = spec.model ?? DEFAULT_MODELS[provider] ?? "unknown";
1125
+ if (provider !== "llama-cpp" || !isLlamaSelector(model)) {
1126
+ return resolveProviderIdentity(resolved);
1127
+ }
1128
+ const tier = model === "auto" ? await probeTier(llamaRuntimeFor(spec)) : model;
1129
+ return { provider, model: aliasForTier(tier) };
1130
+ }
1131
+ function warnIfDownloadPending(spec, model) {
1132
+ const entry = LLAMA_MODELS[model];
1133
+ if (!entry) return;
1134
+ const directory = spec.llamaCpp?.modelsDirectory ?? defaultLlamaModelsDirectory();
1135
+ if (!isModelDownloaded(model, directory)) {
1136
+ warnPendingDownload(model, entry.sizeBytes);
1137
+ }
1138
+ }
1139
+ function llamaRuntimeFor(spec) {
1140
+ return spec.llamaRuntime ?? spec.llamaCpp?.runtime;
1141
+ }
1142
+ async function probeTier(runtime) {
1143
+ const source = runtime ?? defaultLlamaRuntime();
1144
+ return tierForBudget(await source.getMemoryBudgetBytes());
404
1145
  }
405
1146
  function makeProvider(spec) {
1147
+ warnIfUnsupportedNode();
406
1148
  const { model } = resolveProviderIdentity(spec);
407
1149
  switch (spec.provider) {
408
1150
  case "anthropic":
@@ -428,6 +1170,11 @@ function makeProvider(spec) {
428
1170
  );
429
1171
  case "mock":
430
1172
  return new MockProvider(spec.mockResponses ?? [{ json: {} }], model);
1173
+ case "llama-cpp":
1174
+ return new LlamaCppProvider(model, {
1175
+ ...spec.llamaCpp ?? {},
1176
+ ...spec.llamaRuntime ? { runtime: spec.llamaRuntime } : {}
1177
+ });
431
1178
  default:
432
1179
  throw new InferenceError(
433
1180
  `Unknown provider "${String(spec.provider)}". Available: ${Object.keys(
@@ -436,6 +1183,55 @@ function makeProvider(spec) {
436
1183
  );
437
1184
  }
438
1185
  }
1186
+ async function makeProviderAsync(spec) {
1187
+ const { provider, model } = await resolveProviderIdentityAsync(spec);
1188
+ if (provider === "llama-cpp") warnIfDownloadPending(spec, model);
1189
+ return makeProvider({ ...spec, provider, model });
1190
+ }
1191
+
1192
+ // src/providers/llama-clean.ts
1193
+ import { rmSync as rmSync2, statSync as statSync2 } from "fs";
1194
+ import { join as join4 } from "path";
1195
+ async function clearLlamaModels(options = {}) {
1196
+ const directory = options.directory ?? defaultLlamaModelsDirectory();
1197
+ const dryRun = options.dryRun ?? false;
1198
+ const wanted = options.models?.map(blobNameFor);
1199
+ if (!dryRun) await disposeLlamaModels();
1200
+ const files = [];
1201
+ for (const entry of listModelDirectory(directory)) {
1202
+ if (!isModelBlob(entry)) continue;
1203
+ if (wanted && !wanted.some((name) => matchesModelBlob(entry, name))) {
1204
+ continue;
1205
+ }
1206
+ const path = join4(directory, entry);
1207
+ const sizeBytes = sizeOf(path);
1208
+ if (sizeBytes === void 0) continue;
1209
+ if (!dryRun) {
1210
+ try {
1211
+ rmSync2(path);
1212
+ } catch {
1213
+ continue;
1214
+ }
1215
+ }
1216
+ files.push({ path, sizeBytes });
1217
+ }
1218
+ return {
1219
+ files,
1220
+ freedBytes: files.reduce((total, file) => total + file.sizeBytes, 0),
1221
+ directory,
1222
+ dryRun
1223
+ };
1224
+ }
1225
+ function isModelBlob(entry) {
1226
+ return entry.endsWith(".gguf") || entry.endsWith(".gguf.ipull");
1227
+ }
1228
+ function sizeOf(path) {
1229
+ try {
1230
+ return statSync2(path).size;
1231
+ } catch {
1232
+ return void 0;
1233
+ }
1234
+ }
439
1235
 
440
1236
  // src/complete.ts
441
1237
  import { Ajv2020 } from "ajv/dist/2020.js";
@@ -448,6 +1244,7 @@ function validatorFor(schema) {
448
1244
  return compiled;
449
1245
  }
450
1246
  async function completeValidatedJSON(options) {
1247
+ warnIfUnsupportedNode();
451
1248
  const {
452
1249
  provider,
453
1250
  system,
@@ -488,56 +1285,6 @@ async function completeValidatedJSON(options) {
488
1285
  return { ...base, error: lastError, durationMs: Date.now() - start };
489
1286
  }
490
1287
 
491
- // src/cache.ts
492
- import { createHash } from "crypto";
493
- import { existsSync, mkdirSync, readFileSync, writeFileSync } from "fs";
494
- import { join } from "path";
495
- function sha256(text) {
496
- return createHash("sha256").update(text, "utf8").digest("hex");
497
- }
498
- function buildCacheKey(parts) {
499
- return sha256(parts.map((p) => `${p.length}:${p}`).join("|"));
500
- }
501
- var JsonCache = class {
502
- constructor(dir, enabled = true, label = "inference") {
503
- this.dir = dir;
504
- this.enabled = enabled;
505
- this.label = label;
506
- }
507
- dir;
508
- enabled;
509
- label;
510
- /** Cache-write failures warn once per process, not once per entry. */
511
- warned = false;
512
- get(key) {
513
- if (!this.enabled) return void 0;
514
- const path = join(this.dir, `${key}.json`);
515
- if (!existsSync(path)) return void 0;
516
- try {
517
- return JSON.parse(readFileSync(path, "utf8"));
518
- } catch {
519
- return void 0;
520
- }
521
- }
522
- set(key, value) {
523
- if (!this.enabled) return;
524
- try {
525
- mkdirSync(this.dir, { recursive: true });
526
- writeFileSync(
527
- join(this.dir, `${key}.json`),
528
- JSON.stringify(value, null, 2)
529
- );
530
- } catch (e) {
531
- if (!this.warned) {
532
- this.warned = true;
533
- console.warn(
534
- `${this.label}: could not write the cache at ${this.dir} (${e instanceof Error ? e.message : String(e)}). Continuing without caching.`
535
- );
536
- }
537
- }
538
- }
539
- };
540
-
541
1288
  // src/cost.ts
542
1289
  var PRICE_TABLE = {
543
1290
  "claude-sonnet-4-5": { inputPerMTok: 3, outputPerMTok: 15 },
@@ -693,29 +1440,56 @@ export {
693
1440
  DEFAULT_MODELS,
694
1441
  DEFAULT_OPENAI_BASE_URL,
695
1442
  DEFAULT_ZONES,
1443
+ DETECTION_ORDER,
696
1444
  InferenceError,
697
1445
  JsonCache,
1446
+ LLAMA_MODELS,
1447
+ LLAMA_SELECTORS,
1448
+ LLAMA_TIERS,
1449
+ LlamaCppProvider,
698
1450
  MockProvider,
699
1451
  OpenAICompatProvider,
700
1452
  PRICE_TABLE,
701
1453
  VERDICT_SCHEMA,
1454
+ aliasForTier,
1455
+ availableProviders,
1456
+ blobNameFor,
702
1457
  buildCacheKey,
1458
+ clearLlamaModels,
703
1459
  completeValidatedJSON,
704
1460
  computeConsensus,
705
1461
  costOfRuns,
706
1462
  costOfUsage,
1463
+ defaultLlamaModelsDirectory,
1464
+ defaultLlamaRuntime,
1465
+ defaultLlamaRuntimeDirectory,
1466
+ detectProvider,
1467
+ disposeLlamaModels,
707
1468
  extractJson,
1469
+ importNodeLlamaCpp,
1470
+ isLlamaSelector,
1471
+ isModelDownloaded,
708
1472
  judge,
709
1473
  makeProvider,
1474
+ makeProviderAsync,
710
1475
  mockVerdict,
1476
+ nodeLlamaCppStatus,
711
1477
  pricingFor,
712
1478
  realExec,
1479
+ resetClaudeCliProbe,
1480
+ resetNodeVersionWarning,
1481
+ resetProviderDetectionWarning,
1482
+ resetRuntimeInstall,
713
1483
  resetTemperatureWarning,
1484
+ resolveLlamaModelRef,
714
1485
  resolveProviderIdentity,
1486
+ resolveProviderIdentityAsync,
715
1487
  runEnsemble,
716
1488
  sha256,
717
1489
  stripNulls,
1490
+ tierForBudget,
718
1491
  toStrictSchema,
1492
+ uriForTier,
719
1493
  validatorFor,
720
1494
  zoneFor
721
1495
  };