@hawkeyexl/inference 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -6,6 +6,22 @@ var InferenceError = class extends Error {
6
6
  }
7
7
  };
8
8
 
9
+ // src/runtime.ts
10
+ var MINIMUM_NODE_MAJOR = 24;
11
+ var warnedNodeVersion = false;
12
+ function warnIfUnsupportedNode(version = process.versions.node) {
13
+ if (warnedNodeVersion) return;
14
+ const major = Number.parseInt(version, 10);
15
+ if (!Number.isInteger(major) || major >= MINIMUM_NODE_MAJOR) return;
16
+ warnedNodeVersion = true;
17
+ console.warn(
18
+ `inference: running on Node ${version}, older than the Node ${MINIMUM_NODE_MAJOR} this package requires. npm only warns about that at install time (EBADENGINE), so nothing has stopped you yet \u2014 upgrade Node, or expect failures this library cannot explain.`
19
+ );
20
+ }
21
+ function resetNodeVersionWarning() {
22
+ warnedNodeVersion = false;
23
+ }
24
+
9
25
  // src/providers/anthropic.ts
10
26
  import Anthropic from "@anthropic-ai/sdk";
11
27
  var DEFAULT_TOOL_NAME = "record_result";
@@ -331,7 +347,15 @@ var ClaudeCliProvider = class {
331
347
  `Claude CLI exited ${result.code}: ${result.stderr.trim().slice(-300)}`
332
348
  );
333
349
  }
334
- const wrapper = JSON.parse(result.stdout);
350
+ let wrapper;
351
+ try {
352
+ wrapper = JSON.parse(result.stdout);
353
+ } catch {
354
+ const excerpt = result.stdout.trim().replace(/\s+/g, " ").slice(0, 200);
355
+ throw new Error(
356
+ `Claude CLI printed non-JSON output (is it logged in?): ${excerpt || "(no output)"}`
357
+ );
358
+ }
335
359
  if (typeof wrapper.result !== "string") {
336
360
  throw new Error("Claude CLI returned no result field");
337
361
  }
@@ -384,12 +408,671 @@ function mockVerdict(match, confidence, overrides = {}) {
384
408
  };
385
409
  }
386
410
 
411
+ // src/cache.ts
412
+ import { createHash } from "crypto";
413
+ import { existsSync, mkdirSync, readFileSync, writeFileSync } from "fs";
414
+ import { join } from "path";
415
+ function sha256(text) {
416
+ return createHash("sha256").update(text, "utf8").digest("hex");
417
+ }
418
+ function buildCacheKey(parts) {
419
+ return sha256(parts.map((p) => `${p.length}:${p}`).join("|"));
420
+ }
421
+ var JsonCache = class {
422
+ constructor(dir, enabled = true, label = "inference") {
423
+ this.dir = dir;
424
+ this.enabled = enabled;
425
+ this.label = label;
426
+ }
427
+ dir;
428
+ enabled;
429
+ label;
430
+ /** Cache-write failures warn once per process, not once per entry. */
431
+ warned = false;
432
+ get(key) {
433
+ if (!this.enabled) return void 0;
434
+ const path = join(this.dir, `${key}.json`);
435
+ if (!existsSync(path)) return void 0;
436
+ try {
437
+ return JSON.parse(readFileSync(path, "utf8"));
438
+ } catch {
439
+ return void 0;
440
+ }
441
+ }
442
+ set(key, value) {
443
+ if (!this.enabled) return;
444
+ try {
445
+ mkdirSync(this.dir, { recursive: true });
446
+ writeFileSync(
447
+ join(this.dir, `${key}.json`),
448
+ JSON.stringify(value, null, 2)
449
+ );
450
+ } catch (e) {
451
+ if (!this.warned) {
452
+ this.warned = true;
453
+ console.warn(
454
+ `${this.label}: could not write the cache at ${this.dir} (${e instanceof Error ? e.message : String(e)}). Continuing without caching.`
455
+ );
456
+ }
457
+ }
458
+ }
459
+ };
460
+
461
+ // src/providers/llama-models.ts
462
+ import { readdirSync } from "fs";
463
+ import { homedir } from "os";
464
+ import { join as join2 } from "path";
465
+ function defaultLlamaModelsDirectory() {
466
+ return process.env["INFERENCE_MODELS_DIR"] || join2(homedir(), ".hawkeyexl-inference", "models");
467
+ }
468
+ var LLAMA_TIERS = ["fast", "balanced", "quality"];
469
+ var LLAMA_SELECTORS = ["auto", ...LLAMA_TIERS];
470
+ var LLAMA_MODELS = deepFreezeEntries({
471
+ "gemma-4-e2b": {
472
+ uri: "hf:unsloth/gemma-4-E2B-it-qat-GGUF/gemma-4-E2B-it-qat-UD-Q4_K_XL.gguf",
473
+ sizeBytes: 2620370976,
474
+ license: "Apache-2.0",
475
+ tier: "fast",
476
+ notes: "IFEval 94.6. Smallest Gemma 4; the floor for this family."
477
+ },
478
+ "gemma-4-e4b": {
479
+ uri: "hf:unsloth/gemma-4-E4B-it-qat-GGUF/gemma-4-E4B-it-qat-UD-Q4_K_XL.gguf",
480
+ sizeBytes: 4215695776,
481
+ license: "Apache-2.0",
482
+ tier: "balanced",
483
+ notes: "IFEval 96.7. The default for most machines."
484
+ },
485
+ "gemma-4-12b": {
486
+ uri: "hf:unsloth/gemma-4-12B-it-qat-GGUF/gemma-4-12B-it-qat-UD-Q4_K_XL.gguf",
487
+ sizeBytes: 6716356800,
488
+ license: "Apache-2.0",
489
+ tier: "quality",
490
+ notes: "IFEval 97.2. Dense 12B; wants a GPU or plenty of RAM."
491
+ },
492
+ "gemma-4-26b-a4b": {
493
+ uri: "hf:unsloth/gemma-4-26B-A4B-it-qat-GGUF/gemma-4-26B-A4B-it-qat-UD-Q4_K_XL.gguf",
494
+ sizeBytes: 14249047104,
495
+ license: "Apache-2.0",
496
+ notes: "MoE: 25.2B total, 3.8B active \u2014 infers near E4B speed if it fits in memory."
497
+ },
498
+ "gemma-4-e2b-q2": {
499
+ uri: "hf:unsloth/gemma-4-E2B-it-qat-GGUF/gemma-4-E2B-it-qat-UD-Q2_K_XL.gguf",
500
+ sizeBytes: 2186186784,
501
+ license: "Apache-2.0",
502
+ notes: "Q2 build of the fast tier; smallest download, lowest fidelity."
503
+ }
504
+ });
505
+ function deepFreezeEntries(catalog) {
506
+ for (const entry of Object.values(catalog)) Object.freeze(entry);
507
+ return Object.freeze(catalog);
508
+ }
509
+ var TIER_ALIAS = {
510
+ fast: "gemma-4-e2b",
511
+ balanced: "gemma-4-e4b",
512
+ quality: "gemma-4-12b"
513
+ };
514
+ function isLlamaSelector(model) {
515
+ return LLAMA_SELECTORS.includes(model);
516
+ }
517
+ var MEMORY_HEADROOM = 3.5;
518
+ function tierForBudget(budgetBytes) {
519
+ let chosen = "fast";
520
+ for (const tier of LLAMA_TIERS) {
521
+ const entry = LLAMA_MODELS[TIER_ALIAS[tier]];
522
+ if (entry.sizeBytes * MEMORY_HEADROOM <= budgetBytes) chosen = tier;
523
+ }
524
+ return chosen;
525
+ }
526
+ function aliasForTier(tier) {
527
+ return TIER_ALIAS[tier];
528
+ }
529
+ function uriForTier(tier) {
530
+ return LLAMA_MODELS[TIER_ALIAS[tier]].uri;
531
+ }
532
+ function resolveLlamaModelRef(model) {
533
+ if (isLlamaSelector(model)) {
534
+ throw new InferenceError(
535
+ `llama-cpp model "${model}" is a selector and needs a hardware probe to resolve. Use resolveProviderIdentityAsync/makeProviderAsync, or name a concrete model (e.g. "gemma-4-e4b").`
536
+ );
537
+ }
538
+ const entry = LLAMA_MODELS[model];
539
+ if (entry) return entry.uri;
540
+ if (isModelPathOrUri(model)) return model;
541
+ throw new InferenceError(
542
+ `Unknown llama-cpp model "${model}". Use a selector (${LLAMA_SELECTORS.join(
543
+ ", "
544
+ )}), a curated alias (${Object.keys(LLAMA_MODELS).join(
545
+ ", "
546
+ )}), an hf: URI, or a path to a .gguf file.`
547
+ );
548
+ }
549
+ function blobNameFor(model) {
550
+ const ref = resolveLlamaModelRef(model).split("#")[0];
551
+ return ref.split(/[/\\]/).pop();
552
+ }
553
+ function matchesModelBlob(entry, blobName) {
554
+ const base = entry.replace(/\.ipull$/, "");
555
+ if (base.endsWith(blobName)) return true;
556
+ const stem = blobName.replace(/\.gguf$/, "");
557
+ return new RegExp(`${escapeRegExp(stem)}-\\d{5}-of-\\d{5}\\.gguf$`).test(base);
558
+ }
559
+ function escapeRegExp(text) {
560
+ return text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
561
+ }
562
+ function isModelDownloaded(model, directory) {
563
+ const blobName = blobNameFor(model);
564
+ return listModelDirectory(directory).some(
565
+ (entry) => !entry.endsWith(".ipull") && matchesModelBlob(entry, blobName)
566
+ );
567
+ }
568
+ function listModelDirectory(directory) {
569
+ try {
570
+ return readdirSync(directory, { withFileTypes: true }).filter((entry) => entry.isFile()).map((entry) => entry.name);
571
+ } catch {
572
+ return [];
573
+ }
574
+ }
575
+ function isModelPathOrUri(model) {
576
+ return /^(hf|huggingface):/i.test(model) || /^https?:\/\//i.test(model) || /^(hf|huggingface)\.co\//i.test(model) || model.endsWith(".gguf");
577
+ }
578
+
579
+ // src/providers/llama-install.ts
580
+ import {
581
+ existsSync as existsSync2,
582
+ mkdirSync as mkdirSync2,
583
+ rmSync,
584
+ statSync,
585
+ writeFileSync as writeFileSync2
586
+ } from "fs";
587
+ import { homedir as homedir2 } from "os";
588
+ import { join as join3 } from "path";
589
+ import { pathToFileURL } from "url";
590
+ var PACKAGE_SPEC = "node-llama-cpp@^3.19.0";
591
+ var SHIM = "loader.mjs";
592
+ var LOCK = ".install.lock";
593
+ var INSTALL_TIMEOUT_MS = 9e5;
594
+ var LOCK_WAIT_MS = INSTALL_TIMEOUT_MS;
595
+ var LOCK_STALE_MS = INSTALL_TIMEOUT_MS + 6e4;
596
+ function defaultLlamaRuntimeDirectory(env = process.env) {
597
+ return env["INFERENCE_RUNTIME_DIR"] || join3(homedir2(), ".hawkeyexl-inference", "runtime");
598
+ }
599
+ function isModuleNotFound(e) {
600
+ const code = e?.code;
601
+ return code === "ERR_MODULE_NOT_FOUND" || code === "MODULE_NOT_FOUND";
602
+ }
603
+ function describe(e) {
604
+ return e instanceof Error ? e.message : String(e);
605
+ }
606
+ async function nodeLlamaCppStatus(options = {}) {
607
+ const env = options.env ?? process.env;
608
+ const directory = options.directory ?? defaultLlamaRuntimeDirectory(env);
609
+ const probeImport = options.probeImport ?? (() => import("node-llama-cpp"));
610
+ try {
611
+ await probeImport();
612
+ return { state: "present" };
613
+ } catch (e) {
614
+ if (!isModuleNotFound(e)) {
615
+ return {
616
+ state: "refused",
617
+ reason: `node-llama-cpp is installed but failed to load (${describe(e)})`
618
+ };
619
+ }
620
+ }
621
+ if (existsSync2(join3(directory, SHIM))) return { state: "present" };
622
+ if ((env["INFERENCE_NO_AUTO_INSTALL"] ?? "") !== "") {
623
+ return {
624
+ state: "refused",
625
+ reason: `node-llama-cpp is not installed and INFERENCE_NO_AUTO_INSTALL is set`
626
+ };
627
+ }
628
+ return { state: "installable", directory };
629
+ }
630
+ var installs = /* @__PURE__ */ new Map();
631
+ var warnedInstall = false;
632
+ function resetRuntimeInstall() {
633
+ installs.clear();
634
+ warnedInstall = false;
635
+ }
636
+ function importNodeLlamaCpp(options = {}) {
637
+ const env = options.env ?? process.env;
638
+ const directory = options.directory ?? defaultLlamaRuntimeDirectory(env);
639
+ const existing = installs.get(directory);
640
+ if (existing) return existing;
641
+ const pending = fromPrefix(directory, env, options);
642
+ const guarded = pending.catch((e) => {
643
+ if (installs.get(directory) === guarded) installs.delete(directory);
644
+ throw e;
645
+ });
646
+ installs.set(directory, guarded);
647
+ return guarded;
648
+ }
649
+ async function fromPrefix(directory, env, options) {
650
+ const importShim = options.importShim ?? ((url) => import(url));
651
+ const shim = join3(directory, SHIM);
652
+ if (existsSync2(shim)) return importShim(pathToFileURL(shim).href);
653
+ if ((env["INFERENCE_NO_AUTO_INSTALL"] ?? "") !== "") {
654
+ throw new InferenceError(
655
+ `The llama-cpp provider needs node-llama-cpp, and INFERENCE_NO_AUTO_INSTALL is set. Install it yourself (npm i ${PACKAGE_SPEC}), unset INFERENCE_NO_AUTO_INSTALL to allow installing into ${directory}, or name a different provider.`
656
+ );
657
+ }
658
+ mkdirSync2(directory, { recursive: true });
659
+ await withLock(directory, async () => {
660
+ if (existsSync2(shim)) return;
661
+ warnInstalling(directory);
662
+ await runInstall(directory, env, options);
663
+ writeFileSync2(shim, `export * from "node-llama-cpp";
664
+ `, "utf8");
665
+ });
666
+ return importShim(pathToFileURL(shim).href);
667
+ }
668
+ async function runInstall(directory, env, options) {
669
+ const manifest = join3(directory, "package.json");
670
+ if (!existsSync2(manifest)) {
671
+ writeFileSync2(
672
+ manifest,
673
+ `${JSON.stringify(
674
+ {
675
+ name: "hawkeyexl-inference-runtime",
676
+ version: "0.0.0",
677
+ private: true,
678
+ description: "Auto-installed runtime for @hawkeyexl/inference. Safe to delete."
679
+ },
680
+ null,
681
+ 2
682
+ )}
683
+ `,
684
+ "utf8"
685
+ );
686
+ }
687
+ const exec = options.exec ?? realExec;
688
+ const result = await exec(
689
+ [
690
+ "npm",
691
+ "install",
692
+ "--prefix",
693
+ directory,
694
+ PACKAGE_SPEC,
695
+ "--no-audit",
696
+ "--no-fund"
697
+ ],
698
+ { timeoutMs: options.timeoutMs ?? INSTALL_TIMEOUT_MS, env }
699
+ );
700
+ if (result.spawnError != null) {
701
+ throw new InferenceError(
702
+ `Could not run npm to install node-llama-cpp (${result.spawnError}). Install it yourself with: npm i ${PACKAGE_SPEC}, or name a provider that does not need it.`
703
+ );
704
+ }
705
+ if (result.timedOut) {
706
+ throw new InferenceError(
707
+ `Installing node-llama-cpp into ${directory} timed out. A source build can take a while \u2014 retry, raise the timeout, or install it yourself with: npm i ${PACKAGE_SPEC}.`
708
+ );
709
+ }
710
+ if (result.code !== 0) {
711
+ throw new InferenceError(
712
+ `Installing node-llama-cpp into ${directory} failed (exit ${String(
713
+ result.code
714
+ )}).
715
+ ${tail(result.stderr || result.stdout)}
716
+ Install it yourself with: npm i ${PACKAGE_SPEC}, or name a provider that does not need it.`
717
+ );
718
+ }
719
+ }
720
+ function tail(output, lines = 12) {
721
+ return output.trimEnd().split(/\r?\n/).slice(-lines).join("\n");
722
+ }
723
+ function warnInstalling(directory) {
724
+ if (warnedInstall) return;
725
+ warnedInstall = true;
726
+ console.warn(
727
+ `inference: node-llama-cpp is not installed \u2014 fetching it into ${directory} so the local model can run. This is a one-time native install; set INFERENCE_NO_AUTO_INSTALL=1 to refuse it, or name a provider that does not need it.`
728
+ );
729
+ }
730
+ async function withLock(directory, fn) {
731
+ const lock = join3(directory, LOCK);
732
+ const deadline = Date.now() + LOCK_WAIT_MS;
733
+ for (; ; ) {
734
+ try {
735
+ writeFileSync2(lock, String(process.pid), { flag: "wx" });
736
+ break;
737
+ } catch (e) {
738
+ if (e.code !== "EEXIST") throw e;
739
+ if (ageOf(lock) > LOCK_STALE_MS) {
740
+ rmSync(lock, { force: true });
741
+ continue;
742
+ }
743
+ if (existsSync2(join3(directory, SHIM))) return;
744
+ if (Date.now() > deadline) {
745
+ throw new InferenceError(
746
+ `Timed out waiting for another process to install node-llama-cpp into ${directory}. If nothing else is running, remove ${lock} and retry.`
747
+ );
748
+ }
749
+ await delay(250);
750
+ }
751
+ }
752
+ try {
753
+ await fn();
754
+ } finally {
755
+ rmSync(lock, { force: true });
756
+ }
757
+ }
758
+ function ageOf(path) {
759
+ try {
760
+ return Date.now() - statSync(path).mtimeMs;
761
+ } catch {
762
+ return Number.POSITIVE_INFINITY;
763
+ }
764
+ }
765
+ function delay(ms) {
766
+ return new Promise((resolve) => setTimeout(resolve, ms));
767
+ }
768
+
769
+ // src/providers/llama-cpp.ts
770
+ var loadedModels = /* @__PURE__ */ new Map();
771
+ async function disposeLlamaModels() {
772
+ const pending = [...loadedModels.values()];
773
+ loadedModels.clear();
774
+ await Promise.all(
775
+ pending.map((p) => p.then((m) => m.dispose()).catch(() => void 0))
776
+ );
777
+ }
778
+ var LlamaCppProvider = class {
779
+ constructor(model, options = {}) {
780
+ this.model = model;
781
+ if (isLlamaSelector(model)) {
782
+ throw new InferenceError(
783
+ `llama-cpp model "${model}" is a selector. Constructing a provider directly needs a concrete model (e.g. "gemma-4-e4b") \u2014 use makeProviderAsync to resolve a selector against this machine.`
784
+ );
785
+ }
786
+ this.uri = resolveLlamaModelRef(model);
787
+ this.runtime = options.runtime ?? defaultLlamaRuntime();
788
+ this.thoughtTokens = options.thoughtTokens ?? 0;
789
+ this.maxTokens = options.maxTokens;
790
+ this.modelsDirectory = options.modelsDirectory ?? defaultLlamaModelsDirectory();
791
+ this.cacheKey = buildCacheKey([this.modelsDirectory, this.uri]);
792
+ }
793
+ model;
794
+ uri;
795
+ runtime;
796
+ thoughtTokens;
797
+ maxTokens;
798
+ modelsDirectory;
799
+ /**
800
+ * Loaded-model key: the same URI in two directories is two different files.
801
+ * Built with `buildCacheKey` so its parts are length-prefixed — a plain join
802
+ * would let two different (directory, uri) pairs collide and hand a provider
803
+ * back the wrong weights.
804
+ */
805
+ cacheKey;
806
+ provider() {
807
+ return "llama-cpp";
808
+ }
809
+ modelName() {
810
+ return this.model;
811
+ }
812
+ async completeJSON(req) {
813
+ const model = await this.load();
814
+ const session = await model.createSession(systemPromptFor(req));
815
+ try {
816
+ const result = await session.prompt(req.user, {
817
+ schema: req.schema,
818
+ temperature: req.temperature,
819
+ thoughtTokens: this.thoughtTokens,
820
+ ...this.maxTokens != null ? { maxTokens: this.maxTokens } : {}
821
+ });
822
+ if (result.stopReason === "maxTokens") {
823
+ throw new Error(
824
+ `llama-cpp generation hit the token limit before completing the JSON${this.maxTokens != null ? ` (maxTokens: ${this.maxTokens})` : ""} \u2014 raise llamaCpp.maxTokens, or shorten the prompt if the context is full.`
825
+ );
826
+ }
827
+ return { json: extractJson(result.text), usage: result.usage };
828
+ } finally {
829
+ await session.dispose().catch(() => void 0);
830
+ }
831
+ }
832
+ load() {
833
+ const existing = loadedModels.get(this.cacheKey);
834
+ if (existing) return existing;
835
+ const pending = (async () => {
836
+ const path = await this.runtime.resolveModelFile(
837
+ this.uri,
838
+ this.modelsDirectory
839
+ );
840
+ return this.runtime.loadModel(path);
841
+ })();
842
+ const guarded = pending.catch((e) => {
843
+ if (loadedModels.get(this.cacheKey) === guarded) {
844
+ loadedModels.delete(this.cacheKey);
845
+ }
846
+ throw e;
847
+ });
848
+ loadedModels.set(this.cacheKey, guarded);
849
+ return guarded;
850
+ }
851
+ };
852
+ function systemPromptFor(req) {
853
+ return `${req.system}
854
+
855
+ Respond with ONLY a JSON object conforming to this JSON Schema:
856
+ ${JSON.stringify(
857
+ req.schema
858
+ )}`;
859
+ }
860
+ var runtimePromise;
861
+ function defaultLlamaRuntime() {
862
+ const real = () => (
863
+ // Drop a failed init so the next call retries. A GPU that failed to
864
+ // initialise, or a binary still being extracted by a concurrent install,
865
+ // must not poison the runtime for the rest of the process — the same rule
866
+ // `load()` applies to weights.
867
+ runtimePromise ??= loadNodeLlamaCpp().catch((e) => {
868
+ runtimePromise = void 0;
869
+ throw e;
870
+ })
871
+ );
872
+ return {
873
+ resolveModelFile: (uri, directory) => real().then((r) => r.resolveModelFile(uri, directory)),
874
+ loadModel: (path) => real().then((r) => r.loadModel(path)),
875
+ getMemoryBudgetBytes: () => real().then((r) => r.getMemoryBudgetBytes())
876
+ };
877
+ }
878
+ async function loadNodeLlamaCpp() {
879
+ let mod;
880
+ try {
881
+ mod = await import("node-llama-cpp");
882
+ } catch (e) {
883
+ if (!isModuleNotFound(e)) {
884
+ throw new InferenceError(
885
+ `node-llama-cpp is installed but failed to load (${e instanceof Error ? e.message : String(e)}). This is the copy resolved from your own node_modules, so reinstalling it here will not help \u2014 check the Node version and the platform build.`
886
+ );
887
+ }
888
+ mod = await importNodeLlamaCpp();
889
+ }
890
+ const { getLlama, resolveModelFile, LlamaChatSession, TokenMeter } = mod;
891
+ const llama = await getLlama();
892
+ return {
893
+ // `directory` is this library's own, not node-llama-cpp's global default —
894
+ // owning it is what makes `clearLlamaModels` safe.
895
+ resolveModelFile: (uri, directory) => resolveModelFile(uri, { directory }),
896
+ async loadModel(path) {
897
+ const model = await llama.loadModel({ modelPath: path });
898
+ return {
899
+ async createSession(systemPrompt) {
900
+ const context = await model.createContext();
901
+ const sequence = context.getSequence();
902
+ const session = new LlamaChatSession({
903
+ contextSequence: sequence,
904
+ systemPrompt
905
+ });
906
+ return {
907
+ async prompt(text, options) {
908
+ const grammar = await llama.createGrammarForJsonSchema(
909
+ options.schema
910
+ );
911
+ const before = sequence.tokenMeter.getState();
912
+ const result = await session.promptWithMeta(text, {
913
+ grammar,
914
+ temperature: options.temperature,
915
+ budgets: { thoughtTokens: options.thoughtTokens },
916
+ ...options.maxTokens != null ? { maxTokens: options.maxTokens } : {}
917
+ });
918
+ const diff = TokenMeter.diff(sequence.tokenMeter, before);
919
+ return {
920
+ text: result.responseText,
921
+ stopReason: result.stopReason,
922
+ usage: {
923
+ inputTokens: diff.usedInputTokens,
924
+ outputTokens: diff.usedOutputTokens
925
+ }
926
+ };
927
+ },
928
+ async dispose() {
929
+ await context.dispose();
930
+ }
931
+ };
932
+ },
933
+ async dispose() {
934
+ await model.dispose();
935
+ }
936
+ };
937
+ },
938
+ async getMemoryBudgetBytes() {
939
+ const { totalmem } = await import("os");
940
+ const ramBudget = totalmem() / 2;
941
+ try {
942
+ const vram = await llama.getVramState();
943
+ return Math.max(vram.free, ramBudget);
944
+ } catch {
945
+ return ramBudget;
946
+ }
947
+ }
948
+ };
949
+ }
950
+
951
+ // src/providers/detect.ts
952
+ var DETECTION_ORDER = [
953
+ "anthropic",
954
+ "openai",
955
+ "claude-cli",
956
+ "llama-cpp"
957
+ ];
958
+ var DEFAULT_KEY_ENV = {
959
+ anthropic: "ANTHROPIC_API_KEY",
960
+ openai: "OPENAI_API_KEY"
961
+ };
962
+ function hasKey(provider) {
963
+ return (process.env[DEFAULT_KEY_ENV[provider]] ?? "") !== "";
964
+ }
965
+ var cliProbes = /* @__PURE__ */ new Map();
966
+ function resetClaudeCliProbe() {
967
+ cliProbes.clear();
968
+ }
969
+ function probeClaudeCli(spec) {
970
+ const exec = spec.exec ?? realExec;
971
+ const command = spec.command ?? "claude";
972
+ const cached = cliProbes.get(command);
973
+ if (cached) return cached;
974
+ const probing = exec([command, "--version"], { timeoutMs: 1e4 }).then((r) => r.code === 0 && !r.timedOut && r.spawnError == null).catch(() => false);
975
+ cliProbes.set(command, probing);
976
+ return probing;
977
+ }
978
+ async function probeLlamaCpp(spec) {
979
+ const injected = spec.llamaRuntime ?? spec.llamaCpp?.runtime;
980
+ if (injected) return budgetProbe(injected);
981
+ const status = await nodeLlamaCppStatus();
982
+ if (status.state === "refused") {
983
+ return { available: false, reason: status.reason };
984
+ }
985
+ if (status.state === "installable") {
986
+ return { available: true };
987
+ }
988
+ return budgetProbe(defaultLlamaRuntime());
989
+ }
990
+ function budgetProbe(runtime) {
991
+ return runtime.getMemoryBudgetBytes().then(
992
+ () => ({ available: true }),
993
+ (e) => ({
994
+ available: false,
995
+ reason: e instanceof Error && /node-llama-cpp/.test(e.message) ? "node-llama-cpp is not installed (npm i node-llama-cpp)" : `node-llama-cpp could not start (${e instanceof Error ? e.message : String(e)})`
996
+ })
997
+ );
998
+ }
999
+ async function probe(provider, spec) {
1000
+ switch (provider) {
1001
+ case "anthropic":
1002
+ return hasKey("anthropic") ? { available: true } : {
1003
+ available: false,
1004
+ reason: `ANTHROPIC_API_KEY is not set`
1005
+ };
1006
+ case "openai":
1007
+ return hasKey("openai") || spec.baseUrl ? { available: true } : {
1008
+ available: false,
1009
+ reason: `OPENAI_API_KEY is not set and no baseUrl was given`
1010
+ };
1011
+ case "claude-cli":
1012
+ return await probeClaudeCli(spec) ? { available: true } : {
1013
+ available: false,
1014
+ reason: `could not run \`${spec.command ?? "claude"}\` (is the Claude CLI installed?)`
1015
+ };
1016
+ case "llama-cpp":
1017
+ return probeLlamaCpp(spec);
1018
+ default:
1019
+ return { available: false, reason: "not auto-selectable" };
1020
+ }
1021
+ }
1022
+ async function availableProviders(spec = {}) {
1023
+ const probes = await Promise.all(
1024
+ DETECTION_ORDER.map((name) => probe(name, spec))
1025
+ );
1026
+ return DETECTION_ORDER.filter((_, i) => probes[i].available);
1027
+ }
1028
+ async function detectProvider(spec = {}) {
1029
+ const reasons = [];
1030
+ for (const name of DETECTION_ORDER) {
1031
+ const result = await probe(name, spec);
1032
+ if (result.available) {
1033
+ warnSelected(name, spec.provider === "auto");
1034
+ return name;
1035
+ }
1036
+ reasons.push(` ${name.padEnd(10)} \u2014 ${result.reason}`);
1037
+ }
1038
+ throw new InferenceError(
1039
+ `No inference provider is available. Tried:
1040
+ ${reasons.join("\n")}
1041
+ Pass an explicit \`provider\`, set one of the keys above, or install node-llama-cpp.`
1042
+ );
1043
+ }
1044
+ var warnedSelection = false;
1045
+ var warnedDownload = false;
1046
+ function resetProviderDetectionWarning() {
1047
+ warnedSelection = false;
1048
+ warnedDownload = false;
1049
+ }
1050
+ function warnSelected(provider, wasExplicitAuto) {
1051
+ if (warnedSelection) return;
1052
+ warnedSelection = true;
1053
+ const because = wasExplicitAuto ? `provider "auto"` : "no provider specified";
1054
+ console.warn(
1055
+ `inference: ${because} \u2014 auto-selected "${provider}". Pass an explicit \`provider\` to pin it.`
1056
+ );
1057
+ }
1058
+ function warnPendingDownload(model, sizeBytes) {
1059
+ if (warnedDownload) return;
1060
+ warnedDownload = true;
1061
+ console.warn(
1062
+ `inference: "${model}" is not downloaded yet \u2014 the first run will fetch ~${(sizeBytes / 1e9).toFixed(2)} GB. Pre-fetch it, or pass an explicit \`provider\` to avoid the local model entirely.`
1063
+ );
1064
+ }
1065
+
387
1066
  // src/providers/index.ts
388
1067
  var DEFAULT_MODELS = {
389
1068
  anthropic: "claude-sonnet-4-5",
390
1069
  openai: "gpt-4o-mini",
391
1070
  "claude-cli": "claude-sonnet-4-5",
392
- mock: "mock-model"
1071
+ mock: "mock-model",
1072
+ // A selector, not a pinned model: which weights a tier points at is then a
1073
+ // catalog change rather than an API change. Resolving it needs the async
1074
+ // factory — see `resolveProviderIdentityAsync`.
1075
+ "llama-cpp": "auto"
393
1076
  };
394
1077
  var DEFAULT_API_KEY_ENV = {
395
1078
  anthropic: "ANTHROPIC_API_KEY",
@@ -397,12 +1080,51 @@ var DEFAULT_API_KEY_ENV = {
397
1080
  };
398
1081
  var DEFAULT_OPENAI_BASE_URL = "https://api.openai.com/v1";
399
1082
  function resolveProviderIdentity(spec) {
400
- return {
401
- provider: spec.provider,
402
- model: spec.model ?? DEFAULT_MODELS[spec.provider] ?? "unknown"
403
- };
1083
+ if (spec.provider == null || spec.provider === "auto") {
1084
+ throw new InferenceError(
1085
+ `No provider specified. Detecting one probes the environment, the Claude CLI and the local model runtime, which cannot be done synchronously \u2014 use resolveProviderIdentityAsync/makeProviderAsync, or name a provider (${Object.keys(DEFAULT_MODELS).join(", ")}).`
1086
+ );
1087
+ }
1088
+ const model = spec.model ?? DEFAULT_MODELS[spec.provider] ?? "unknown";
1089
+ if (spec.provider === "llama-cpp" && isLlamaSelector(model)) {
1090
+ throw new InferenceError(
1091
+ `llama-cpp model "${model}" is a selector and cannot be resolved synchronously \u2014 picking a tier probes GPU memory. Use resolveProviderIdentityAsync/makeProviderAsync, or name a concrete model (e.g. "gemma-4-e4b").`
1092
+ );
1093
+ }
1094
+ return { provider: spec.provider, model };
1095
+ }
1096
+ async function resolveProviderIdentityAsync(spec) {
1097
+ if ((spec.provider == null || spec.provider === "auto") && spec.model != null) {
1098
+ throw new InferenceError(
1099
+ `Model "${spec.model}" was given without a provider, and a model name does not say which provider owns it. Name the provider too (${Object.keys(DEFAULT_MODELS).join(", ")}), or drop the model to take the detected provider's default.`
1100
+ );
1101
+ }
1102
+ const provider = spec.provider == null || spec.provider === "auto" ? await detectProvider(spec) : spec.provider;
1103
+ const resolved = { ...spec, provider };
1104
+ const model = spec.model ?? DEFAULT_MODELS[provider] ?? "unknown";
1105
+ if (provider !== "llama-cpp" || !isLlamaSelector(model)) {
1106
+ return resolveProviderIdentity(resolved);
1107
+ }
1108
+ const tier = model === "auto" ? await probeTier(llamaRuntimeFor(spec)) : model;
1109
+ return { provider, model: aliasForTier(tier) };
1110
+ }
1111
+ function warnIfDownloadPending(spec, model) {
1112
+ const entry = LLAMA_MODELS[model];
1113
+ if (!entry) return;
1114
+ const directory = spec.llamaCpp?.modelsDirectory ?? defaultLlamaModelsDirectory();
1115
+ if (!isModelDownloaded(model, directory)) {
1116
+ warnPendingDownload(model, entry.sizeBytes);
1117
+ }
1118
+ }
1119
+ function llamaRuntimeFor(spec) {
1120
+ return spec.llamaRuntime ?? spec.llamaCpp?.runtime;
1121
+ }
1122
+ async function probeTier(runtime) {
1123
+ const source = runtime ?? defaultLlamaRuntime();
1124
+ return tierForBudget(await source.getMemoryBudgetBytes());
404
1125
  }
405
1126
  function makeProvider(spec) {
1127
+ warnIfUnsupportedNode();
406
1128
  const { model } = resolveProviderIdentity(spec);
407
1129
  switch (spec.provider) {
408
1130
  case "anthropic":
@@ -428,6 +1150,11 @@ function makeProvider(spec) {
428
1150
  );
429
1151
  case "mock":
430
1152
  return new MockProvider(spec.mockResponses ?? [{ json: {} }], model);
1153
+ case "llama-cpp":
1154
+ return new LlamaCppProvider(model, {
1155
+ ...spec.llamaCpp ?? {},
1156
+ ...spec.llamaRuntime ? { runtime: spec.llamaRuntime } : {}
1157
+ });
431
1158
  default:
432
1159
  throw new InferenceError(
433
1160
  `Unknown provider "${String(spec.provider)}". Available: ${Object.keys(
@@ -436,6 +1163,55 @@ function makeProvider(spec) {
436
1163
  );
437
1164
  }
438
1165
  }
1166
+ async function makeProviderAsync(spec) {
1167
+ const { provider, model } = await resolveProviderIdentityAsync(spec);
1168
+ if (provider === "llama-cpp") warnIfDownloadPending(spec, model);
1169
+ return makeProvider({ ...spec, provider, model });
1170
+ }
1171
+
1172
+ // src/providers/llama-clean.ts
1173
+ import { rmSync as rmSync2, statSync as statSync2 } from "fs";
1174
+ import { join as join4 } from "path";
1175
+ async function clearLlamaModels(options = {}) {
1176
+ const directory = options.directory ?? defaultLlamaModelsDirectory();
1177
+ const dryRun = options.dryRun ?? false;
1178
+ const wanted = options.models?.map(blobNameFor);
1179
+ if (!dryRun) await disposeLlamaModels();
1180
+ const files = [];
1181
+ for (const entry of listModelDirectory(directory)) {
1182
+ if (!isModelBlob(entry)) continue;
1183
+ if (wanted && !wanted.some((name) => matchesModelBlob(entry, name))) {
1184
+ continue;
1185
+ }
1186
+ const path = join4(directory, entry);
1187
+ const sizeBytes = sizeOf(path);
1188
+ if (sizeBytes === void 0) continue;
1189
+ if (!dryRun) {
1190
+ try {
1191
+ rmSync2(path);
1192
+ } catch {
1193
+ continue;
1194
+ }
1195
+ }
1196
+ files.push({ path, sizeBytes });
1197
+ }
1198
+ return {
1199
+ files,
1200
+ freedBytes: files.reduce((total, file) => total + file.sizeBytes, 0),
1201
+ directory,
1202
+ dryRun
1203
+ };
1204
+ }
1205
+ function isModelBlob(entry) {
1206
+ return entry.endsWith(".gguf") || entry.endsWith(".gguf.ipull");
1207
+ }
1208
+ function sizeOf(path) {
1209
+ try {
1210
+ return statSync2(path).size;
1211
+ } catch {
1212
+ return void 0;
1213
+ }
1214
+ }
439
1215
 
440
1216
  // src/complete.ts
441
1217
  import { Ajv2020 } from "ajv/dist/2020.js";
@@ -448,6 +1224,7 @@ function validatorFor(schema) {
448
1224
  return compiled;
449
1225
  }
450
1226
  async function completeValidatedJSON(options) {
1227
+ warnIfUnsupportedNode();
451
1228
  const {
452
1229
  provider,
453
1230
  system,
@@ -488,56 +1265,6 @@ async function completeValidatedJSON(options) {
488
1265
  return { ...base, error: lastError, durationMs: Date.now() - start };
489
1266
  }
490
1267
 
491
- // src/cache.ts
492
- import { createHash } from "crypto";
493
- import { existsSync, mkdirSync, readFileSync, writeFileSync } from "fs";
494
- import { join } from "path";
495
- function sha256(text) {
496
- return createHash("sha256").update(text, "utf8").digest("hex");
497
- }
498
- function buildCacheKey(parts) {
499
- return sha256(parts.map((p) => `${p.length}:${p}`).join("|"));
500
- }
501
- var JsonCache = class {
502
- constructor(dir, enabled = true, label = "inference") {
503
- this.dir = dir;
504
- this.enabled = enabled;
505
- this.label = label;
506
- }
507
- dir;
508
- enabled;
509
- label;
510
- /** Cache-write failures warn once per process, not once per entry. */
511
- warned = false;
512
- get(key) {
513
- if (!this.enabled) return void 0;
514
- const path = join(this.dir, `${key}.json`);
515
- if (!existsSync(path)) return void 0;
516
- try {
517
- return JSON.parse(readFileSync(path, "utf8"));
518
- } catch {
519
- return void 0;
520
- }
521
- }
522
- set(key, value) {
523
- if (!this.enabled) return;
524
- try {
525
- mkdirSync(this.dir, { recursive: true });
526
- writeFileSync(
527
- join(this.dir, `${key}.json`),
528
- JSON.stringify(value, null, 2)
529
- );
530
- } catch (e) {
531
- if (!this.warned) {
532
- this.warned = true;
533
- console.warn(
534
- `${this.label}: could not write the cache at ${this.dir} (${e instanceof Error ? e.message : String(e)}). Continuing without caching.`
535
- );
536
- }
537
- }
538
- }
539
- };
540
-
541
1268
  // src/cost.ts
542
1269
  var PRICE_TABLE = {
543
1270
  "claude-sonnet-4-5": { inputPerMTok: 3, outputPerMTok: 15 },
@@ -693,29 +1420,56 @@ export {
693
1420
  DEFAULT_MODELS,
694
1421
  DEFAULT_OPENAI_BASE_URL,
695
1422
  DEFAULT_ZONES,
1423
+ DETECTION_ORDER,
696
1424
  InferenceError,
697
1425
  JsonCache,
1426
+ LLAMA_MODELS,
1427
+ LLAMA_SELECTORS,
1428
+ LLAMA_TIERS,
1429
+ LlamaCppProvider,
698
1430
  MockProvider,
699
1431
  OpenAICompatProvider,
700
1432
  PRICE_TABLE,
701
1433
  VERDICT_SCHEMA,
1434
+ aliasForTier,
1435
+ availableProviders,
1436
+ blobNameFor,
702
1437
  buildCacheKey,
1438
+ clearLlamaModels,
703
1439
  completeValidatedJSON,
704
1440
  computeConsensus,
705
1441
  costOfRuns,
706
1442
  costOfUsage,
1443
+ defaultLlamaModelsDirectory,
1444
+ defaultLlamaRuntime,
1445
+ defaultLlamaRuntimeDirectory,
1446
+ detectProvider,
1447
+ disposeLlamaModels,
707
1448
  extractJson,
1449
+ importNodeLlamaCpp,
1450
+ isLlamaSelector,
1451
+ isModelDownloaded,
708
1452
  judge,
709
1453
  makeProvider,
1454
+ makeProviderAsync,
710
1455
  mockVerdict,
1456
+ nodeLlamaCppStatus,
711
1457
  pricingFor,
712
1458
  realExec,
1459
+ resetClaudeCliProbe,
1460
+ resetNodeVersionWarning,
1461
+ resetProviderDetectionWarning,
1462
+ resetRuntimeInstall,
713
1463
  resetTemperatureWarning,
1464
+ resolveLlamaModelRef,
714
1465
  resolveProviderIdentity,
1466
+ resolveProviderIdentityAsync,
715
1467
  runEnsemble,
716
1468
  sha256,
717
1469
  stripNulls,
1470
+ tierForBudget,
718
1471
  toStrictSchema,
1472
+ uriForTier,
719
1473
  validatorFor,
720
1474
  zoneFor
721
1475
  };