@hawkeyexl/inference 0.0.1 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +64 -138
- package/dist/index.d.ts +346 -8
- package/dist/index.js +810 -56
- package/dist/index.js.map +1 -1
- package/package.json +15 -3
package/dist/index.js
CHANGED
|
@@ -6,6 +6,22 @@ var InferenceError = class extends Error {
|
|
|
6
6
|
}
|
|
7
7
|
};
|
|
8
8
|
|
|
9
|
+
// src/runtime.ts
|
|
10
|
+
var MINIMUM_NODE_MAJOR = 24;
|
|
11
|
+
var warnedNodeVersion = false;
|
|
12
|
+
function warnIfUnsupportedNode(version = process.versions.node) {
|
|
13
|
+
if (warnedNodeVersion) return;
|
|
14
|
+
const major = Number.parseInt(version, 10);
|
|
15
|
+
if (!Number.isInteger(major) || major >= MINIMUM_NODE_MAJOR) return;
|
|
16
|
+
warnedNodeVersion = true;
|
|
17
|
+
console.warn(
|
|
18
|
+
`inference: running on Node ${version}, older than the Node ${MINIMUM_NODE_MAJOR} this package requires. npm only warns about that at install time (EBADENGINE), so nothing has stopped you yet \u2014 upgrade Node, or expect failures this library cannot explain.`
|
|
19
|
+
);
|
|
20
|
+
}
|
|
21
|
+
function resetNodeVersionWarning() {
|
|
22
|
+
warnedNodeVersion = false;
|
|
23
|
+
}
|
|
24
|
+
|
|
9
25
|
// src/providers/anthropic.ts
|
|
10
26
|
import Anthropic from "@anthropic-ai/sdk";
|
|
11
27
|
var DEFAULT_TOOL_NAME = "record_result";
|
|
@@ -331,7 +347,15 @@ var ClaudeCliProvider = class {
|
|
|
331
347
|
`Claude CLI exited ${result.code}: ${result.stderr.trim().slice(-300)}`
|
|
332
348
|
);
|
|
333
349
|
}
|
|
334
|
-
|
|
350
|
+
let wrapper;
|
|
351
|
+
try {
|
|
352
|
+
wrapper = JSON.parse(result.stdout);
|
|
353
|
+
} catch {
|
|
354
|
+
const excerpt = result.stdout.trim().replace(/\s+/g, " ").slice(0, 200);
|
|
355
|
+
throw new Error(
|
|
356
|
+
`Claude CLI printed non-JSON output (is it logged in?): ${excerpt || "(no output)"}`
|
|
357
|
+
);
|
|
358
|
+
}
|
|
335
359
|
if (typeof wrapper.result !== "string") {
|
|
336
360
|
throw new Error("Claude CLI returned no result field");
|
|
337
361
|
}
|
|
@@ -384,12 +408,671 @@ function mockVerdict(match, confidence, overrides = {}) {
|
|
|
384
408
|
};
|
|
385
409
|
}
|
|
386
410
|
|
|
411
|
+
// src/cache.ts
|
|
412
|
+
import { createHash } from "crypto";
|
|
413
|
+
import { existsSync, mkdirSync, readFileSync, writeFileSync } from "fs";
|
|
414
|
+
import { join } from "path";
|
|
415
|
+
function sha256(text) {
|
|
416
|
+
return createHash("sha256").update(text, "utf8").digest("hex");
|
|
417
|
+
}
|
|
418
|
+
function buildCacheKey(parts) {
|
|
419
|
+
return sha256(parts.map((p) => `${p.length}:${p}`).join("|"));
|
|
420
|
+
}
|
|
421
|
+
var JsonCache = class {
|
|
422
|
+
constructor(dir, enabled = true, label = "inference") {
|
|
423
|
+
this.dir = dir;
|
|
424
|
+
this.enabled = enabled;
|
|
425
|
+
this.label = label;
|
|
426
|
+
}
|
|
427
|
+
dir;
|
|
428
|
+
enabled;
|
|
429
|
+
label;
|
|
430
|
+
/** Cache-write failures warn once per process, not once per entry. */
|
|
431
|
+
warned = false;
|
|
432
|
+
get(key) {
|
|
433
|
+
if (!this.enabled) return void 0;
|
|
434
|
+
const path = join(this.dir, `${key}.json`);
|
|
435
|
+
if (!existsSync(path)) return void 0;
|
|
436
|
+
try {
|
|
437
|
+
return JSON.parse(readFileSync(path, "utf8"));
|
|
438
|
+
} catch {
|
|
439
|
+
return void 0;
|
|
440
|
+
}
|
|
441
|
+
}
|
|
442
|
+
set(key, value) {
|
|
443
|
+
if (!this.enabled) return;
|
|
444
|
+
try {
|
|
445
|
+
mkdirSync(this.dir, { recursive: true });
|
|
446
|
+
writeFileSync(
|
|
447
|
+
join(this.dir, `${key}.json`),
|
|
448
|
+
JSON.stringify(value, null, 2)
|
|
449
|
+
);
|
|
450
|
+
} catch (e) {
|
|
451
|
+
if (!this.warned) {
|
|
452
|
+
this.warned = true;
|
|
453
|
+
console.warn(
|
|
454
|
+
`${this.label}: could not write the cache at ${this.dir} (${e instanceof Error ? e.message : String(e)}). Continuing without caching.`
|
|
455
|
+
);
|
|
456
|
+
}
|
|
457
|
+
}
|
|
458
|
+
}
|
|
459
|
+
};
|
|
460
|
+
|
|
461
|
+
// src/providers/llama-models.ts
|
|
462
|
+
import { readdirSync } from "fs";
|
|
463
|
+
import { homedir } from "os";
|
|
464
|
+
import { join as join2 } from "path";
|
|
465
|
+
function defaultLlamaModelsDirectory() {
|
|
466
|
+
return process.env["INFERENCE_MODELS_DIR"] || join2(homedir(), ".hawkeyexl-inference", "models");
|
|
467
|
+
}
|
|
468
|
+
var LLAMA_TIERS = ["fast", "balanced", "quality"];
|
|
469
|
+
var LLAMA_SELECTORS = ["auto", ...LLAMA_TIERS];
|
|
470
|
+
var LLAMA_MODELS = deepFreezeEntries({
|
|
471
|
+
"gemma-4-e2b": {
|
|
472
|
+
uri: "hf:unsloth/gemma-4-E2B-it-qat-GGUF/gemma-4-E2B-it-qat-UD-Q4_K_XL.gguf",
|
|
473
|
+
sizeBytes: 2620370976,
|
|
474
|
+
license: "Apache-2.0",
|
|
475
|
+
tier: "fast",
|
|
476
|
+
notes: "IFEval 94.6. Smallest Gemma 4; the floor for this family."
|
|
477
|
+
},
|
|
478
|
+
"gemma-4-e4b": {
|
|
479
|
+
uri: "hf:unsloth/gemma-4-E4B-it-qat-GGUF/gemma-4-E4B-it-qat-UD-Q4_K_XL.gguf",
|
|
480
|
+
sizeBytes: 4215695776,
|
|
481
|
+
license: "Apache-2.0",
|
|
482
|
+
tier: "balanced",
|
|
483
|
+
notes: "IFEval 96.7. The default for most machines."
|
|
484
|
+
},
|
|
485
|
+
"gemma-4-12b": {
|
|
486
|
+
uri: "hf:unsloth/gemma-4-12B-it-qat-GGUF/gemma-4-12B-it-qat-UD-Q4_K_XL.gguf",
|
|
487
|
+
sizeBytes: 6716356800,
|
|
488
|
+
license: "Apache-2.0",
|
|
489
|
+
tier: "quality",
|
|
490
|
+
notes: "IFEval 97.2. Dense 12B; wants a GPU or plenty of RAM."
|
|
491
|
+
},
|
|
492
|
+
"gemma-4-26b-a4b": {
|
|
493
|
+
uri: "hf:unsloth/gemma-4-26B-A4B-it-qat-GGUF/gemma-4-26B-A4B-it-qat-UD-Q4_K_XL.gguf",
|
|
494
|
+
sizeBytes: 14249047104,
|
|
495
|
+
license: "Apache-2.0",
|
|
496
|
+
notes: "MoE: 25.2B total, 3.8B active \u2014 infers near E4B speed if it fits in memory."
|
|
497
|
+
},
|
|
498
|
+
"gemma-4-e2b-q2": {
|
|
499
|
+
uri: "hf:unsloth/gemma-4-E2B-it-qat-GGUF/gemma-4-E2B-it-qat-UD-Q2_K_XL.gguf",
|
|
500
|
+
sizeBytes: 2186186784,
|
|
501
|
+
license: "Apache-2.0",
|
|
502
|
+
notes: "Q2 build of the fast tier; smallest download, lowest fidelity."
|
|
503
|
+
}
|
|
504
|
+
});
|
|
505
|
+
function deepFreezeEntries(catalog) {
|
|
506
|
+
for (const entry of Object.values(catalog)) Object.freeze(entry);
|
|
507
|
+
return Object.freeze(catalog);
|
|
508
|
+
}
|
|
509
|
+
var TIER_ALIAS = {
|
|
510
|
+
fast: "gemma-4-e2b",
|
|
511
|
+
balanced: "gemma-4-e4b",
|
|
512
|
+
quality: "gemma-4-12b"
|
|
513
|
+
};
|
|
514
|
+
function isLlamaSelector(model) {
|
|
515
|
+
return LLAMA_SELECTORS.includes(model);
|
|
516
|
+
}
|
|
517
|
+
var MEMORY_HEADROOM = 3.5;
|
|
518
|
+
function tierForBudget(budgetBytes) {
|
|
519
|
+
let chosen = "fast";
|
|
520
|
+
for (const tier of LLAMA_TIERS) {
|
|
521
|
+
const entry = LLAMA_MODELS[TIER_ALIAS[tier]];
|
|
522
|
+
if (entry.sizeBytes * MEMORY_HEADROOM <= budgetBytes) chosen = tier;
|
|
523
|
+
}
|
|
524
|
+
return chosen;
|
|
525
|
+
}
|
|
526
|
+
function aliasForTier(tier) {
|
|
527
|
+
return TIER_ALIAS[tier];
|
|
528
|
+
}
|
|
529
|
+
function uriForTier(tier) {
|
|
530
|
+
return LLAMA_MODELS[TIER_ALIAS[tier]].uri;
|
|
531
|
+
}
|
|
532
|
+
function resolveLlamaModelRef(model) {
|
|
533
|
+
if (isLlamaSelector(model)) {
|
|
534
|
+
throw new InferenceError(
|
|
535
|
+
`llama-cpp model "${model}" is a selector and needs a hardware probe to resolve. Use resolveProviderIdentityAsync/makeProviderAsync, or name a concrete model (e.g. "gemma-4-e4b").`
|
|
536
|
+
);
|
|
537
|
+
}
|
|
538
|
+
const entry = LLAMA_MODELS[model];
|
|
539
|
+
if (entry) return entry.uri;
|
|
540
|
+
if (isModelPathOrUri(model)) return model;
|
|
541
|
+
throw new InferenceError(
|
|
542
|
+
`Unknown llama-cpp model "${model}". Use a selector (${LLAMA_SELECTORS.join(
|
|
543
|
+
", "
|
|
544
|
+
)}), a curated alias (${Object.keys(LLAMA_MODELS).join(
|
|
545
|
+
", "
|
|
546
|
+
)}), an hf: URI, or a path to a .gguf file.`
|
|
547
|
+
);
|
|
548
|
+
}
|
|
549
|
+
function blobNameFor(model) {
|
|
550
|
+
const ref = resolveLlamaModelRef(model).split("#")[0];
|
|
551
|
+
return ref.split(/[/\\]/).pop();
|
|
552
|
+
}
|
|
553
|
+
function matchesModelBlob(entry, blobName) {
|
|
554
|
+
const base = entry.replace(/\.ipull$/, "");
|
|
555
|
+
if (base.endsWith(blobName)) return true;
|
|
556
|
+
const stem = blobName.replace(/\.gguf$/, "");
|
|
557
|
+
return new RegExp(`${escapeRegExp(stem)}-\\d{5}-of-\\d{5}\\.gguf$`).test(base);
|
|
558
|
+
}
|
|
559
|
+
function escapeRegExp(text) {
|
|
560
|
+
return text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
561
|
+
}
|
|
562
|
+
function isModelDownloaded(model, directory) {
|
|
563
|
+
const blobName = blobNameFor(model);
|
|
564
|
+
return listModelDirectory(directory).some(
|
|
565
|
+
(entry) => !entry.endsWith(".ipull") && matchesModelBlob(entry, blobName)
|
|
566
|
+
);
|
|
567
|
+
}
|
|
568
|
+
function listModelDirectory(directory) {
|
|
569
|
+
try {
|
|
570
|
+
return readdirSync(directory, { withFileTypes: true }).filter((entry) => entry.isFile()).map((entry) => entry.name);
|
|
571
|
+
} catch {
|
|
572
|
+
return [];
|
|
573
|
+
}
|
|
574
|
+
}
|
|
575
|
+
function isModelPathOrUri(model) {
|
|
576
|
+
return /^(hf|huggingface):/i.test(model) || /^https?:\/\//i.test(model) || /^(hf|huggingface)\.co\//i.test(model) || model.endsWith(".gguf");
|
|
577
|
+
}
|
|
578
|
+
|
|
579
|
+
// src/providers/llama-install.ts
|
|
580
|
+
import {
|
|
581
|
+
existsSync as existsSync2,
|
|
582
|
+
mkdirSync as mkdirSync2,
|
|
583
|
+
rmSync,
|
|
584
|
+
statSync,
|
|
585
|
+
writeFileSync as writeFileSync2
|
|
586
|
+
} from "fs";
|
|
587
|
+
import { homedir as homedir2 } from "os";
|
|
588
|
+
import { join as join3 } from "path";
|
|
589
|
+
import { pathToFileURL } from "url";
|
|
590
|
+
var PACKAGE_SPEC = "node-llama-cpp@^3.19.0";
|
|
591
|
+
var SHIM = "loader.mjs";
|
|
592
|
+
var LOCK = ".install.lock";
|
|
593
|
+
var INSTALL_TIMEOUT_MS = 9e5;
|
|
594
|
+
var LOCK_WAIT_MS = INSTALL_TIMEOUT_MS;
|
|
595
|
+
var LOCK_STALE_MS = INSTALL_TIMEOUT_MS + 6e4;
|
|
596
|
+
function defaultLlamaRuntimeDirectory(env = process.env) {
|
|
597
|
+
return env["INFERENCE_RUNTIME_DIR"] || join3(homedir2(), ".hawkeyexl-inference", "runtime");
|
|
598
|
+
}
|
|
599
|
+
function isModuleNotFound(e) {
|
|
600
|
+
const code = e?.code;
|
|
601
|
+
return code === "ERR_MODULE_NOT_FOUND" || code === "MODULE_NOT_FOUND";
|
|
602
|
+
}
|
|
603
|
+
function describe(e) {
|
|
604
|
+
return e instanceof Error ? e.message : String(e);
|
|
605
|
+
}
|
|
606
|
+
async function nodeLlamaCppStatus(options = {}) {
|
|
607
|
+
const env = options.env ?? process.env;
|
|
608
|
+
const directory = options.directory ?? defaultLlamaRuntimeDirectory(env);
|
|
609
|
+
const probeImport = options.probeImport ?? (() => import("node-llama-cpp"));
|
|
610
|
+
try {
|
|
611
|
+
await probeImport();
|
|
612
|
+
return { state: "present" };
|
|
613
|
+
} catch (e) {
|
|
614
|
+
if (!isModuleNotFound(e)) {
|
|
615
|
+
return {
|
|
616
|
+
state: "refused",
|
|
617
|
+
reason: `node-llama-cpp is installed but failed to load (${describe(e)})`
|
|
618
|
+
};
|
|
619
|
+
}
|
|
620
|
+
}
|
|
621
|
+
if (existsSync2(join3(directory, SHIM))) return { state: "present" };
|
|
622
|
+
if ((env["INFERENCE_NO_AUTO_INSTALL"] ?? "") !== "") {
|
|
623
|
+
return {
|
|
624
|
+
state: "refused",
|
|
625
|
+
reason: `node-llama-cpp is not installed and INFERENCE_NO_AUTO_INSTALL is set`
|
|
626
|
+
};
|
|
627
|
+
}
|
|
628
|
+
return { state: "installable", directory };
|
|
629
|
+
}
|
|
630
|
+
var installs = /* @__PURE__ */ new Map();
|
|
631
|
+
var warnedInstall = false;
|
|
632
|
+
function resetRuntimeInstall() {
|
|
633
|
+
installs.clear();
|
|
634
|
+
warnedInstall = false;
|
|
635
|
+
}
|
|
636
|
+
function importNodeLlamaCpp(options = {}) {
|
|
637
|
+
const env = options.env ?? process.env;
|
|
638
|
+
const directory = options.directory ?? defaultLlamaRuntimeDirectory(env);
|
|
639
|
+
const existing = installs.get(directory);
|
|
640
|
+
if (existing) return existing;
|
|
641
|
+
const pending = fromPrefix(directory, env, options);
|
|
642
|
+
const guarded = pending.catch((e) => {
|
|
643
|
+
if (installs.get(directory) === guarded) installs.delete(directory);
|
|
644
|
+
throw e;
|
|
645
|
+
});
|
|
646
|
+
installs.set(directory, guarded);
|
|
647
|
+
return guarded;
|
|
648
|
+
}
|
|
649
|
+
async function fromPrefix(directory, env, options) {
|
|
650
|
+
const importShim = options.importShim ?? ((url) => import(url));
|
|
651
|
+
const shim = join3(directory, SHIM);
|
|
652
|
+
if (existsSync2(shim)) return importShim(pathToFileURL(shim).href);
|
|
653
|
+
if ((env["INFERENCE_NO_AUTO_INSTALL"] ?? "") !== "") {
|
|
654
|
+
throw new InferenceError(
|
|
655
|
+
`The llama-cpp provider needs node-llama-cpp, and INFERENCE_NO_AUTO_INSTALL is set. Install it yourself (npm i ${PACKAGE_SPEC}), unset INFERENCE_NO_AUTO_INSTALL to allow installing into ${directory}, or name a different provider.`
|
|
656
|
+
);
|
|
657
|
+
}
|
|
658
|
+
mkdirSync2(directory, { recursive: true });
|
|
659
|
+
await withLock(directory, async () => {
|
|
660
|
+
if (existsSync2(shim)) return;
|
|
661
|
+
warnInstalling(directory);
|
|
662
|
+
await runInstall(directory, env, options);
|
|
663
|
+
writeFileSync2(shim, `export * from "node-llama-cpp";
|
|
664
|
+
`, "utf8");
|
|
665
|
+
});
|
|
666
|
+
return importShim(pathToFileURL(shim).href);
|
|
667
|
+
}
|
|
668
|
+
async function runInstall(directory, env, options) {
|
|
669
|
+
const manifest = join3(directory, "package.json");
|
|
670
|
+
if (!existsSync2(manifest)) {
|
|
671
|
+
writeFileSync2(
|
|
672
|
+
manifest,
|
|
673
|
+
`${JSON.stringify(
|
|
674
|
+
{
|
|
675
|
+
name: "hawkeyexl-inference-runtime",
|
|
676
|
+
version: "0.0.0",
|
|
677
|
+
private: true,
|
|
678
|
+
description: "Auto-installed runtime for @hawkeyexl/inference. Safe to delete."
|
|
679
|
+
},
|
|
680
|
+
null,
|
|
681
|
+
2
|
|
682
|
+
)}
|
|
683
|
+
`,
|
|
684
|
+
"utf8"
|
|
685
|
+
);
|
|
686
|
+
}
|
|
687
|
+
const exec = options.exec ?? realExec;
|
|
688
|
+
const result = await exec(
|
|
689
|
+
[
|
|
690
|
+
"npm",
|
|
691
|
+
"install",
|
|
692
|
+
"--prefix",
|
|
693
|
+
directory,
|
|
694
|
+
PACKAGE_SPEC,
|
|
695
|
+
"--no-audit",
|
|
696
|
+
"--no-fund"
|
|
697
|
+
],
|
|
698
|
+
{ timeoutMs: options.timeoutMs ?? INSTALL_TIMEOUT_MS, env }
|
|
699
|
+
);
|
|
700
|
+
if (result.spawnError != null) {
|
|
701
|
+
throw new InferenceError(
|
|
702
|
+
`Could not run npm to install node-llama-cpp (${result.spawnError}). Install it yourself with: npm i ${PACKAGE_SPEC}, or name a provider that does not need it.`
|
|
703
|
+
);
|
|
704
|
+
}
|
|
705
|
+
if (result.timedOut) {
|
|
706
|
+
throw new InferenceError(
|
|
707
|
+
`Installing node-llama-cpp into ${directory} timed out. A source build can take a while \u2014 retry, raise the timeout, or install it yourself with: npm i ${PACKAGE_SPEC}.`
|
|
708
|
+
);
|
|
709
|
+
}
|
|
710
|
+
if (result.code !== 0) {
|
|
711
|
+
throw new InferenceError(
|
|
712
|
+
`Installing node-llama-cpp into ${directory} failed (exit ${String(
|
|
713
|
+
result.code
|
|
714
|
+
)}).
|
|
715
|
+
${tail(result.stderr || result.stdout)}
|
|
716
|
+
Install it yourself with: npm i ${PACKAGE_SPEC}, or name a provider that does not need it.`
|
|
717
|
+
);
|
|
718
|
+
}
|
|
719
|
+
}
|
|
720
|
+
function tail(output, lines = 12) {
|
|
721
|
+
return output.trimEnd().split(/\r?\n/).slice(-lines).join("\n");
|
|
722
|
+
}
|
|
723
|
+
function warnInstalling(directory) {
|
|
724
|
+
if (warnedInstall) return;
|
|
725
|
+
warnedInstall = true;
|
|
726
|
+
console.warn(
|
|
727
|
+
`inference: node-llama-cpp is not installed \u2014 fetching it into ${directory} so the local model can run. This is a one-time native install; set INFERENCE_NO_AUTO_INSTALL=1 to refuse it, or name a provider that does not need it.`
|
|
728
|
+
);
|
|
729
|
+
}
|
|
730
|
+
async function withLock(directory, fn) {
|
|
731
|
+
const lock = join3(directory, LOCK);
|
|
732
|
+
const deadline = Date.now() + LOCK_WAIT_MS;
|
|
733
|
+
for (; ; ) {
|
|
734
|
+
try {
|
|
735
|
+
writeFileSync2(lock, String(process.pid), { flag: "wx" });
|
|
736
|
+
break;
|
|
737
|
+
} catch (e) {
|
|
738
|
+
if (e.code !== "EEXIST") throw e;
|
|
739
|
+
if (ageOf(lock) > LOCK_STALE_MS) {
|
|
740
|
+
rmSync(lock, { force: true });
|
|
741
|
+
continue;
|
|
742
|
+
}
|
|
743
|
+
if (existsSync2(join3(directory, SHIM))) return;
|
|
744
|
+
if (Date.now() > deadline) {
|
|
745
|
+
throw new InferenceError(
|
|
746
|
+
`Timed out waiting for another process to install node-llama-cpp into ${directory}. If nothing else is running, remove ${lock} and retry.`
|
|
747
|
+
);
|
|
748
|
+
}
|
|
749
|
+
await delay(250);
|
|
750
|
+
}
|
|
751
|
+
}
|
|
752
|
+
try {
|
|
753
|
+
await fn();
|
|
754
|
+
} finally {
|
|
755
|
+
rmSync(lock, { force: true });
|
|
756
|
+
}
|
|
757
|
+
}
|
|
758
|
+
function ageOf(path) {
|
|
759
|
+
try {
|
|
760
|
+
return Date.now() - statSync(path).mtimeMs;
|
|
761
|
+
} catch {
|
|
762
|
+
return Number.POSITIVE_INFINITY;
|
|
763
|
+
}
|
|
764
|
+
}
|
|
765
|
+
function delay(ms) {
|
|
766
|
+
return new Promise((resolve) => setTimeout(resolve, ms));
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
// src/providers/llama-cpp.ts
|
|
770
|
+
var loadedModels = /* @__PURE__ */ new Map();
|
|
771
|
+
async function disposeLlamaModels() {
|
|
772
|
+
const pending = [...loadedModels.values()];
|
|
773
|
+
loadedModels.clear();
|
|
774
|
+
await Promise.all(
|
|
775
|
+
pending.map((p) => p.then((m) => m.dispose()).catch(() => void 0))
|
|
776
|
+
);
|
|
777
|
+
}
|
|
778
|
+
var LlamaCppProvider = class {
|
|
779
|
+
constructor(model, options = {}) {
|
|
780
|
+
this.model = model;
|
|
781
|
+
if (isLlamaSelector(model)) {
|
|
782
|
+
throw new InferenceError(
|
|
783
|
+
`llama-cpp model "${model}" is a selector. Constructing a provider directly needs a concrete model (e.g. "gemma-4-e4b") \u2014 use makeProviderAsync to resolve a selector against this machine.`
|
|
784
|
+
);
|
|
785
|
+
}
|
|
786
|
+
this.uri = resolveLlamaModelRef(model);
|
|
787
|
+
this.runtime = options.runtime ?? defaultLlamaRuntime();
|
|
788
|
+
this.thoughtTokens = options.thoughtTokens ?? 0;
|
|
789
|
+
this.maxTokens = options.maxTokens;
|
|
790
|
+
this.modelsDirectory = options.modelsDirectory ?? defaultLlamaModelsDirectory();
|
|
791
|
+
this.cacheKey = buildCacheKey([this.modelsDirectory, this.uri]);
|
|
792
|
+
}
|
|
793
|
+
model;
|
|
794
|
+
uri;
|
|
795
|
+
runtime;
|
|
796
|
+
thoughtTokens;
|
|
797
|
+
maxTokens;
|
|
798
|
+
modelsDirectory;
|
|
799
|
+
/**
|
|
800
|
+
* Loaded-model key: the same URI in two directories is two different files.
|
|
801
|
+
* Built with `buildCacheKey` so its parts are length-prefixed — a plain join
|
|
802
|
+
* would let two different (directory, uri) pairs collide and hand a provider
|
|
803
|
+
* back the wrong weights.
|
|
804
|
+
*/
|
|
805
|
+
cacheKey;
|
|
806
|
+
provider() {
|
|
807
|
+
return "llama-cpp";
|
|
808
|
+
}
|
|
809
|
+
modelName() {
|
|
810
|
+
return this.model;
|
|
811
|
+
}
|
|
812
|
+
async completeJSON(req) {
|
|
813
|
+
const model = await this.load();
|
|
814
|
+
const session = await model.createSession(systemPromptFor(req));
|
|
815
|
+
try {
|
|
816
|
+
const result = await session.prompt(req.user, {
|
|
817
|
+
schema: req.schema,
|
|
818
|
+
temperature: req.temperature,
|
|
819
|
+
thoughtTokens: this.thoughtTokens,
|
|
820
|
+
...this.maxTokens != null ? { maxTokens: this.maxTokens } : {}
|
|
821
|
+
});
|
|
822
|
+
if (result.stopReason === "maxTokens") {
|
|
823
|
+
throw new Error(
|
|
824
|
+
`llama-cpp generation hit the token limit before completing the JSON${this.maxTokens != null ? ` (maxTokens: ${this.maxTokens})` : ""} \u2014 raise llamaCpp.maxTokens, or shorten the prompt if the context is full.`
|
|
825
|
+
);
|
|
826
|
+
}
|
|
827
|
+
return { json: extractJson(result.text), usage: result.usage };
|
|
828
|
+
} finally {
|
|
829
|
+
await session.dispose().catch(() => void 0);
|
|
830
|
+
}
|
|
831
|
+
}
|
|
832
|
+
load() {
|
|
833
|
+
const existing = loadedModels.get(this.cacheKey);
|
|
834
|
+
if (existing) return existing;
|
|
835
|
+
const pending = (async () => {
|
|
836
|
+
const path = await this.runtime.resolveModelFile(
|
|
837
|
+
this.uri,
|
|
838
|
+
this.modelsDirectory
|
|
839
|
+
);
|
|
840
|
+
return this.runtime.loadModel(path);
|
|
841
|
+
})();
|
|
842
|
+
const guarded = pending.catch((e) => {
|
|
843
|
+
if (loadedModels.get(this.cacheKey) === guarded) {
|
|
844
|
+
loadedModels.delete(this.cacheKey);
|
|
845
|
+
}
|
|
846
|
+
throw e;
|
|
847
|
+
});
|
|
848
|
+
loadedModels.set(this.cacheKey, guarded);
|
|
849
|
+
return guarded;
|
|
850
|
+
}
|
|
851
|
+
};
|
|
852
|
+
function systemPromptFor(req) {
|
|
853
|
+
return `${req.system}
|
|
854
|
+
|
|
855
|
+
Respond with ONLY a JSON object conforming to this JSON Schema:
|
|
856
|
+
${JSON.stringify(
|
|
857
|
+
req.schema
|
|
858
|
+
)}`;
|
|
859
|
+
}
|
|
860
|
+
var runtimePromise;
|
|
861
|
+
function defaultLlamaRuntime() {
|
|
862
|
+
const real = () => (
|
|
863
|
+
// Drop a failed init so the next call retries. A GPU that failed to
|
|
864
|
+
// initialise, or a binary still being extracted by a concurrent install,
|
|
865
|
+
// must not poison the runtime for the rest of the process — the same rule
|
|
866
|
+
// `load()` applies to weights.
|
|
867
|
+
runtimePromise ??= loadNodeLlamaCpp().catch((e) => {
|
|
868
|
+
runtimePromise = void 0;
|
|
869
|
+
throw e;
|
|
870
|
+
})
|
|
871
|
+
);
|
|
872
|
+
return {
|
|
873
|
+
resolveModelFile: (uri, directory) => real().then((r) => r.resolveModelFile(uri, directory)),
|
|
874
|
+
loadModel: (path) => real().then((r) => r.loadModel(path)),
|
|
875
|
+
getMemoryBudgetBytes: () => real().then((r) => r.getMemoryBudgetBytes())
|
|
876
|
+
};
|
|
877
|
+
}
|
|
878
|
+
async function loadNodeLlamaCpp() {
|
|
879
|
+
let mod;
|
|
880
|
+
try {
|
|
881
|
+
mod = await import("node-llama-cpp");
|
|
882
|
+
} catch (e) {
|
|
883
|
+
if (!isModuleNotFound(e)) {
|
|
884
|
+
throw new InferenceError(
|
|
885
|
+
`node-llama-cpp is installed but failed to load (${e instanceof Error ? e.message : String(e)}). This is the copy resolved from your own node_modules, so reinstalling it here will not help \u2014 check the Node version and the platform build.`
|
|
886
|
+
);
|
|
887
|
+
}
|
|
888
|
+
mod = await importNodeLlamaCpp();
|
|
889
|
+
}
|
|
890
|
+
const { getLlama, resolveModelFile, LlamaChatSession, TokenMeter } = mod;
|
|
891
|
+
const llama = await getLlama();
|
|
892
|
+
return {
|
|
893
|
+
// `directory` is this library's own, not node-llama-cpp's global default —
|
|
894
|
+
// owning it is what makes `clearLlamaModels` safe.
|
|
895
|
+
resolveModelFile: (uri, directory) => resolveModelFile(uri, { directory }),
|
|
896
|
+
async loadModel(path) {
|
|
897
|
+
const model = await llama.loadModel({ modelPath: path });
|
|
898
|
+
return {
|
|
899
|
+
async createSession(systemPrompt) {
|
|
900
|
+
const context = await model.createContext();
|
|
901
|
+
const sequence = context.getSequence();
|
|
902
|
+
const session = new LlamaChatSession({
|
|
903
|
+
contextSequence: sequence,
|
|
904
|
+
systemPrompt
|
|
905
|
+
});
|
|
906
|
+
return {
|
|
907
|
+
async prompt(text, options) {
|
|
908
|
+
const grammar = await llama.createGrammarForJsonSchema(
|
|
909
|
+
options.schema
|
|
910
|
+
);
|
|
911
|
+
const before = sequence.tokenMeter.getState();
|
|
912
|
+
const result = await session.promptWithMeta(text, {
|
|
913
|
+
grammar,
|
|
914
|
+
temperature: options.temperature,
|
|
915
|
+
budgets: { thoughtTokens: options.thoughtTokens },
|
|
916
|
+
...options.maxTokens != null ? { maxTokens: options.maxTokens } : {}
|
|
917
|
+
});
|
|
918
|
+
const diff = TokenMeter.diff(sequence.tokenMeter, before);
|
|
919
|
+
return {
|
|
920
|
+
text: result.responseText,
|
|
921
|
+
stopReason: result.stopReason,
|
|
922
|
+
usage: {
|
|
923
|
+
inputTokens: diff.usedInputTokens,
|
|
924
|
+
outputTokens: diff.usedOutputTokens
|
|
925
|
+
}
|
|
926
|
+
};
|
|
927
|
+
},
|
|
928
|
+
async dispose() {
|
|
929
|
+
await context.dispose();
|
|
930
|
+
}
|
|
931
|
+
};
|
|
932
|
+
},
|
|
933
|
+
async dispose() {
|
|
934
|
+
await model.dispose();
|
|
935
|
+
}
|
|
936
|
+
};
|
|
937
|
+
},
|
|
938
|
+
async getMemoryBudgetBytes() {
|
|
939
|
+
const { totalmem } = await import("os");
|
|
940
|
+
const ramBudget = totalmem() / 2;
|
|
941
|
+
try {
|
|
942
|
+
const vram = await llama.getVramState();
|
|
943
|
+
return Math.max(vram.free, ramBudget);
|
|
944
|
+
} catch {
|
|
945
|
+
return ramBudget;
|
|
946
|
+
}
|
|
947
|
+
}
|
|
948
|
+
};
|
|
949
|
+
}
|
|
950
|
+
|
|
951
|
+
// src/providers/detect.ts
|
|
952
|
+
var DETECTION_ORDER = [
|
|
953
|
+
"anthropic",
|
|
954
|
+
"openai",
|
|
955
|
+
"claude-cli",
|
|
956
|
+
"llama-cpp"
|
|
957
|
+
];
|
|
958
|
+
var DEFAULT_KEY_ENV = {
|
|
959
|
+
anthropic: "ANTHROPIC_API_KEY",
|
|
960
|
+
openai: "OPENAI_API_KEY"
|
|
961
|
+
};
|
|
962
|
+
function hasKey(provider) {
|
|
963
|
+
return (process.env[DEFAULT_KEY_ENV[provider]] ?? "") !== "";
|
|
964
|
+
}
|
|
965
|
+
var cliProbes = /* @__PURE__ */ new Map();
|
|
966
|
+
function resetClaudeCliProbe() {
|
|
967
|
+
cliProbes.clear();
|
|
968
|
+
}
|
|
969
|
+
function probeClaudeCli(spec) {
|
|
970
|
+
const exec = spec.exec ?? realExec;
|
|
971
|
+
const command = spec.command ?? "claude";
|
|
972
|
+
const cached = cliProbes.get(command);
|
|
973
|
+
if (cached) return cached;
|
|
974
|
+
const probing = exec([command, "--version"], { timeoutMs: 1e4 }).then((r) => r.code === 0 && !r.timedOut && r.spawnError == null).catch(() => false);
|
|
975
|
+
cliProbes.set(command, probing);
|
|
976
|
+
return probing;
|
|
977
|
+
}
|
|
978
|
+
async function probeLlamaCpp(spec) {
|
|
979
|
+
const injected = spec.llamaRuntime ?? spec.llamaCpp?.runtime;
|
|
980
|
+
if (injected) return budgetProbe(injected);
|
|
981
|
+
const status = await nodeLlamaCppStatus();
|
|
982
|
+
if (status.state === "refused") {
|
|
983
|
+
return { available: false, reason: status.reason };
|
|
984
|
+
}
|
|
985
|
+
if (status.state === "installable") {
|
|
986
|
+
return { available: true };
|
|
987
|
+
}
|
|
988
|
+
return budgetProbe(defaultLlamaRuntime());
|
|
989
|
+
}
|
|
990
|
+
function budgetProbe(runtime) {
|
|
991
|
+
return runtime.getMemoryBudgetBytes().then(
|
|
992
|
+
() => ({ available: true }),
|
|
993
|
+
(e) => ({
|
|
994
|
+
available: false,
|
|
995
|
+
reason: e instanceof Error && /node-llama-cpp/.test(e.message) ? "node-llama-cpp is not installed (npm i node-llama-cpp)" : `node-llama-cpp could not start (${e instanceof Error ? e.message : String(e)})`
|
|
996
|
+
})
|
|
997
|
+
);
|
|
998
|
+
}
|
|
999
|
+
async function probe(provider, spec) {
|
|
1000
|
+
switch (provider) {
|
|
1001
|
+
case "anthropic":
|
|
1002
|
+
return hasKey("anthropic") ? { available: true } : {
|
|
1003
|
+
available: false,
|
|
1004
|
+
reason: `ANTHROPIC_API_KEY is not set`
|
|
1005
|
+
};
|
|
1006
|
+
case "openai":
|
|
1007
|
+
return hasKey("openai") || spec.baseUrl ? { available: true } : {
|
|
1008
|
+
available: false,
|
|
1009
|
+
reason: `OPENAI_API_KEY is not set and no baseUrl was given`
|
|
1010
|
+
};
|
|
1011
|
+
case "claude-cli":
|
|
1012
|
+
return await probeClaudeCli(spec) ? { available: true } : {
|
|
1013
|
+
available: false,
|
|
1014
|
+
reason: `could not run \`${spec.command ?? "claude"}\` (is the Claude CLI installed?)`
|
|
1015
|
+
};
|
|
1016
|
+
case "llama-cpp":
|
|
1017
|
+
return probeLlamaCpp(spec);
|
|
1018
|
+
default:
|
|
1019
|
+
return { available: false, reason: "not auto-selectable" };
|
|
1020
|
+
}
|
|
1021
|
+
}
|
|
1022
|
+
async function availableProviders(spec = {}) {
|
|
1023
|
+
const probes = await Promise.all(
|
|
1024
|
+
DETECTION_ORDER.map((name) => probe(name, spec))
|
|
1025
|
+
);
|
|
1026
|
+
return DETECTION_ORDER.filter((_, i) => probes[i].available);
|
|
1027
|
+
}
|
|
1028
|
+
async function detectProvider(spec = {}) {
|
|
1029
|
+
const reasons = [];
|
|
1030
|
+
for (const name of DETECTION_ORDER) {
|
|
1031
|
+
const result = await probe(name, spec);
|
|
1032
|
+
if (result.available) {
|
|
1033
|
+
warnSelected(name, spec.provider === "auto");
|
|
1034
|
+
return name;
|
|
1035
|
+
}
|
|
1036
|
+
reasons.push(` ${name.padEnd(10)} \u2014 ${result.reason}`);
|
|
1037
|
+
}
|
|
1038
|
+
throw new InferenceError(
|
|
1039
|
+
`No inference provider is available. Tried:
|
|
1040
|
+
${reasons.join("\n")}
|
|
1041
|
+
Pass an explicit \`provider\`, set one of the keys above, or install node-llama-cpp.`
|
|
1042
|
+
);
|
|
1043
|
+
}
|
|
1044
|
+
var warnedSelection = false;
|
|
1045
|
+
var warnedDownload = false;
|
|
1046
|
+
function resetProviderDetectionWarning() {
|
|
1047
|
+
warnedSelection = false;
|
|
1048
|
+
warnedDownload = false;
|
|
1049
|
+
}
|
|
1050
|
+
function warnSelected(provider, wasExplicitAuto) {
|
|
1051
|
+
if (warnedSelection) return;
|
|
1052
|
+
warnedSelection = true;
|
|
1053
|
+
const because = wasExplicitAuto ? `provider "auto"` : "no provider specified";
|
|
1054
|
+
console.warn(
|
|
1055
|
+
`inference: ${because} \u2014 auto-selected "${provider}". Pass an explicit \`provider\` to pin it.`
|
|
1056
|
+
);
|
|
1057
|
+
}
|
|
1058
|
+
function warnPendingDownload(model, sizeBytes) {
|
|
1059
|
+
if (warnedDownload) return;
|
|
1060
|
+
warnedDownload = true;
|
|
1061
|
+
console.warn(
|
|
1062
|
+
`inference: "${model}" is not downloaded yet \u2014 the first run will fetch ~${(sizeBytes / 1e9).toFixed(2)} GB. Pre-fetch it, or pass an explicit \`provider\` to avoid the local model entirely.`
|
|
1063
|
+
);
|
|
1064
|
+
}
|
|
1065
|
+
|
|
387
1066
|
// src/providers/index.ts
|
|
388
1067
|
var DEFAULT_MODELS = {
|
|
389
1068
|
anthropic: "claude-sonnet-4-5",
|
|
390
1069
|
openai: "gpt-4o-mini",
|
|
391
1070
|
"claude-cli": "claude-sonnet-4-5",
|
|
392
|
-
mock: "mock-model"
|
|
1071
|
+
mock: "mock-model",
|
|
1072
|
+
// A selector, not a pinned model: which weights a tier points at is then a
|
|
1073
|
+
// catalog change rather than an API change. Resolving it needs the async
|
|
1074
|
+
// factory — see `resolveProviderIdentityAsync`.
|
|
1075
|
+
"llama-cpp": "auto"
|
|
393
1076
|
};
|
|
394
1077
|
var DEFAULT_API_KEY_ENV = {
|
|
395
1078
|
anthropic: "ANTHROPIC_API_KEY",
|
|
@@ -397,12 +1080,51 @@ var DEFAULT_API_KEY_ENV = {
|
|
|
397
1080
|
};
|
|
398
1081
|
var DEFAULT_OPENAI_BASE_URL = "https://api.openai.com/v1";
|
|
399
1082
|
function resolveProviderIdentity(spec) {
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
1083
|
+
if (spec.provider == null || spec.provider === "auto") {
|
|
1084
|
+
throw new InferenceError(
|
|
1085
|
+
`No provider specified. Detecting one probes the environment, the Claude CLI and the local model runtime, which cannot be done synchronously \u2014 use resolveProviderIdentityAsync/makeProviderAsync, or name a provider (${Object.keys(DEFAULT_MODELS).join(", ")}).`
|
|
1086
|
+
);
|
|
1087
|
+
}
|
|
1088
|
+
const model = spec.model ?? DEFAULT_MODELS[spec.provider] ?? "unknown";
|
|
1089
|
+
if (spec.provider === "llama-cpp" && isLlamaSelector(model)) {
|
|
1090
|
+
throw new InferenceError(
|
|
1091
|
+
`llama-cpp model "${model}" is a selector and cannot be resolved synchronously \u2014 picking a tier probes GPU memory. Use resolveProviderIdentityAsync/makeProviderAsync, or name a concrete model (e.g. "gemma-4-e4b").`
|
|
1092
|
+
);
|
|
1093
|
+
}
|
|
1094
|
+
return { provider: spec.provider, model };
|
|
1095
|
+
}
|
|
1096
|
+
async function resolveProviderIdentityAsync(spec) {
|
|
1097
|
+
if ((spec.provider == null || spec.provider === "auto") && spec.model != null) {
|
|
1098
|
+
throw new InferenceError(
|
|
1099
|
+
`Model "${spec.model}" was given without a provider, and a model name does not say which provider owns it. Name the provider too (${Object.keys(DEFAULT_MODELS).join(", ")}), or drop the model to take the detected provider's default.`
|
|
1100
|
+
);
|
|
1101
|
+
}
|
|
1102
|
+
const provider = spec.provider == null || spec.provider === "auto" ? await detectProvider(spec) : spec.provider;
|
|
1103
|
+
const resolved = { ...spec, provider };
|
|
1104
|
+
const model = spec.model ?? DEFAULT_MODELS[provider] ?? "unknown";
|
|
1105
|
+
if (provider !== "llama-cpp" || !isLlamaSelector(model)) {
|
|
1106
|
+
return resolveProviderIdentity(resolved);
|
|
1107
|
+
}
|
|
1108
|
+
const tier = model === "auto" ? await probeTier(llamaRuntimeFor(spec)) : model;
|
|
1109
|
+
return { provider, model: aliasForTier(tier) };
|
|
1110
|
+
}
|
|
1111
|
+
function warnIfDownloadPending(spec, model) {
|
|
1112
|
+
const entry = LLAMA_MODELS[model];
|
|
1113
|
+
if (!entry) return;
|
|
1114
|
+
const directory = spec.llamaCpp?.modelsDirectory ?? defaultLlamaModelsDirectory();
|
|
1115
|
+
if (!isModelDownloaded(model, directory)) {
|
|
1116
|
+
warnPendingDownload(model, entry.sizeBytes);
|
|
1117
|
+
}
|
|
1118
|
+
}
|
|
1119
|
+
function llamaRuntimeFor(spec) {
|
|
1120
|
+
return spec.llamaRuntime ?? spec.llamaCpp?.runtime;
|
|
1121
|
+
}
|
|
1122
|
+
async function probeTier(runtime) {
|
|
1123
|
+
const source = runtime ?? defaultLlamaRuntime();
|
|
1124
|
+
return tierForBudget(await source.getMemoryBudgetBytes());
|
|
404
1125
|
}
|
|
405
1126
|
function makeProvider(spec) {
|
|
1127
|
+
warnIfUnsupportedNode();
|
|
406
1128
|
const { model } = resolveProviderIdentity(spec);
|
|
407
1129
|
switch (spec.provider) {
|
|
408
1130
|
case "anthropic":
|
|
@@ -428,6 +1150,11 @@ function makeProvider(spec) {
|
|
|
428
1150
|
);
|
|
429
1151
|
case "mock":
|
|
430
1152
|
return new MockProvider(spec.mockResponses ?? [{ json: {} }], model);
|
|
1153
|
+
case "llama-cpp":
|
|
1154
|
+
return new LlamaCppProvider(model, {
|
|
1155
|
+
...spec.llamaCpp ?? {},
|
|
1156
|
+
...spec.llamaRuntime ? { runtime: spec.llamaRuntime } : {}
|
|
1157
|
+
});
|
|
431
1158
|
default:
|
|
432
1159
|
throw new InferenceError(
|
|
433
1160
|
`Unknown provider "${String(spec.provider)}". Available: ${Object.keys(
|
|
@@ -436,6 +1163,55 @@ function makeProvider(spec) {
|
|
|
436
1163
|
);
|
|
437
1164
|
}
|
|
438
1165
|
}
|
|
1166
|
+
async function makeProviderAsync(spec) {
|
|
1167
|
+
const { provider, model } = await resolveProviderIdentityAsync(spec);
|
|
1168
|
+
if (provider === "llama-cpp") warnIfDownloadPending(spec, model);
|
|
1169
|
+
return makeProvider({ ...spec, provider, model });
|
|
1170
|
+
}
|
|
1171
|
+
|
|
1172
|
+
// src/providers/llama-clean.ts
|
|
1173
|
+
import { rmSync as rmSync2, statSync as statSync2 } from "fs";
|
|
1174
|
+
import { join as join4 } from "path";
|
|
1175
|
+
async function clearLlamaModels(options = {}) {
|
|
1176
|
+
const directory = options.directory ?? defaultLlamaModelsDirectory();
|
|
1177
|
+
const dryRun = options.dryRun ?? false;
|
|
1178
|
+
const wanted = options.models?.map(blobNameFor);
|
|
1179
|
+
if (!dryRun) await disposeLlamaModels();
|
|
1180
|
+
const files = [];
|
|
1181
|
+
for (const entry of listModelDirectory(directory)) {
|
|
1182
|
+
if (!isModelBlob(entry)) continue;
|
|
1183
|
+
if (wanted && !wanted.some((name) => matchesModelBlob(entry, name))) {
|
|
1184
|
+
continue;
|
|
1185
|
+
}
|
|
1186
|
+
const path = join4(directory, entry);
|
|
1187
|
+
const sizeBytes = sizeOf(path);
|
|
1188
|
+
if (sizeBytes === void 0) continue;
|
|
1189
|
+
if (!dryRun) {
|
|
1190
|
+
try {
|
|
1191
|
+
rmSync2(path);
|
|
1192
|
+
} catch {
|
|
1193
|
+
continue;
|
|
1194
|
+
}
|
|
1195
|
+
}
|
|
1196
|
+
files.push({ path, sizeBytes });
|
|
1197
|
+
}
|
|
1198
|
+
return {
|
|
1199
|
+
files,
|
|
1200
|
+
freedBytes: files.reduce((total, file) => total + file.sizeBytes, 0),
|
|
1201
|
+
directory,
|
|
1202
|
+
dryRun
|
|
1203
|
+
};
|
|
1204
|
+
}
|
|
1205
|
+
function isModelBlob(entry) {
|
|
1206
|
+
return entry.endsWith(".gguf") || entry.endsWith(".gguf.ipull");
|
|
1207
|
+
}
|
|
1208
|
+
function sizeOf(path) {
|
|
1209
|
+
try {
|
|
1210
|
+
return statSync2(path).size;
|
|
1211
|
+
} catch {
|
|
1212
|
+
return void 0;
|
|
1213
|
+
}
|
|
1214
|
+
}
|
|
439
1215
|
|
|
440
1216
|
// src/complete.ts
|
|
441
1217
|
import { Ajv2020 } from "ajv/dist/2020.js";
|
|
@@ -448,6 +1224,7 @@ function validatorFor(schema) {
|
|
|
448
1224
|
return compiled;
|
|
449
1225
|
}
|
|
450
1226
|
async function completeValidatedJSON(options) {
|
|
1227
|
+
warnIfUnsupportedNode();
|
|
451
1228
|
const {
|
|
452
1229
|
provider,
|
|
453
1230
|
system,
|
|
@@ -488,56 +1265,6 @@ async function completeValidatedJSON(options) {
|
|
|
488
1265
|
return { ...base, error: lastError, durationMs: Date.now() - start };
|
|
489
1266
|
}
|
|
490
1267
|
|
|
491
|
-
// src/cache.ts
|
|
492
|
-
import { createHash } from "crypto";
|
|
493
|
-
import { existsSync, mkdirSync, readFileSync, writeFileSync } from "fs";
|
|
494
|
-
import { join } from "path";
|
|
495
|
-
function sha256(text) {
|
|
496
|
-
return createHash("sha256").update(text, "utf8").digest("hex");
|
|
497
|
-
}
|
|
498
|
-
function buildCacheKey(parts) {
|
|
499
|
-
return sha256(parts.map((p) => `${p.length}:${p}`).join("|"));
|
|
500
|
-
}
|
|
501
|
-
var JsonCache = class {
|
|
502
|
-
constructor(dir, enabled = true, label = "inference") {
|
|
503
|
-
this.dir = dir;
|
|
504
|
-
this.enabled = enabled;
|
|
505
|
-
this.label = label;
|
|
506
|
-
}
|
|
507
|
-
dir;
|
|
508
|
-
enabled;
|
|
509
|
-
label;
|
|
510
|
-
/** Cache-write failures warn once per process, not once per entry. */
|
|
511
|
-
warned = false;
|
|
512
|
-
get(key) {
|
|
513
|
-
if (!this.enabled) return void 0;
|
|
514
|
-
const path = join(this.dir, `${key}.json`);
|
|
515
|
-
if (!existsSync(path)) return void 0;
|
|
516
|
-
try {
|
|
517
|
-
return JSON.parse(readFileSync(path, "utf8"));
|
|
518
|
-
} catch {
|
|
519
|
-
return void 0;
|
|
520
|
-
}
|
|
521
|
-
}
|
|
522
|
-
set(key, value) {
|
|
523
|
-
if (!this.enabled) return;
|
|
524
|
-
try {
|
|
525
|
-
mkdirSync(this.dir, { recursive: true });
|
|
526
|
-
writeFileSync(
|
|
527
|
-
join(this.dir, `${key}.json`),
|
|
528
|
-
JSON.stringify(value, null, 2)
|
|
529
|
-
);
|
|
530
|
-
} catch (e) {
|
|
531
|
-
if (!this.warned) {
|
|
532
|
-
this.warned = true;
|
|
533
|
-
console.warn(
|
|
534
|
-
`${this.label}: could not write the cache at ${this.dir} (${e instanceof Error ? e.message : String(e)}). Continuing without caching.`
|
|
535
|
-
);
|
|
536
|
-
}
|
|
537
|
-
}
|
|
538
|
-
}
|
|
539
|
-
};
|
|
540
|
-
|
|
541
1268
|
// src/cost.ts
|
|
542
1269
|
var PRICE_TABLE = {
|
|
543
1270
|
"claude-sonnet-4-5": { inputPerMTok: 3, outputPerMTok: 15 },
|
|
@@ -693,29 +1420,56 @@ export {
|
|
|
693
1420
|
DEFAULT_MODELS,
|
|
694
1421
|
DEFAULT_OPENAI_BASE_URL,
|
|
695
1422
|
DEFAULT_ZONES,
|
|
1423
|
+
DETECTION_ORDER,
|
|
696
1424
|
InferenceError,
|
|
697
1425
|
JsonCache,
|
|
1426
|
+
LLAMA_MODELS,
|
|
1427
|
+
LLAMA_SELECTORS,
|
|
1428
|
+
LLAMA_TIERS,
|
|
1429
|
+
LlamaCppProvider,
|
|
698
1430
|
MockProvider,
|
|
699
1431
|
OpenAICompatProvider,
|
|
700
1432
|
PRICE_TABLE,
|
|
701
1433
|
VERDICT_SCHEMA,
|
|
1434
|
+
aliasForTier,
|
|
1435
|
+
availableProviders,
|
|
1436
|
+
blobNameFor,
|
|
702
1437
|
buildCacheKey,
|
|
1438
|
+
clearLlamaModels,
|
|
703
1439
|
completeValidatedJSON,
|
|
704
1440
|
computeConsensus,
|
|
705
1441
|
costOfRuns,
|
|
706
1442
|
costOfUsage,
|
|
1443
|
+
defaultLlamaModelsDirectory,
|
|
1444
|
+
defaultLlamaRuntime,
|
|
1445
|
+
defaultLlamaRuntimeDirectory,
|
|
1446
|
+
detectProvider,
|
|
1447
|
+
disposeLlamaModels,
|
|
707
1448
|
extractJson,
|
|
1449
|
+
importNodeLlamaCpp,
|
|
1450
|
+
isLlamaSelector,
|
|
1451
|
+
isModelDownloaded,
|
|
708
1452
|
judge,
|
|
709
1453
|
makeProvider,
|
|
1454
|
+
makeProviderAsync,
|
|
710
1455
|
mockVerdict,
|
|
1456
|
+
nodeLlamaCppStatus,
|
|
711
1457
|
pricingFor,
|
|
712
1458
|
realExec,
|
|
1459
|
+
resetClaudeCliProbe,
|
|
1460
|
+
resetNodeVersionWarning,
|
|
1461
|
+
resetProviderDetectionWarning,
|
|
1462
|
+
resetRuntimeInstall,
|
|
713
1463
|
resetTemperatureWarning,
|
|
1464
|
+
resolveLlamaModelRef,
|
|
714
1465
|
resolveProviderIdentity,
|
|
1466
|
+
resolveProviderIdentityAsync,
|
|
715
1467
|
runEnsemble,
|
|
716
1468
|
sha256,
|
|
717
1469
|
stripNulls,
|
|
1470
|
+
tierForBudget,
|
|
718
1471
|
toStrictSchema,
|
|
1472
|
+
uriForTier,
|
|
719
1473
|
validatorFor,
|
|
720
1474
|
zoneFor
|
|
721
1475
|
};
|