@hawkeyexl/inference 0.3.1 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/chunk-UFHBIS5K.js +185 -0
- package/dist/chunk-UFHBIS5K.js.map +1 -0
- package/dist/index.d.ts +3 -152
- package/dist/index.js +628 -95
- package/dist/index.js.map +1 -1
- package/dist/llama-cpp-CdVKP63Z.d.ts +308 -0
- package/dist/llama-worker.d.ts +1 -0
- package/dist/llama-worker.js +9 -0
- package/dist/llama-worker.js.map +1 -0
- package/package.json +2 -2
package/README.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
# @hawkeyexl/inference
|
|
2
2
|
|
|
3
3
|
Shared LLM inference layer for the docs-as-tests toolchain: schema-constrained completion across
|
|
4
|
-
Anthropic, OpenAI-compatible, Claude CLI, and
|
|
4
|
+
Anthropic, OpenAI-compatible, Claude CLI, and local (llama.cpp) providers, with result
|
|
5
5
|
caching, cost accounting, and an LLM-as-judge ensemble on top.
|
|
6
6
|
|
|
7
7
|
Extracted from three projects that had each grown their own copy —
|
|
@@ -101,7 +101,7 @@ tokens makes a budget gate inert. See
|
|
|
101
101
|
| [Get started](https://hawkeyexl.github.io/inference/get-started/) | Install, one validated call with no key, choosing a provider |
|
|
102
102
|
| [Judge & consensus](https://hawkeyexl.github.io/inference/judge/) | Ensembles, consensus math, confidence zones, caching, budgets |
|
|
103
103
|
| [Structured extraction](https://hawkeyexl.github.io/inference/extract/) | One schema-constrained call, honest failures, the subprocess seam |
|
|
104
|
-
| [Run models locally](https://hawkeyexl.github.io/inference/local/) | GGUF weights
|
|
104
|
+
| [Run models locally](https://hawkeyexl.github.io/inference/local/) | GGUF weights run locally, model selection, managing weights on disk |
|
|
105
105
|
| [Keep it working](https://hawkeyexl.github.io/inference/keep-it-working/testing/) | Testing without a network, upgrading without losing a cache |
|
|
106
106
|
| [Reference](https://hawkeyexl.github.io/inference/reference/providers/) | Full signatures for every export |
|
|
107
107
|
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
// src/providers/llama-worker.ts
|
|
2
|
+
import { totalmem } from "os";
|
|
3
|
+
var DEFAULT_CONTEXT_SIZE = 8192;
|
|
4
|
+
function openBackend(mod, options, hooks) {
|
|
5
|
+
const custom = mod.createWorkerBackend;
|
|
6
|
+
if (typeof custom === "function") return custom(options, hooks);
|
|
7
|
+
return nodeLlamaCppBackend(mod, options, hooks);
|
|
8
|
+
}
|
|
9
|
+
async function nodeLlamaCppBackend(mod, options, hooks) {
|
|
10
|
+
const { getLlama, getLlamaGpuTypes, LlamaChatSession, TokenMeter } = mod;
|
|
11
|
+
if (options.gpu === "auto" || typeof options.gpu === "object") {
|
|
12
|
+
const exclude = typeof options.gpu === "object" ? options.gpu.exclude : [];
|
|
13
|
+
const supported = await getLlamaGpuTypes("supported").catch(
|
|
14
|
+
() => []
|
|
15
|
+
);
|
|
16
|
+
hooks.trying(supported.find((g) => g !== false && !exclude.includes(g)) ?? false);
|
|
17
|
+
} else {
|
|
18
|
+
hooks.trying(options.gpu);
|
|
19
|
+
}
|
|
20
|
+
const llama = await getLlama({
|
|
21
|
+
gpu: options.gpu,
|
|
22
|
+
...options.build ? { build: options.build } : {}
|
|
23
|
+
});
|
|
24
|
+
return {
|
|
25
|
+
gpu: llama.gpu,
|
|
26
|
+
async memoryBudget() {
|
|
27
|
+
const ramBudget = totalmem() / 2;
|
|
28
|
+
try {
|
|
29
|
+
const vram = await llama.getVramState();
|
|
30
|
+
return Math.max(vram.free, ramBudget);
|
|
31
|
+
} catch {
|
|
32
|
+
return ramBudget;
|
|
33
|
+
}
|
|
34
|
+
},
|
|
35
|
+
async loadModel(path) {
|
|
36
|
+
const model = await llama.loadModel({ modelPath: path });
|
|
37
|
+
return {
|
|
38
|
+
trainContextSize: model.trainContextSize,
|
|
39
|
+
countTokens: (text) => model.tokenize(text).length,
|
|
40
|
+
async createSession(systemPrompt, contextSize) {
|
|
41
|
+
const context = await model.createContext({
|
|
42
|
+
contextSize: contextSize ?? DEFAULT_CONTEXT_SIZE
|
|
43
|
+
});
|
|
44
|
+
const sequence = context.getSequence();
|
|
45
|
+
const session = new LlamaChatSession({
|
|
46
|
+
contextSequence: sequence,
|
|
47
|
+
systemPrompt
|
|
48
|
+
});
|
|
49
|
+
return {
|
|
50
|
+
contextSize: context.contextSize,
|
|
51
|
+
async prompt(text, promptOptions) {
|
|
52
|
+
const grammar = await llama.createGrammarForJsonSchema(
|
|
53
|
+
promptOptions.schema
|
|
54
|
+
);
|
|
55
|
+
const before = sequence.tokenMeter.getState();
|
|
56
|
+
const result = await session.promptWithMeta(text, {
|
|
57
|
+
grammar,
|
|
58
|
+
temperature: promptOptions.temperature,
|
|
59
|
+
budgets: { thoughtTokens: promptOptions.thoughtTokens },
|
|
60
|
+
...promptOptions.maxTokens != null ? { maxTokens: promptOptions.maxTokens } : {}
|
|
61
|
+
});
|
|
62
|
+
const diff = TokenMeter.diff(sequence.tokenMeter, before);
|
|
63
|
+
return {
|
|
64
|
+
text: result.responseText,
|
|
65
|
+
stopReason: result.stopReason,
|
|
66
|
+
usage: {
|
|
67
|
+
inputTokens: diff.usedInputTokens,
|
|
68
|
+
outputTokens: diff.usedOutputTokens
|
|
69
|
+
}
|
|
70
|
+
};
|
|
71
|
+
},
|
|
72
|
+
async dispose() {
|
|
73
|
+
await context.dispose();
|
|
74
|
+
}
|
|
75
|
+
};
|
|
76
|
+
},
|
|
77
|
+
async dispose() {
|
|
78
|
+
await model.dispose();
|
|
79
|
+
}
|
|
80
|
+
};
|
|
81
|
+
}
|
|
82
|
+
};
|
|
83
|
+
}
|
|
84
|
+
var WORKER_ENV_FLAG = "INFERENCE_LLAMA_WORKER";
|
|
85
|
+
function serve() {
|
|
86
|
+
delete process.env[WORKER_ENV_FLAG];
|
|
87
|
+
let backend;
|
|
88
|
+
const models = /* @__PURE__ */ new Map();
|
|
89
|
+
const sessions = /* @__PURE__ */ new Map();
|
|
90
|
+
let nextId = 1;
|
|
91
|
+
const send = (message, then) => {
|
|
92
|
+
try {
|
|
93
|
+
process.send(message, void 0, {}, () => then?.());
|
|
94
|
+
} catch {
|
|
95
|
+
}
|
|
96
|
+
};
|
|
97
|
+
const ready = () => {
|
|
98
|
+
if (!backend) throw new Error("The local-model worker was not initialised.");
|
|
99
|
+
return backend;
|
|
100
|
+
};
|
|
101
|
+
const model = (id) => {
|
|
102
|
+
const found = models.get(id);
|
|
103
|
+
if (!found) throw new Error(`The local-model worker has no model ${id}.`);
|
|
104
|
+
return found;
|
|
105
|
+
};
|
|
106
|
+
const session = (id) => {
|
|
107
|
+
const found = sessions.get(id);
|
|
108
|
+
if (!found) throw new Error(`The local-model worker has no session ${id}.`);
|
|
109
|
+
return found;
|
|
110
|
+
};
|
|
111
|
+
async function dispatch(request) {
|
|
112
|
+
switch (request.op) {
|
|
113
|
+
case "init": {
|
|
114
|
+
const mod = await import(request.moduleUrl);
|
|
115
|
+
backend = openBackend(mod, request.options, {
|
|
116
|
+
trying: (gpu) => send({ event: "trying", gpu })
|
|
117
|
+
});
|
|
118
|
+
return { gpu: (await backend).gpu };
|
|
119
|
+
}
|
|
120
|
+
case "memoryBudget":
|
|
121
|
+
return (await ready()).memoryBudget();
|
|
122
|
+
case "loadModel": {
|
|
123
|
+
const loaded = await (await ready()).loadModel(request.path);
|
|
124
|
+
const modelId = nextId++;
|
|
125
|
+
models.set(modelId, loaded);
|
|
126
|
+
return { modelId, trainContextSize: loaded.trainContextSize };
|
|
127
|
+
}
|
|
128
|
+
case "countTokens":
|
|
129
|
+
return model(request.modelId).countTokens(request.text);
|
|
130
|
+
case "createSession": {
|
|
131
|
+
const opened = await model(request.modelId).createSession(
|
|
132
|
+
request.systemPrompt,
|
|
133
|
+
request.contextSize
|
|
134
|
+
);
|
|
135
|
+
const sessionId = nextId++;
|
|
136
|
+
sessions.set(sessionId, opened);
|
|
137
|
+
return { sessionId, contextSize: opened.contextSize };
|
|
138
|
+
}
|
|
139
|
+
case "prompt":
|
|
140
|
+
return session(request.sessionId).prompt(request.text, request.options);
|
|
141
|
+
case "disposeSession": {
|
|
142
|
+
const found = sessions.get(request.sessionId);
|
|
143
|
+
sessions.delete(request.sessionId);
|
|
144
|
+
await found?.dispose();
|
|
145
|
+
return null;
|
|
146
|
+
}
|
|
147
|
+
case "disposeModel": {
|
|
148
|
+
const found = models.get(request.modelId);
|
|
149
|
+
models.delete(request.modelId);
|
|
150
|
+
await found?.dispose();
|
|
151
|
+
return null;
|
|
152
|
+
}
|
|
153
|
+
case "shutdown": {
|
|
154
|
+
await Promise.allSettled([...sessions.values()].map((s) => s.dispose()));
|
|
155
|
+
await Promise.allSettled([...models.values()].map((m) => m.dispose()));
|
|
156
|
+
sessions.clear();
|
|
157
|
+
models.clear();
|
|
158
|
+
return null;
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
process.on("message", (request) => {
|
|
163
|
+
void dispatch(request).then(
|
|
164
|
+
(value) => send({ id: request.id, ok: true, value: value ?? null }, () => {
|
|
165
|
+
if (request.op === "shutdown") process.exit(0);
|
|
166
|
+
}),
|
|
167
|
+
(e) => send({
|
|
168
|
+
id: request.id,
|
|
169
|
+
ok: false,
|
|
170
|
+
error: {
|
|
171
|
+
name: e instanceof Error ? e.name : "Error",
|
|
172
|
+
message: e instanceof Error ? e.message : String(e)
|
|
173
|
+
}
|
|
174
|
+
})
|
|
175
|
+
);
|
|
176
|
+
});
|
|
177
|
+
process.on("disconnect", () => process.exit(0));
|
|
178
|
+
}
|
|
179
|
+
if (process.send && process.env[WORKER_ENV_FLAG] === "1") serve();
|
|
180
|
+
|
|
181
|
+
export {
|
|
182
|
+
openBackend,
|
|
183
|
+
WORKER_ENV_FLAG
|
|
184
|
+
};
|
|
185
|
+
//# sourceMappingURL=chunk-UFHBIS5K.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["../src/providers/llama-worker.ts"],"sourcesContent":["/**\n * The local-model worker: a child process that owns node-llama-cpp.\n *\n * llama.cpp reports a fatal GPU error by aborting the process (GGML_ABORT).\n * In-process, that ended the consumer — every result it had computed, gone,\n * with no exception to catch. Here it ends only this worker, and the parent\n * sees a child that exited mid-request, which it can retry on another backend.\n * ADR 01012.\n *\n * This file is forked straight from `src/` under the test suite, where Node\n * strips its types but cannot map a sibling's `.js` import to its `.ts`. So it\n * imports only Node builtins at runtime — `source-hygiene.test.ts` pins that.\n * Types are erased, so `import type` from siblings is fine.\n */\nimport { totalmem } from \"node:os\";\nimport type { LlamaPromptOptions, LlamaPromptResult } from \"./llama-cpp.js\";\n\n/** A llama.cpp backend, as node-llama-cpp names it; `false` is the CPU. */\nexport type WorkerGpu = \"metal\" | \"cuda\" | \"vulkan\" | false;\n\n/** What `getLlama` is asked for: one backend, or the best one not excluded. */\nexport type WorkerGpuRequest =\n | \"auto\"\n | WorkerGpu\n | { type: \"auto\"; exclude: WorkerGpu[] };\n\nexport interface WorkerBackendOptions {\n gpu: WorkerGpuRequest;\n /**\n * `\"never\"` on a fallback, so switching backends uses a prebuilt binary or\n * fails — it never starts a multi-minute CMake build mid-run.\n */\n build?: \"never\";\n}\n\nexport interface WorkerBackendHooks {\n /**\n * Names the backend about to initialise, before it does. A crash inside\n * initialisation is then attributable, so the fallback knows what to skip.\n */\n trying(gpu: WorkerGpu): void;\n}\n\nexport interface WorkerSession {\n readonly contextSize?: number;\n prompt(text: string, options: LlamaPromptOptions): Promise<LlamaPromptResult>;\n dispose(): Promise<void>;\n}\n\nexport interface WorkerModel {\n readonly trainContextSize?: number;\n countTokens(text: string): number;\n createSession(systemPrompt: string, contextSize?: number): Promise<WorkerSession>;\n dispose(): Promise<void>;\n}\n\nexport interface WorkerBackend {\n /** The backend that actually initialised. */\n readonly gpu: WorkerGpu;\n /** Memory available for weights, in bytes — see `getMemoryBudgetBytes`. */\n memoryBudget(): Promise<number>;\n loadModel(path: string): Promise<WorkerModel>;\n}\n\n/**\n * The context an unsized session gets — the same default `llama-cpp.ts` uses.\n * Restated rather than imported: this file may import only Node builtins.\n */\nconst DEFAULT_CONTEXT_SIZE = 8192;\n\ntype CreateWorkerBackend = (\n options: WorkerBackendOptions,\n hooks: WorkerBackendHooks,\n) => Promise<WorkerBackend>;\n\n/**\n * Open a backend from an imported module: node-llama-cpp itself, or — in the\n * test suite — a module exporting `createWorkerBackend`, which stands in for\n * the inference while the process, the IPC, and the crash stay real.\n */\nexport function openBackend(\n mod: unknown,\n options: WorkerBackendOptions,\n hooks: WorkerBackendHooks,\n): Promise<WorkerBackend> {\n const custom = (mod as { createWorkerBackend?: CreateWorkerBackend })\n .createWorkerBackend;\n if (typeof custom === \"function\") return custom(options, hooks);\n return nodeLlamaCppBackend(mod as typeof import(\"node-llama-cpp\"), options, hooks);\n}\n\nasync function nodeLlamaCppBackend(\n mod: typeof import(\"node-llama-cpp\"),\n options: WorkerBackendOptions,\n hooks: WorkerBackendHooks,\n): Promise<WorkerBackend> {\n const { getLlama, getLlamaGpuTypes, LlamaChatSession, TokenMeter } = mod;\n\n if (options.gpu === \"auto\" || typeof options.gpu === \"object\") {\n // getLlama picks the best supported backend not excluded; name that one\n // first. If getLlama settles on another, `llama.gpu` below is the truth.\n const exclude = typeof options.gpu === \"object\" ? options.gpu.exclude : [];\n const supported = await getLlamaGpuTypes(\"supported\").catch(\n (): WorkerGpu[] => [],\n );\n hooks.trying(supported.find((g) => g !== false && !exclude.includes(g)) ?? false);\n } else {\n hooks.trying(options.gpu);\n }\n\n const llama = await getLlama({\n gpu: options.gpu,\n ...(options.build ? { build: options.build } : {}),\n });\n\n return {\n gpu: llama.gpu,\n\n async memoryBudget() {\n // Half of RAM is what a judge can reasonably claim on a shared machine;\n // a GPU's free VRAM is usable outright.\n const ramBudget = totalmem() / 2;\n try {\n const vram = await llama.getVramState();\n // The LARGER of the two, not VRAM in preference to RAM: llama.cpp\n // offloads the layers that fit onto the GPU and keeps the rest in\n // system RAM, so a small GPU beside plenty of RAM still runs a big\n // model. Sizing off VRAM alone would idle most of such a machine.\n return Math.max(vram.free, ramBudget);\n } catch {\n // CPU-only builds and probe failures are normal, never fatal.\n return ramBudget;\n }\n },\n\n async loadModel(path) {\n const model = await llama.loadModel({ modelPath: path });\n return {\n trainContextSize: model.trainContextSize,\n countTokens: (text) => model.tokenize(text).length,\n async createSession(systemPrompt, contextSize) {\n // Always an explicit size. Left out, node-llama-cpp sizes the context\n // to free memory, which is the bug ADR 01011 records.\n const context = await model.createContext({\n contextSize: contextSize ?? DEFAULT_CONTEXT_SIZE,\n });\n const sequence = context.getSequence();\n const session = new LlamaChatSession({\n contextSequence: sequence,\n systemPrompt,\n });\n return {\n contextSize: context.contextSize,\n async prompt(text, promptOptions) {\n const grammar = await llama.createGrammarForJsonSchema(\n promptOptions.schema as Parameters<\n typeof llama.createGrammarForJsonSchema\n >[0],\n );\n const before = sequence.tokenMeter.getState();\n const result = await session.promptWithMeta(text, {\n grammar,\n temperature: promptOptions.temperature,\n budgets: { thoughtTokens: promptOptions.thoughtTokens },\n ...(promptOptions.maxTokens != null\n ? { maxTokens: promptOptions.maxTokens }\n : {}),\n });\n // promptWithMeta does not report usage; the sequence's meter does.\n const diff = TokenMeter.diff(sequence.tokenMeter, before);\n return {\n text: result.responseText,\n stopReason: result.stopReason,\n usage: {\n inputTokens: diff.usedInputTokens,\n outputTokens: diff.usedOutputTokens,\n },\n };\n },\n async dispose() {\n await context.dispose();\n },\n };\n },\n async dispose() {\n await model.dispose();\n },\n };\n },\n };\n}\n\n/** One request from the parent. Every one gets exactly one reply. */\nexport type WorkerRequest = { id: number } & (\n | { op: \"init\"; moduleUrl: string; options: WorkerBackendOptions }\n | { op: \"memoryBudget\" }\n | { op: \"loadModel\"; path: string }\n | { op: \"countTokens\"; modelId: number; text: string }\n | { op: \"createSession\"; modelId: number; systemPrompt: string; contextSize?: number }\n | { op: \"prompt\"; sessionId: number; text: string; options: LlamaPromptOptions }\n | { op: \"disposeSession\"; sessionId: number }\n | { op: \"disposeModel\"; modelId: number }\n | { op: \"shutdown\" }\n);\n\nexport type WorkerMessage =\n | { id: number; ok: true; value: unknown }\n | { id: number; ok: false; error: { name: string; message: string } }\n | { event: \"trying\"; gpu: WorkerGpu };\n\n/** Set by the parent at fork time, so importing this file never serves. */\nexport const WORKER_ENV_FLAG = \"INFERENCE_LLAMA_WORKER\";\n\nfunction serve(): void {\n // Anything this worker spawns is not a worker.\n delete process.env[WORKER_ENV_FLAG];\n let backend: Promise<WorkerBackend> | undefined;\n const models = new Map<number, WorkerModel>();\n const sessions = new Map<number, WorkerSession>();\n let nextId = 1;\n\n const send = (message: WorkerMessage, then?: () => void): void => {\n try {\n process.send!(message, undefined, {}, () => then?.());\n } catch {\n // The parent is gone; `disconnect` below ends this process.\n }\n };\n\n const ready = (): Promise<WorkerBackend> => {\n if (!backend) throw new Error(\"The local-model worker was not initialised.\");\n return backend;\n };\n const model = (id: number): WorkerModel => {\n const found = models.get(id);\n if (!found) throw new Error(`The local-model worker has no model ${id}.`);\n return found;\n };\n const session = (id: number): WorkerSession => {\n const found = sessions.get(id);\n if (!found) throw new Error(`The local-model worker has no session ${id}.`);\n return found;\n };\n\n async function dispatch(request: WorkerRequest): Promise<unknown> {\n switch (request.op) {\n case \"init\": {\n const mod: unknown = await import(request.moduleUrl);\n backend = openBackend(mod, request.options, {\n trying: (gpu) => send({ event: \"trying\", gpu }),\n });\n return { gpu: (await backend).gpu };\n }\n case \"memoryBudget\":\n return (await ready()).memoryBudget();\n case \"loadModel\": {\n const loaded = await (await ready()).loadModel(request.path);\n const modelId = nextId++;\n models.set(modelId, loaded);\n return { modelId, trainContextSize: loaded.trainContextSize };\n }\n case \"countTokens\":\n return model(request.modelId).countTokens(request.text);\n case \"createSession\": {\n const opened = await model(request.modelId).createSession(\n request.systemPrompt,\n request.contextSize,\n );\n const sessionId = nextId++;\n sessions.set(sessionId, opened);\n return { sessionId, contextSize: opened.contextSize };\n }\n case \"prompt\":\n return session(request.sessionId).prompt(request.text, request.options);\n case \"disposeSession\": {\n const found = sessions.get(request.sessionId);\n sessions.delete(request.sessionId);\n await found?.dispose();\n return null;\n }\n case \"disposeModel\": {\n const found = models.get(request.modelId);\n models.delete(request.modelId);\n await found?.dispose();\n return null;\n }\n case \"shutdown\": {\n await Promise.allSettled([...sessions.values()].map((s) => s.dispose()));\n await Promise.allSettled([...models.values()].map((m) => m.dispose()));\n sessions.clear();\n models.clear();\n return null;\n }\n }\n }\n\n process.on(\"message\", (request: WorkerRequest) => {\n void dispatch(request).then(\n (value) =>\n send({ id: request.id, ok: true, value: value ?? null }, () => {\n if (request.op === \"shutdown\") process.exit(0);\n }),\n (e: unknown) =>\n send({\n id: request.id,\n ok: false,\n error: {\n name: e instanceof Error ? e.name : \"Error\",\n message: e instanceof Error ? e.message : String(e),\n },\n }),\n );\n });\n\n // The parent exited, crashed, or was killed: nothing is left to answer.\n process.on(\"disconnect\", () => process.exit(0));\n}\n\nif (process.send && process.env[WORKER_ENV_FLAG] === \"1\") serve();\n"],"mappings":";AAcA,SAAS,gBAAgB;AAsDzB,IAAM,uBAAuB;AAYtB,SAAS,YACd,KACA,SACA,OACwB;AACxB,QAAM,SAAU,IACb;AACH,MAAI,OAAO,WAAW,WAAY,QAAO,OAAO,SAAS,KAAK;AAC9D,SAAO,oBAAoB,KAAwC,SAAS,KAAK;AACnF;AAEA,eAAe,oBACb,KACA,SACA,OACwB;AACxB,QAAM,EAAE,UAAU,kBAAkB,kBAAkB,WAAW,IAAI;AAErE,MAAI,QAAQ,QAAQ,UAAU,OAAO,QAAQ,QAAQ,UAAU;AAG7D,UAAM,UAAU,OAAO,QAAQ,QAAQ,WAAW,QAAQ,IAAI,UAAU,CAAC;AACzE,UAAM,YAAY,MAAM,iBAAiB,WAAW,EAAE;AAAA,MACpD,MAAmB,CAAC;AAAA,IACtB;AACA,UAAM,OAAO,UAAU,KAAK,CAAC,MAAM,MAAM,SAAS,CAAC,QAAQ,SAAS,CAAC,CAAC,KAAK,KAAK;AAAA,EAClF,OAAO;AACL,UAAM,OAAO,QAAQ,GAAG;AAAA,EAC1B;AAEA,QAAM,QAAQ,MAAM,SAAS;AAAA,IAC3B,KAAK,QAAQ;AAAA,IACb,GAAI,QAAQ,QAAQ,EAAE,OAAO,QAAQ,MAAM,IAAI,CAAC;AAAA,EAClD,CAAC;AAED,SAAO;AAAA,IACL,KAAK,MAAM;AAAA,IAEX,MAAM,eAAe;AAGnB,YAAM,YAAY,SAAS,IAAI;AAC/B,UAAI;AACF,cAAM,OAAO,MAAM,MAAM,aAAa;AAKtC,eAAO,KAAK,IAAI,KAAK,MAAM,SAAS;AAAA,MACtC,QAAQ;AAEN,eAAO;AAAA,MACT;AAAA,IACF;AAAA,IAEA,MAAM,UAAU,MAAM;AACpB,YAAM,QAAQ,MAAM,MAAM,UAAU,EAAE,WAAW,KAAK,CAAC;AACvD,aAAO;AAAA,QACL,kBAAkB,MAAM;AAAA,QACxB,aAAa,CAAC,SAAS,MAAM,SAAS,IAAI,EAAE;AAAA,QAC5C,MAAM,cAAc,cAAc,aAAa;AAG7C,gBAAM,UAAU,MAAM,MAAM,cAAc;AAAA,YACxC,aAAa,eAAe;AAAA,UAC9B,CAAC;AACD,gBAAM,WAAW,QAAQ,YAAY;AACrC,gBAAM,UAAU,IAAI,iBAAiB;AAAA,YACnC,iBAAiB;AAAA,YACjB;AAAA,UACF,CAAC;AACD,iBAAO;AAAA,YACL,aAAa,QAAQ;AAAA,YACrB,MAAM,OAAO,MAAM,eAAe;AAChC,oBAAM,UAAU,MAAM,MAAM;AAAA,gBAC1B,cAAc;AAAA,cAGhB;AACA,oBAAM,SAAS,SAAS,WAAW,SAAS;AAC5C,oBAAM,SAAS,MAAM,QAAQ,eAAe,MAAM;AAAA,gBAChD;AAAA,gBACA,aAAa,cAAc;AAAA,gBAC3B,SAAS,EAAE,eAAe,cAAc,cAAc;AAAA,gBACtD,GAAI,cAAc,aAAa,OAC3B,EAAE,WAAW,cAAc,UAAU,IACrC,CAAC;AAAA,cACP,CAAC;AAED,oBAAM,OAAO,WAAW,KAAK,SAAS,YAAY,MAAM;AACxD,qBAAO;AAAA,gBACL,MAAM,OAAO;AAAA,gBACb,YAAY,OAAO;AAAA,gBACnB,OAAO;AAAA,kBACL,aAAa,KAAK;AAAA,kBAClB,cAAc,KAAK;AAAA,gBACrB;AAAA,cACF;AAAA,YACF;AAAA,YACA,MAAM,UAAU;AACd,oBAAM,QAAQ,QAAQ;AAAA,YACxB;AAAA,UACF;AAAA,QACF;AAAA,QACA,MAAM,UAAU;AACd,gBAAM,MAAM,QAAQ;AAAA,QACtB;AAAA,MACF;AAAA,IACF;AAAA,EACF;AACF;AAqBO,IAAM,kBAAkB;AAE/B,SAAS,QAAc;AAErB,SAAO,QAAQ,IAAI,eAAe;AAClC,MAAI;AACJ,QAAM,SAAS,oBAAI,IAAyB;AAC5C,QAAM,WAAW,oBAAI,IAA2B;AAChD,MAAI,SAAS;AAEb,QAAM,OAAO,CAAC,SAAwB,SAA4B;AAChE,QAAI;AACF,cAAQ,KAAM,SAAS,QAAW,CAAC,GAAG,MAAM,OAAO,CAAC;AAAA,IACtD,QAAQ;AAAA,IAER;AAAA,EACF;AAEA,QAAM,QAAQ,MAA8B;AAC1C,QAAI,CAAC,QAAS,OAAM,IAAI,MAAM,6CAA6C;AAC3E,WAAO;AAAA,EACT;AACA,QAAM,QAAQ,CAAC,OAA4B;AACzC,UAAM,QAAQ,OAAO,IAAI,EAAE;AAC3B,QAAI,CAAC,MAAO,OAAM,IAAI,MAAM,uCAAuC,EAAE,GAAG;AACxE,WAAO;AAAA,EACT;AACA,QAAM,UAAU,CAAC,OAA8B;AAC7C,UAAM,QAAQ,SAAS,IAAI,EAAE;AAC7B,QAAI,CAAC,MAAO,OAAM,IAAI,MAAM,yCAAyC,EAAE,GAAG;AAC1E,WAAO;AAAA,EACT;AAEA,iBAAe,SAAS,SAA0C;AAChE,YAAQ,QAAQ,IAAI;AAAA,MAClB,KAAK,QAAQ;AACX,cAAM,MAAe,MAAM,OAAO,QAAQ;AAC1C,kBAAU,YAAY,KAAK,QAAQ,SAAS;AAAA,UAC1C,QAAQ,CAAC,QAAQ,KAAK,EAAE,OAAO,UAAU,IAAI,CAAC;AAAA,QAChD,CAAC;AACD,eAAO,EAAE,MAAM,MAAM,SAAS,IAAI;AAAA,MACpC;AAAA,MACA,KAAK;AACH,gBAAQ,MAAM,MAAM,GAAG,aAAa;AAAA,MACtC,KAAK,aAAa;AAChB,cAAM,SAAS,OAAO,MAAM,MAAM,GAAG,UAAU,QAAQ,IAAI;AAC3D,cAAM,UAAU;AAChB,eAAO,IAAI,SAAS,MAAM;AAC1B,eAAO,EAAE,SAAS,kBAAkB,OAAO,iBAAiB;AAAA,MAC9D;AAAA,MACA,KAAK;AACH,eAAO,MAAM,QAAQ,OAAO,EAAE,YAAY,QAAQ,IAAI;AAAA,MACxD,KAAK,iBAAiB;AACpB,cAAM,SAAS,MAAM,MAAM,QAAQ,OAAO,EAAE;AAAA,UAC1C,QAAQ;AAAA,UACR,QAAQ;AAAA,QACV;AACA,cAAM,YAAY;AAClB,iBAAS,IAAI,WAAW,MAAM;AAC9B,eAAO,EAAE,WAAW,aAAa,OAAO,YAAY;AAAA,MACtD;AAAA,MACA,KAAK;AACH,eAAO,QAAQ,QAAQ,SAAS,EAAE,OAAO,QAAQ,MAAM,QAAQ,OAAO;AAAA,MACxE,KAAK,kBAAkB;AACrB,cAAM,QAAQ,SAAS,IAAI,QAAQ,SAAS;AAC5C,iBAAS,OAAO,QAAQ,SAAS;AACjC,cAAM,OAAO,QAAQ;AACrB,eAAO;AAAA,MACT;AAAA,MACA,KAAK,gBAAgB;AACnB,cAAM,QAAQ,OAAO,IAAI,QAAQ,OAAO;AACxC,eAAO,OAAO,QAAQ,OAAO;AAC7B,cAAM,OAAO,QAAQ;AACrB,eAAO;AAAA,MACT;AAAA,MACA,KAAK,YAAY;AACf,cAAM,QAAQ,WAAW,CAAC,GAAG,SAAS,OAAO,CAAC,EAAE,IAAI,CAAC,MAAM,EAAE,QAAQ,CAAC,CAAC;AACvE,cAAM,QAAQ,WAAW,CAAC,GAAG,OAAO,OAAO,CAAC,EAAE,IAAI,CAAC,MAAM,EAAE,QAAQ,CAAC,CAAC;AACrE,iBAAS,MAAM;AACf,eAAO,MAAM;AACb,eAAO;AAAA,MACT;AAAA,IACF;AAAA,EACF;AAEA,UAAQ,GAAG,WAAW,CAAC,YAA2B;AAChD,SAAK,SAAS,OAAO,EAAE;AAAA,MACrB,CAAC,UACC,KAAK,EAAE,IAAI,QAAQ,IAAI,IAAI,MAAM,OAAO,SAAS,KAAK,GAAG,MAAM;AAC7D,YAAI,QAAQ,OAAO,WAAY,SAAQ,KAAK,CAAC;AAAA,MAC/C,CAAC;AAAA,MACH,CAAC,MACC,KAAK;AAAA,QACH,IAAI,QAAQ;AAAA,QACZ,IAAI;AAAA,QACJ,OAAO;AAAA,UACL,MAAM,aAAa,QAAQ,EAAE,OAAO;AAAA,UACpC,SAAS,aAAa,QAAQ,EAAE,UAAU,OAAO,CAAC;AAAA,QACpD;AAAA,MACF,CAAC;AAAA,IACL;AAAA,EACF,CAAC;AAGD,UAAQ,GAAG,cAAc,MAAM,QAAQ,KAAK,CAAC,CAAC;AAChD;AAEA,IAAI,QAAQ,QAAQ,QAAQ,IAAI,eAAe,MAAM,IAAK,OAAM;","names":[]}
|
package/dist/index.d.ts
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { T as TokenUsage, I as InferenceProvider, C as CompleteJSONRequest, a as CompleteJSONResponse, L as LlamaCppProviderOptions, E as ExecFn, b as LlamaRuntime } from './llama-cpp-CdVKP63Z.js';
|
|
2
|
+
export { c as ExecOptions, d as ExecResult, e as LlamaCppProvider, f as LlamaGpu, g as LlamaLoadedModel, h as LlamaPromptOptions, i as LlamaPromptResult, j as LlamaSession, k as defaultLlamaRuntime, l as disposeLlamaModels } from './llama-cpp-CdVKP63Z.js';
|
|
1
3
|
import { ValidateFunction } from 'ajv';
|
|
2
4
|
|
|
3
5
|
/**
|
|
@@ -12,63 +14,6 @@ declare class InferenceError extends Error {
|
|
|
12
14
|
/** Test seam: reset the once-per-process Node version warning. */
|
|
13
15
|
declare function resetNodeVersionWarning(): void;
|
|
14
16
|
|
|
15
|
-
/**
|
|
16
|
-
* The provider contract. A provider turns a (system, user, schema) request
|
|
17
|
-
* into schema-conforming JSON. `provider()` and `modelName()` feed cache keys
|
|
18
|
-
* and pricing lookups, so two providers/models never share a cached result.
|
|
19
|
-
*
|
|
20
|
-
* This is deliberately the narrowest useful surface: no streaming, no
|
|
21
|
-
* multi-turn, no tool loops. Everything downstream of it — judging,
|
|
22
|
-
* extraction, classification — is schema-constrained single-shot completion.
|
|
23
|
-
*/
|
|
24
|
-
interface CompleteJSONRequest {
|
|
25
|
-
system: string;
|
|
26
|
-
user: string;
|
|
27
|
-
/** JSON Schema the response must conform to. */
|
|
28
|
-
schema: Record<string, unknown>;
|
|
29
|
-
temperature: number;
|
|
30
|
-
}
|
|
31
|
-
interface TokenUsage {
|
|
32
|
-
inputTokens: number;
|
|
33
|
-
outputTokens: number;
|
|
34
|
-
}
|
|
35
|
-
interface CompleteJSONResponse {
|
|
36
|
-
json: unknown;
|
|
37
|
-
/** Absent when the provider does not report usage (e.g. the Claude CLI). */
|
|
38
|
-
usage?: TokenUsage;
|
|
39
|
-
}
|
|
40
|
-
interface InferenceProvider {
|
|
41
|
-
/** Stable provider id — feeds cache keys. */
|
|
42
|
-
provider(): string;
|
|
43
|
-
/** Model id — feeds cache keys and pricing. */
|
|
44
|
-
modelName(): string;
|
|
45
|
-
completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
|
|
46
|
-
}
|
|
47
|
-
interface ExecResult {
|
|
48
|
-
code: number | null;
|
|
49
|
-
stdout: string;
|
|
50
|
-
stderr: string;
|
|
51
|
-
timedOut: boolean;
|
|
52
|
-
/** Set when the process could not be spawned (e.g. binary not found). */
|
|
53
|
-
spawnError?: string;
|
|
54
|
-
}
|
|
55
|
-
interface ExecOptions {
|
|
56
|
-
cwd?: string;
|
|
57
|
-
timeoutMs?: number;
|
|
58
|
-
/**
|
|
59
|
-
* Overrides on the ambient environment. A key mapped to `undefined`
|
|
60
|
-
* *unsets* that variable for the child rather than passing it through —
|
|
61
|
-
* Node omits undefined-valued keys when it builds the child's environment.
|
|
62
|
-
* Clearing inherited state (`GIT_*`, say) needs this, so the value type is
|
|
63
|
-
* deliberately wider than `string`.
|
|
64
|
-
*/
|
|
65
|
-
env?: Record<string, string | undefined>;
|
|
66
|
-
/** Text piped to the child's stdin (stdin is closed after writing). */
|
|
67
|
-
input?: string;
|
|
68
|
-
}
|
|
69
|
-
/** Injectable process-execution seam — subprocess providers take one for tests. */
|
|
70
|
-
type ExecFn = (cmd: string[], opts?: ExecOptions) => Promise<ExecResult>;
|
|
71
|
-
|
|
72
17
|
/**
|
|
73
18
|
* Cost tracking: token usage priced from a small static table, overridable per
|
|
74
19
|
* model by the caller. Unknown models cost 0 (unknown), never a guess — a
|
|
@@ -176,100 +121,6 @@ declare function mockVerdict(match: "pass" | "fail" | "partial", confidence: num
|
|
|
176
121
|
json: unknown;
|
|
177
122
|
};
|
|
178
123
|
|
|
179
|
-
interface LlamaPromptOptions {
|
|
180
|
-
/** JSON Schema converted to a GBNF grammar by the runtime. */
|
|
181
|
-
schema: Record<string, unknown>;
|
|
182
|
-
temperature: number;
|
|
183
|
-
/** Thinking budget; 0 disables it. See the note in `completeJSON`. */
|
|
184
|
-
thoughtTokens: number;
|
|
185
|
-
maxTokens?: number;
|
|
186
|
-
}
|
|
187
|
-
interface LlamaPromptResult {
|
|
188
|
-
text: string;
|
|
189
|
-
usage?: TokenUsage;
|
|
190
|
-
/**
|
|
191
|
-
* Why generation stopped. `"maxTokens"` means the output was cut off, so the
|
|
192
|
-
* text is almost certainly truncated JSON — see the guard in `completeJSON`.
|
|
193
|
-
*/
|
|
194
|
-
stopReason?: string;
|
|
195
|
-
}
|
|
196
|
-
interface LlamaSession {
|
|
197
|
-
prompt(text: string, options: LlamaPromptOptions): Promise<LlamaPromptResult>;
|
|
198
|
-
dispose(): Promise<void>;
|
|
199
|
-
}
|
|
200
|
-
interface LlamaLoadedModel {
|
|
201
|
-
createSession(systemPrompt: string): Promise<LlamaSession>;
|
|
202
|
-
dispose(): Promise<void>;
|
|
203
|
-
}
|
|
204
|
-
/**
|
|
205
|
-
* The whole of `node-llama-cpp` that this provider uses. Kept this narrow so a
|
|
206
|
-
* test fake is a few lines and so the real adapter is the only place that
|
|
207
|
-
* knows the upstream API shape.
|
|
208
|
-
*/
|
|
209
|
-
interface LlamaRuntime {
|
|
210
|
-
/**
|
|
211
|
-
* Resolve an `hf:` URI or path to a local file inside `directory`,
|
|
212
|
-
* downloading if needed.
|
|
213
|
-
*/
|
|
214
|
-
resolveModelFile(uri: string, directory: string): Promise<string>;
|
|
215
|
-
loadModel(path: string): Promise<LlamaLoadedModel>;
|
|
216
|
-
/** Memory available for weights, in bytes — VRAM if there is a GPU, else RAM. */
|
|
217
|
-
getMemoryBudgetBytes(): Promise<number>;
|
|
218
|
-
}
|
|
219
|
-
interface LlamaCppProviderOptions {
|
|
220
|
-
/** Injected for tests; defaults to the real `node-llama-cpp` adapter. */
|
|
221
|
-
runtime?: LlamaRuntime;
|
|
222
|
-
/**
|
|
223
|
-
* Thinking budget in tokens, default 0.
|
|
224
|
-
*
|
|
225
|
-
* Gemma 4 has a thinking mode, but a grammar constrains generation from
|
|
226
|
-
* token 0 — so an unbudgeted model starts reasoning and gets cut off
|
|
227
|
-
* mid-thought. Zero is the deterministic choice for judging; raise it if you
|
|
228
|
-
* want reasoning before the JSON.
|
|
229
|
-
*/
|
|
230
|
-
thoughtTokens?: number;
|
|
231
|
-
maxTokens?: number;
|
|
232
|
-
/**
|
|
233
|
-
* Where to download and look for weights. Defaults to this library's own
|
|
234
|
-
* directory — see `defaultLlamaModelsDirectory`.
|
|
235
|
-
*/
|
|
236
|
-
modelsDirectory?: string;
|
|
237
|
-
}
|
|
238
|
-
/**
|
|
239
|
-
* Free every loaded model.
|
|
240
|
-
*
|
|
241
|
-
* A standalone function rather than a `dispose()` on `InferenceProvider`:
|
|
242
|
-
* adding one to the contract would make all five providers carry a lifecycle
|
|
243
|
-
* only this one has. Short-lived processes can skip it.
|
|
244
|
-
*/
|
|
245
|
-
declare function disposeLlamaModels(): Promise<void>;
|
|
246
|
-
declare class LlamaCppProvider implements InferenceProvider {
|
|
247
|
-
private readonly model;
|
|
248
|
-
private readonly uri;
|
|
249
|
-
private readonly runtime;
|
|
250
|
-
private readonly thoughtTokens;
|
|
251
|
-
private readonly maxTokens;
|
|
252
|
-
private readonly modelsDirectory;
|
|
253
|
-
/**
|
|
254
|
-
* Loaded-model key: the same URI in two directories is two different files.
|
|
255
|
-
* Built with `buildCacheKey` so its parts are length-prefixed — a plain join
|
|
256
|
-
* would let two different (directory, uri) pairs collide and hand a provider
|
|
257
|
-
* back the wrong weights.
|
|
258
|
-
*/
|
|
259
|
-
private readonly cacheKey;
|
|
260
|
-
constructor(model: string, options?: LlamaCppProviderOptions);
|
|
261
|
-
provider(): string;
|
|
262
|
-
modelName(): string;
|
|
263
|
-
completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
|
|
264
|
-
private load;
|
|
265
|
-
}
|
|
266
|
-
/**
|
|
267
|
-
* Lazy adapter over the real `node-llama-cpp`. Every method defers to the
|
|
268
|
-
* dynamic import, so constructing a provider for a fully-cached run never
|
|
269
|
-
* loads the native binary.
|
|
270
|
-
*/
|
|
271
|
-
declare function defaultLlamaRuntime(): LlamaRuntime;
|
|
272
|
-
|
|
273
124
|
type ProviderName = "anthropic" | "openai" | "claude-cli" | "mock" | "llama-cpp";
|
|
274
125
|
/** A concrete provider, or `"auto"` to detect one. */
|
|
275
126
|
type ProviderSelector = ProviderName | "auto";
|
|
@@ -706,4 +557,4 @@ declare function judge(options: EnsembleOptions & {
|
|
|
706
557
|
/** Test seam: reset the once-per-process temperature warning. */
|
|
707
558
|
declare function resetTemperatureWarning(): void;
|
|
708
559
|
|
|
709
|
-
export { AnthropicProvider, type AnthropicProviderOptions, ClaudeCliProvider, type ClearLlamaModelsOptions, type ClearLlamaModelsResult, type ClearedModelFile,
|
|
560
|
+
export { AnthropicProvider, type AnthropicProviderOptions, ClaudeCliProvider, type ClearLlamaModelsOptions, type ClearLlamaModelsResult, type ClearedModelFile, CompleteJSONRequest, CompleteJSONResponse, type CompleteValidatedOptions, type ConsensusResult, DEFAULT_MODELS, DEFAULT_OPENAI_BASE_URL, DEFAULT_ZONES, DETECTION_ORDER, type EnsembleOptions, ExecFn, InferenceError, InferenceProvider, type InferenceRun, JsonCache, type JudgeRun, type JudgeVerdict, LLAMA_MODELS, LLAMA_SELECTORS, LLAMA_TIERS, LlamaCppProviderOptions, type LlamaModelEntry, LlamaRuntime, type LlamaSelector, type LlamaTier, type Match, MockProvider, type MockResponse, OpenAICompatProvider, type OpenAICompatProviderOptions, PRICE_TABLE, type Pricing, type ProviderIdentity, type ProviderName, type ProviderSelector, type ProviderSpec, type RuntimeInstallOptions, type RuntimeStatus, TokenUsage, VERDICT_SCHEMA, type Zone, type ZoneThresholds, aliasForTier, availableProviders, blobNameFor, buildCacheKey, clearLlamaModels, completeValidatedJSON, computeConsensus, costOfRuns, costOfUsage, defaultLlamaModelsDirectory, defaultLlamaRuntimeDirectory, detectProvider, extractJson, importNodeLlamaCpp, isLlamaSelector, isModelDownloaded, judge, makeProvider, makeProviderAsync, mockVerdict, nodeLlamaCppStatus, pricingFor, realExec, resetClaudeCliProbe, resetNodeVersionWarning, resetProviderDetectionWarning, resetRuntimeInstall, resetTemperatureWarning, resolveLlamaModelRef, resolveProviderIdentity, resolveProviderIdentityAsync, runEnsemble, sha256, stripNulls, tierForBudget, toStrictSchema, uriForTier, validatorFor, zoneFor };
|