@hawkeyexl/inference 0.3.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,308 @@
1
+ /**
2
+ * The provider contract. A provider turns a (system, user, schema) request
3
+ * into schema-conforming JSON. `provider()` and `modelName()` feed cache keys
4
+ * and pricing lookups, so two providers/models never share a cached result.
5
+ *
6
+ * This is deliberately the narrowest useful surface: no streaming, no
7
+ * multi-turn, no tool loops. Everything downstream of it — judging,
8
+ * extraction, classification — is schema-constrained single-shot completion.
9
+ */
10
+ interface CompleteJSONRequest {
11
+ system: string;
12
+ user: string;
13
+ /** JSON Schema the response must conform to. */
14
+ schema: Record<string, unknown>;
15
+ temperature: number;
16
+ }
17
+ interface TokenUsage {
18
+ inputTokens: number;
19
+ outputTokens: number;
20
+ }
21
+ interface CompleteJSONResponse {
22
+ json: unknown;
23
+ /** Absent when the provider does not report usage (e.g. the Claude CLI). */
24
+ usage?: TokenUsage;
25
+ }
26
+ interface InferenceProvider {
27
+ /** Stable provider id — feeds cache keys. */
28
+ provider(): string;
29
+ /** Model id — feeds cache keys and pricing. */
30
+ modelName(): string;
31
+ completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
32
+ }
33
+ interface ExecResult {
34
+ code: number | null;
35
+ stdout: string;
36
+ stderr: string;
37
+ timedOut: boolean;
38
+ /** Set when the process could not be spawned (e.g. binary not found). */
39
+ spawnError?: string;
40
+ }
41
+ interface ExecOptions {
42
+ cwd?: string;
43
+ timeoutMs?: number;
44
+ /**
45
+ * Overrides on the ambient environment. A key mapped to `undefined`
46
+ * *unsets* that variable for the child rather than passing it through —
47
+ * Node omits undefined-valued keys when it builds the child's environment.
48
+ * Clearing inherited state (`GIT_*`, say) needs this, so the value type is
49
+ * deliberately wider than `string`.
50
+ */
51
+ env?: Record<string, string | undefined>;
52
+ /** Text piped to the child's stdin (stdin is closed after writing). */
53
+ input?: string;
54
+ }
55
+ /** Injectable process-execution seam — subprocess providers take one for tests. */
56
+ type ExecFn = (cmd: string[], opts?: ExecOptions) => Promise<ExecResult>;
57
+
58
+ /** A llama.cpp backend, as node-llama-cpp names it; `false` is the CPU. */
59
+ type WorkerGpu = "metal" | "cuda" | "vulkan" | false;
60
+ /** What `getLlama` is asked for: one backend, or the best one not excluded. */
61
+ type WorkerGpuRequest = "auto" | WorkerGpu | {
62
+ type: "auto";
63
+ exclude: WorkerGpu[];
64
+ };
65
+ interface WorkerBackendOptions {
66
+ gpu: WorkerGpuRequest;
67
+ /**
68
+ * `"never"` on a fallback, so switching backends uses a prebuilt binary or
69
+ * fails — it never starts a multi-minute CMake build mid-run.
70
+ */
71
+ build?: "never";
72
+ }
73
+ interface WorkerBackendHooks {
74
+ /**
75
+ * Names the backend about to initialise, before it does. A crash inside
76
+ * initialisation is then attributable, so the fallback knows what to skip.
77
+ */
78
+ trying(gpu: WorkerGpu): void;
79
+ }
80
+ interface WorkerSession {
81
+ readonly contextSize?: number;
82
+ prompt(text: string, options: LlamaPromptOptions): Promise<LlamaPromptResult>;
83
+ dispose(): Promise<void>;
84
+ }
85
+ interface WorkerModel {
86
+ readonly trainContextSize?: number;
87
+ countTokens(text: string): number;
88
+ createSession(systemPrompt: string, contextSize?: number): Promise<WorkerSession>;
89
+ dispose(): Promise<void>;
90
+ }
91
+ interface WorkerBackend {
92
+ /** The backend that actually initialised. */
93
+ readonly gpu: WorkerGpu;
94
+ /** Memory available for weights, in bytes — see `getMemoryBudgetBytes`. */
95
+ memoryBudget(): Promise<number>;
96
+ loadModel(path: string): Promise<WorkerModel>;
97
+ }
98
+ /**
99
+ * Open a backend from an imported module: node-llama-cpp itself, or — in the
100
+ * test suite — a module exporting `createWorkerBackend`, which stands in for
101
+ * the inference while the process, the IPC, and the crash stay real.
102
+ */
103
+ declare function openBackend(mod: unknown, options: WorkerBackendOptions, hooks: WorkerBackendHooks): Promise<WorkerBackend>;
104
+ /** One request from the parent. Every one gets exactly one reply. */
105
+ type WorkerRequest = {
106
+ id: number;
107
+ } & ({
108
+ op: "init";
109
+ moduleUrl: string;
110
+ options: WorkerBackendOptions;
111
+ } | {
112
+ op: "memoryBudget";
113
+ } | {
114
+ op: "loadModel";
115
+ path: string;
116
+ } | {
117
+ op: "countTokens";
118
+ modelId: number;
119
+ text: string;
120
+ } | {
121
+ op: "createSession";
122
+ modelId: number;
123
+ systemPrompt: string;
124
+ contextSize?: number;
125
+ } | {
126
+ op: "prompt";
127
+ sessionId: number;
128
+ text: string;
129
+ options: LlamaPromptOptions;
130
+ } | {
131
+ op: "disposeSession";
132
+ sessionId: number;
133
+ } | {
134
+ op: "disposeModel";
135
+ modelId: number;
136
+ } | {
137
+ op: "shutdown";
138
+ });
139
+ type WorkerMessage = {
140
+ id: number;
141
+ ok: true;
142
+ value: unknown;
143
+ } | {
144
+ id: number;
145
+ ok: false;
146
+ error: {
147
+ name: string;
148
+ message: string;
149
+ };
150
+ } | {
151
+ event: "trying";
152
+ gpu: WorkerGpu;
153
+ };
154
+ /** Set by the parent at fork time, so importing this file never serves. */
155
+ declare const WORKER_ENV_FLAG = "INFERENCE_LLAMA_WORKER";
156
+
157
+ /**
158
+ * Which llama.cpp backend to run on. `"auto"` picks the best one this machine
159
+ * has and falls back from one that crashes; anything else pins it.
160
+ */
161
+ type LlamaGpu = "auto" | WorkerGpu;
162
+
163
+ interface LlamaPromptOptions {
164
+ /** JSON Schema converted to a GBNF grammar by the runtime. */
165
+ schema: Record<string, unknown>;
166
+ temperature: number;
167
+ /** Thinking budget; 0 disables it. See the note in `completeJSON`. */
168
+ thoughtTokens: number;
169
+ maxTokens?: number;
170
+ }
171
+ interface LlamaPromptResult {
172
+ text: string;
173
+ usage?: TokenUsage;
174
+ /**
175
+ * Why generation stopped. `"maxTokens"` means the output was cut off, so the
176
+ * text is almost certainly truncated JSON — see the guard in `completeJSON`.
177
+ */
178
+ stopReason?: string;
179
+ }
180
+ interface LlamaSession {
181
+ prompt(text: string, options: LlamaPromptOptions): Promise<LlamaPromptResult>;
182
+ dispose(): Promise<void>;
183
+ /**
184
+ * Tokens of context the runtime actually created. llama.cpp may round a
185
+ * requested size up to a multiple of 256. Optional: a fake has no context.
186
+ */
187
+ readonly contextSize?: number;
188
+ }
189
+ interface LlamaLoadedModel {
190
+ /**
191
+ * Open a single-turn session on a fresh context of `contextSize` tokens.
192
+ * The provider always passes it, sized from the prompt it is about to send;
193
+ * the real runtime treats an absent size as the 8192-token default.
194
+ */
195
+ createSession(systemPrompt: string, contextSize?: number): Promise<LlamaSession>;
196
+ dispose(): Promise<void>;
197
+ /**
198
+ * The context length the model was trained on, the most a context can
199
+ * usefully hold. Optional so a runtime written before it existed still
200
+ * satisfies the seam; without it the provider assumes no ceiling.
201
+ */
202
+ readonly trainContextSize?: number;
203
+ /**
204
+ * Count `text` in this model's own tokens. Optional for the same reason;
205
+ * without it the provider uses the default size and cannot check fit. May
206
+ * be async: the real runtime's tokenizer lives in the worker process.
207
+ */
208
+ countTokens?(text: string): number | Promise<number>;
209
+ }
210
+ /**
211
+ * The whole of `node-llama-cpp` that this provider uses. Kept this narrow so a
212
+ * test fake is a few lines and so the real adapter is the only place that
213
+ * knows the upstream API shape.
214
+ */
215
+ interface LlamaRuntime {
216
+ /**
217
+ * Resolve an `hf:` URI or path to a local file inside `directory`,
218
+ * downloading if needed.
219
+ */
220
+ resolveModelFile(uri: string, directory: string): Promise<string>;
221
+ loadModel(path: string): Promise<LlamaLoadedModel>;
222
+ /** Memory available for weights, in bytes — VRAM if there is a GPU, else RAM. */
223
+ getMemoryBudgetBytes(): Promise<number>;
224
+ }
225
+ interface LlamaCppProviderOptions {
226
+ /** Injected for tests; defaults to the real `node-llama-cpp` adapter. */
227
+ runtime?: LlamaRuntime;
228
+ /**
229
+ * Thinking budget in tokens, default 0.
230
+ *
231
+ * Gemma 4 has a thinking mode, but a grammar constrains generation from
232
+ * token 0 — so an unbudgeted model starts reasoning and gets cut off
233
+ * mid-thought. Zero is the deterministic choice for judging; raise it if you
234
+ * want reasoning before the JSON.
235
+ */
236
+ thoughtTokens?: number;
237
+ maxTokens?: number;
238
+ /**
239
+ * A fixed context size in tokens, used for every call.
240
+ *
241
+ * Unset, the context is sized to the work: 8192 tokens, or more when the
242
+ * prompt and its response reserve need it, up to the model's training
243
+ * context. Either way, a prompt that does not fit fails with an
244
+ * `InferenceError` rather than being truncated. ADR 01011.
245
+ */
246
+ contextSize?: number;
247
+ /**
248
+ * Where to download and look for weights. Defaults to this library's own
249
+ * directory — see `defaultLlamaModelsDirectory`.
250
+ */
251
+ modelsDirectory?: string;
252
+ /**
253
+ * The llama.cpp backend: `"cuda"`, `"vulkan"`, `"metal"`, or `false` for the
254
+ * CPU. Unset, `NODE_LLAMA_CPP_GPU` decides, and failing that `"auto"`.
255
+ *
256
+ * `"auto"` picks the best backend this machine has and, if it crashes the
257
+ * local-model worker, retries on the next — CUDA, then Vulkan, then the CPU —
258
+ * for the rest of the process. A named backend is never replaced: a crash on
259
+ * it is an error. Ignored when `runtime` is injected. ADR 01012.
260
+ */
261
+ gpu?: LlamaGpu;
262
+ }
263
+ /**
264
+ * Free every loaded model.
265
+ *
266
+ * A standalone function rather than a `dispose()` on `InferenceProvider`:
267
+ * adding one to the contract would make all five providers carry a lifecycle
268
+ * only this one has. Short-lived processes can skip it.
269
+ */
270
+ declare function disposeLlamaModels(): Promise<void>;
271
+ declare class LlamaCppProvider implements InferenceProvider {
272
+ private readonly model;
273
+ private readonly uri;
274
+ private readonly runtime;
275
+ private readonly thoughtTokens;
276
+ private readonly maxTokens;
277
+ private readonly contextSize;
278
+ private readonly modelsDirectory;
279
+ /**
280
+ * Loaded-model key: the same URI in two directories is two different files.
281
+ * Built with `buildCacheKey` so its parts are length-prefixed — a plain join
282
+ * would let two different (directory, uri) pairs collide and hand a provider
283
+ * back the wrong weights.
284
+ */
285
+ private readonly cacheKey;
286
+ constructor(model: string, options?: LlamaCppProviderOptions);
287
+ provider(): string;
288
+ modelName(): string;
289
+ completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
290
+ /**
291
+ * The context this call needs: both prompts in the model's own tokens, the
292
+ * chat template's overhead, and room for the response. A prompt that does not
293
+ * fit is refused here, before anything is created. llama.cpp would otherwise
294
+ * shift the overflow out of the context and answer a prompt nobody sent.
295
+ */
296
+ private contextFor;
297
+ private load;
298
+ }
299
+ /**
300
+ * The real runtime: node-llama-cpp in a worker process, falling back from a
301
+ * GPU backend that crashes it. Lazy — constructing a provider for a
302
+ * fully-cached run starts no process and loads no native binary.
303
+ */
304
+ declare function defaultLlamaRuntime(options?: {
305
+ gpu?: LlamaGpu;
306
+ }): LlamaRuntime;
307
+
308
+ export { type CompleteJSONRequest as C, type ExecFn as E, type InferenceProvider as I, type LlamaCppProviderOptions as L, type TokenUsage as T, WORKER_ENV_FLAG as W, type CompleteJSONResponse as a, type LlamaRuntime as b, type ExecOptions as c, type ExecResult as d, LlamaCppProvider as e, type LlamaGpu as f, type LlamaLoadedModel as g, type LlamaPromptOptions as h, type LlamaPromptResult as i, type LlamaSession as j, defaultLlamaRuntime as k, disposeLlamaModels as l, type WorkerBackend as m, type WorkerBackendHooks as n, type WorkerBackendOptions as o, type WorkerGpu as p, type WorkerGpuRequest as q, type WorkerMessage as r, type WorkerModel as s, type WorkerRequest as t, type WorkerSession as u, openBackend as v };
@@ -0,0 +1 @@
1
+ export { W as WORKER_ENV_FLAG, m as WorkerBackend, n as WorkerBackendHooks, o as WorkerBackendOptions, p as WorkerGpu, q as WorkerGpuRequest, r as WorkerMessage, s as WorkerModel, t as WorkerRequest, u as WorkerSession, v as openBackend } from './llama-cpp-CdVKP63Z.js';
@@ -0,0 +1,9 @@
1
+ import {
2
+ WORKER_ENV_FLAG,
3
+ openBackend
4
+ } from "./chunk-UFHBIS5K.js";
5
+ export {
6
+ WORKER_ENV_FLAG,
7
+ openBackend
8
+ };
9
+ //# sourceMappingURL=llama-worker.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"sources":[],"sourcesContent":[],"mappings":"","names":[]}
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@hawkeyexl/inference",
3
- "version": "0.3.1",
4
- "description": "Shared TypeScript LLM inference layer: schema-constrained completion across Anthropic, OpenAI-compatible, Claude CLI, and in-process local llama.cpp providers, with caching, cost accounting, and an LLM-as-judge ensemble.",
3
+ "version": "0.4.0",
4
+ "description": "Shared TypeScript LLM inference layer: schema-constrained completion across Anthropic, OpenAI-compatible, Claude CLI, and local llama.cpp providers, with caching, cost accounting, and an LLM-as-judge ensemble.",
5
5
  "type": "module",
6
6
  "main": "dist/index.js",
7
7
  "module": "dist/index.js",