@hawkeyexl/inference 0.3.1 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/chunk-UFHBIS5K.js +185 -0
- package/dist/chunk-UFHBIS5K.js.map +1 -0
- package/dist/index.d.ts +3 -152
- package/dist/index.js +628 -95
- package/dist/index.js.map +1 -1
- package/dist/llama-cpp-CdVKP63Z.d.ts +308 -0
- package/dist/llama-worker.d.ts +1 -0
- package/dist/llama-worker.js +9 -0
- package/dist/llama-worker.js.map +1 -0
- package/package.json +2 -2
|
@@ -0,0 +1,308 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The provider contract. A provider turns a (system, user, schema) request
|
|
3
|
+
* into schema-conforming JSON. `provider()` and `modelName()` feed cache keys
|
|
4
|
+
* and pricing lookups, so two providers/models never share a cached result.
|
|
5
|
+
*
|
|
6
|
+
* This is deliberately the narrowest useful surface: no streaming, no
|
|
7
|
+
* multi-turn, no tool loops. Everything downstream of it — judging,
|
|
8
|
+
* extraction, classification — is schema-constrained single-shot completion.
|
|
9
|
+
*/
|
|
10
|
+
interface CompleteJSONRequest {
|
|
11
|
+
system: string;
|
|
12
|
+
user: string;
|
|
13
|
+
/** JSON Schema the response must conform to. */
|
|
14
|
+
schema: Record<string, unknown>;
|
|
15
|
+
temperature: number;
|
|
16
|
+
}
|
|
17
|
+
interface TokenUsage {
|
|
18
|
+
inputTokens: number;
|
|
19
|
+
outputTokens: number;
|
|
20
|
+
}
|
|
21
|
+
interface CompleteJSONResponse {
|
|
22
|
+
json: unknown;
|
|
23
|
+
/** Absent when the provider does not report usage (e.g. the Claude CLI). */
|
|
24
|
+
usage?: TokenUsage;
|
|
25
|
+
}
|
|
26
|
+
interface InferenceProvider {
|
|
27
|
+
/** Stable provider id — feeds cache keys. */
|
|
28
|
+
provider(): string;
|
|
29
|
+
/** Model id — feeds cache keys and pricing. */
|
|
30
|
+
modelName(): string;
|
|
31
|
+
completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
|
|
32
|
+
}
|
|
33
|
+
interface ExecResult {
|
|
34
|
+
code: number | null;
|
|
35
|
+
stdout: string;
|
|
36
|
+
stderr: string;
|
|
37
|
+
timedOut: boolean;
|
|
38
|
+
/** Set when the process could not be spawned (e.g. binary not found). */
|
|
39
|
+
spawnError?: string;
|
|
40
|
+
}
|
|
41
|
+
interface ExecOptions {
|
|
42
|
+
cwd?: string;
|
|
43
|
+
timeoutMs?: number;
|
|
44
|
+
/**
|
|
45
|
+
* Overrides on the ambient environment. A key mapped to `undefined`
|
|
46
|
+
* *unsets* that variable for the child rather than passing it through —
|
|
47
|
+
* Node omits undefined-valued keys when it builds the child's environment.
|
|
48
|
+
* Clearing inherited state (`GIT_*`, say) needs this, so the value type is
|
|
49
|
+
* deliberately wider than `string`.
|
|
50
|
+
*/
|
|
51
|
+
env?: Record<string, string | undefined>;
|
|
52
|
+
/** Text piped to the child's stdin (stdin is closed after writing). */
|
|
53
|
+
input?: string;
|
|
54
|
+
}
|
|
55
|
+
/** Injectable process-execution seam — subprocess providers take one for tests. */
|
|
56
|
+
type ExecFn = (cmd: string[], opts?: ExecOptions) => Promise<ExecResult>;
|
|
57
|
+
|
|
58
|
+
/** A llama.cpp backend, as node-llama-cpp names it; `false` is the CPU. */
|
|
59
|
+
type WorkerGpu = "metal" | "cuda" | "vulkan" | false;
|
|
60
|
+
/** What `getLlama` is asked for: one backend, or the best one not excluded. */
|
|
61
|
+
type WorkerGpuRequest = "auto" | WorkerGpu | {
|
|
62
|
+
type: "auto";
|
|
63
|
+
exclude: WorkerGpu[];
|
|
64
|
+
};
|
|
65
|
+
interface WorkerBackendOptions {
|
|
66
|
+
gpu: WorkerGpuRequest;
|
|
67
|
+
/**
|
|
68
|
+
* `"never"` on a fallback, so switching backends uses a prebuilt binary or
|
|
69
|
+
* fails — it never starts a multi-minute CMake build mid-run.
|
|
70
|
+
*/
|
|
71
|
+
build?: "never";
|
|
72
|
+
}
|
|
73
|
+
interface WorkerBackendHooks {
|
|
74
|
+
/**
|
|
75
|
+
* Names the backend about to initialise, before it does. A crash inside
|
|
76
|
+
* initialisation is then attributable, so the fallback knows what to skip.
|
|
77
|
+
*/
|
|
78
|
+
trying(gpu: WorkerGpu): void;
|
|
79
|
+
}
|
|
80
|
+
interface WorkerSession {
|
|
81
|
+
readonly contextSize?: number;
|
|
82
|
+
prompt(text: string, options: LlamaPromptOptions): Promise<LlamaPromptResult>;
|
|
83
|
+
dispose(): Promise<void>;
|
|
84
|
+
}
|
|
85
|
+
interface WorkerModel {
|
|
86
|
+
readonly trainContextSize?: number;
|
|
87
|
+
countTokens(text: string): number;
|
|
88
|
+
createSession(systemPrompt: string, contextSize?: number): Promise<WorkerSession>;
|
|
89
|
+
dispose(): Promise<void>;
|
|
90
|
+
}
|
|
91
|
+
interface WorkerBackend {
|
|
92
|
+
/** The backend that actually initialised. */
|
|
93
|
+
readonly gpu: WorkerGpu;
|
|
94
|
+
/** Memory available for weights, in bytes — see `getMemoryBudgetBytes`. */
|
|
95
|
+
memoryBudget(): Promise<number>;
|
|
96
|
+
loadModel(path: string): Promise<WorkerModel>;
|
|
97
|
+
}
|
|
98
|
+
/**
|
|
99
|
+
* Open a backend from an imported module: node-llama-cpp itself, or — in the
|
|
100
|
+
* test suite — a module exporting `createWorkerBackend`, which stands in for
|
|
101
|
+
* the inference while the process, the IPC, and the crash stay real.
|
|
102
|
+
*/
|
|
103
|
+
declare function openBackend(mod: unknown, options: WorkerBackendOptions, hooks: WorkerBackendHooks): Promise<WorkerBackend>;
|
|
104
|
+
/** One request from the parent. Every one gets exactly one reply. */
|
|
105
|
+
type WorkerRequest = {
|
|
106
|
+
id: number;
|
|
107
|
+
} & ({
|
|
108
|
+
op: "init";
|
|
109
|
+
moduleUrl: string;
|
|
110
|
+
options: WorkerBackendOptions;
|
|
111
|
+
} | {
|
|
112
|
+
op: "memoryBudget";
|
|
113
|
+
} | {
|
|
114
|
+
op: "loadModel";
|
|
115
|
+
path: string;
|
|
116
|
+
} | {
|
|
117
|
+
op: "countTokens";
|
|
118
|
+
modelId: number;
|
|
119
|
+
text: string;
|
|
120
|
+
} | {
|
|
121
|
+
op: "createSession";
|
|
122
|
+
modelId: number;
|
|
123
|
+
systemPrompt: string;
|
|
124
|
+
contextSize?: number;
|
|
125
|
+
} | {
|
|
126
|
+
op: "prompt";
|
|
127
|
+
sessionId: number;
|
|
128
|
+
text: string;
|
|
129
|
+
options: LlamaPromptOptions;
|
|
130
|
+
} | {
|
|
131
|
+
op: "disposeSession";
|
|
132
|
+
sessionId: number;
|
|
133
|
+
} | {
|
|
134
|
+
op: "disposeModel";
|
|
135
|
+
modelId: number;
|
|
136
|
+
} | {
|
|
137
|
+
op: "shutdown";
|
|
138
|
+
});
|
|
139
|
+
type WorkerMessage = {
|
|
140
|
+
id: number;
|
|
141
|
+
ok: true;
|
|
142
|
+
value: unknown;
|
|
143
|
+
} | {
|
|
144
|
+
id: number;
|
|
145
|
+
ok: false;
|
|
146
|
+
error: {
|
|
147
|
+
name: string;
|
|
148
|
+
message: string;
|
|
149
|
+
};
|
|
150
|
+
} | {
|
|
151
|
+
event: "trying";
|
|
152
|
+
gpu: WorkerGpu;
|
|
153
|
+
};
|
|
154
|
+
/** Set by the parent at fork time, so importing this file never serves. */
|
|
155
|
+
declare const WORKER_ENV_FLAG = "INFERENCE_LLAMA_WORKER";
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* Which llama.cpp backend to run on. `"auto"` picks the best one this machine
|
|
159
|
+
* has and falls back from one that crashes; anything else pins it.
|
|
160
|
+
*/
|
|
161
|
+
type LlamaGpu = "auto" | WorkerGpu;
|
|
162
|
+
|
|
163
|
+
interface LlamaPromptOptions {
|
|
164
|
+
/** JSON Schema converted to a GBNF grammar by the runtime. */
|
|
165
|
+
schema: Record<string, unknown>;
|
|
166
|
+
temperature: number;
|
|
167
|
+
/** Thinking budget; 0 disables it. See the note in `completeJSON`. */
|
|
168
|
+
thoughtTokens: number;
|
|
169
|
+
maxTokens?: number;
|
|
170
|
+
}
|
|
171
|
+
interface LlamaPromptResult {
|
|
172
|
+
text: string;
|
|
173
|
+
usage?: TokenUsage;
|
|
174
|
+
/**
|
|
175
|
+
* Why generation stopped. `"maxTokens"` means the output was cut off, so the
|
|
176
|
+
* text is almost certainly truncated JSON — see the guard in `completeJSON`.
|
|
177
|
+
*/
|
|
178
|
+
stopReason?: string;
|
|
179
|
+
}
|
|
180
|
+
interface LlamaSession {
|
|
181
|
+
prompt(text: string, options: LlamaPromptOptions): Promise<LlamaPromptResult>;
|
|
182
|
+
dispose(): Promise<void>;
|
|
183
|
+
/**
|
|
184
|
+
* Tokens of context the runtime actually created. llama.cpp may round a
|
|
185
|
+
* requested size up to a multiple of 256. Optional: a fake has no context.
|
|
186
|
+
*/
|
|
187
|
+
readonly contextSize?: number;
|
|
188
|
+
}
|
|
189
|
+
interface LlamaLoadedModel {
|
|
190
|
+
/**
|
|
191
|
+
* Open a single-turn session on a fresh context of `contextSize` tokens.
|
|
192
|
+
* The provider always passes it, sized from the prompt it is about to send;
|
|
193
|
+
* the real runtime treats an absent size as the 8192-token default.
|
|
194
|
+
*/
|
|
195
|
+
createSession(systemPrompt: string, contextSize?: number): Promise<LlamaSession>;
|
|
196
|
+
dispose(): Promise<void>;
|
|
197
|
+
/**
|
|
198
|
+
* The context length the model was trained on, the most a context can
|
|
199
|
+
* usefully hold. Optional so a runtime written before it existed still
|
|
200
|
+
* satisfies the seam; without it the provider assumes no ceiling.
|
|
201
|
+
*/
|
|
202
|
+
readonly trainContextSize?: number;
|
|
203
|
+
/**
|
|
204
|
+
* Count `text` in this model's own tokens. Optional for the same reason;
|
|
205
|
+
* without it the provider uses the default size and cannot check fit. May
|
|
206
|
+
* be async: the real runtime's tokenizer lives in the worker process.
|
|
207
|
+
*/
|
|
208
|
+
countTokens?(text: string): number | Promise<number>;
|
|
209
|
+
}
|
|
210
|
+
/**
|
|
211
|
+
* The whole of `node-llama-cpp` that this provider uses. Kept this narrow so a
|
|
212
|
+
* test fake is a few lines and so the real adapter is the only place that
|
|
213
|
+
* knows the upstream API shape.
|
|
214
|
+
*/
|
|
215
|
+
interface LlamaRuntime {
|
|
216
|
+
/**
|
|
217
|
+
* Resolve an `hf:` URI or path to a local file inside `directory`,
|
|
218
|
+
* downloading if needed.
|
|
219
|
+
*/
|
|
220
|
+
resolveModelFile(uri: string, directory: string): Promise<string>;
|
|
221
|
+
loadModel(path: string): Promise<LlamaLoadedModel>;
|
|
222
|
+
/** Memory available for weights, in bytes — VRAM if there is a GPU, else RAM. */
|
|
223
|
+
getMemoryBudgetBytes(): Promise<number>;
|
|
224
|
+
}
|
|
225
|
+
interface LlamaCppProviderOptions {
|
|
226
|
+
/** Injected for tests; defaults to the real `node-llama-cpp` adapter. */
|
|
227
|
+
runtime?: LlamaRuntime;
|
|
228
|
+
/**
|
|
229
|
+
* Thinking budget in tokens, default 0.
|
|
230
|
+
*
|
|
231
|
+
* Gemma 4 has a thinking mode, but a grammar constrains generation from
|
|
232
|
+
* token 0 — so an unbudgeted model starts reasoning and gets cut off
|
|
233
|
+
* mid-thought. Zero is the deterministic choice for judging; raise it if you
|
|
234
|
+
* want reasoning before the JSON.
|
|
235
|
+
*/
|
|
236
|
+
thoughtTokens?: number;
|
|
237
|
+
maxTokens?: number;
|
|
238
|
+
/**
|
|
239
|
+
* A fixed context size in tokens, used for every call.
|
|
240
|
+
*
|
|
241
|
+
* Unset, the context is sized to the work: 8192 tokens, or more when the
|
|
242
|
+
* prompt and its response reserve need it, up to the model's training
|
|
243
|
+
* context. Either way, a prompt that does not fit fails with an
|
|
244
|
+
* `InferenceError` rather than being truncated. ADR 01011.
|
|
245
|
+
*/
|
|
246
|
+
contextSize?: number;
|
|
247
|
+
/**
|
|
248
|
+
* Where to download and look for weights. Defaults to this library's own
|
|
249
|
+
* directory — see `defaultLlamaModelsDirectory`.
|
|
250
|
+
*/
|
|
251
|
+
modelsDirectory?: string;
|
|
252
|
+
/**
|
|
253
|
+
* The llama.cpp backend: `"cuda"`, `"vulkan"`, `"metal"`, or `false` for the
|
|
254
|
+
* CPU. Unset, `NODE_LLAMA_CPP_GPU` decides, and failing that `"auto"`.
|
|
255
|
+
*
|
|
256
|
+
* `"auto"` picks the best backend this machine has and, if it crashes the
|
|
257
|
+
* local-model worker, retries on the next — CUDA, then Vulkan, then the CPU —
|
|
258
|
+
* for the rest of the process. A named backend is never replaced: a crash on
|
|
259
|
+
* it is an error. Ignored when `runtime` is injected. ADR 01012.
|
|
260
|
+
*/
|
|
261
|
+
gpu?: LlamaGpu;
|
|
262
|
+
}
|
|
263
|
+
/**
|
|
264
|
+
* Free every loaded model.
|
|
265
|
+
*
|
|
266
|
+
* A standalone function rather than a `dispose()` on `InferenceProvider`:
|
|
267
|
+
* adding one to the contract would make all five providers carry a lifecycle
|
|
268
|
+
* only this one has. Short-lived processes can skip it.
|
|
269
|
+
*/
|
|
270
|
+
declare function disposeLlamaModels(): Promise<void>;
|
|
271
|
+
declare class LlamaCppProvider implements InferenceProvider {
|
|
272
|
+
private readonly model;
|
|
273
|
+
private readonly uri;
|
|
274
|
+
private readonly runtime;
|
|
275
|
+
private readonly thoughtTokens;
|
|
276
|
+
private readonly maxTokens;
|
|
277
|
+
private readonly contextSize;
|
|
278
|
+
private readonly modelsDirectory;
|
|
279
|
+
/**
|
|
280
|
+
* Loaded-model key: the same URI in two directories is two different files.
|
|
281
|
+
* Built with `buildCacheKey` so its parts are length-prefixed — a plain join
|
|
282
|
+
* would let two different (directory, uri) pairs collide and hand a provider
|
|
283
|
+
* back the wrong weights.
|
|
284
|
+
*/
|
|
285
|
+
private readonly cacheKey;
|
|
286
|
+
constructor(model: string, options?: LlamaCppProviderOptions);
|
|
287
|
+
provider(): string;
|
|
288
|
+
modelName(): string;
|
|
289
|
+
completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
|
|
290
|
+
/**
|
|
291
|
+
* The context this call needs: both prompts in the model's own tokens, the
|
|
292
|
+
* chat template's overhead, and room for the response. A prompt that does not
|
|
293
|
+
* fit is refused here, before anything is created. llama.cpp would otherwise
|
|
294
|
+
* shift the overflow out of the context and answer a prompt nobody sent.
|
|
295
|
+
*/
|
|
296
|
+
private contextFor;
|
|
297
|
+
private load;
|
|
298
|
+
}
|
|
299
|
+
/**
|
|
300
|
+
* The real runtime: node-llama-cpp in a worker process, falling back from a
|
|
301
|
+
* GPU backend that crashes it. Lazy — constructing a provider for a
|
|
302
|
+
* fully-cached run starts no process and loads no native binary.
|
|
303
|
+
*/
|
|
304
|
+
declare function defaultLlamaRuntime(options?: {
|
|
305
|
+
gpu?: LlamaGpu;
|
|
306
|
+
}): LlamaRuntime;
|
|
307
|
+
|
|
308
|
+
export { type CompleteJSONRequest as C, type ExecFn as E, type InferenceProvider as I, type LlamaCppProviderOptions as L, type TokenUsage as T, WORKER_ENV_FLAG as W, type CompleteJSONResponse as a, type LlamaRuntime as b, type ExecOptions as c, type ExecResult as d, LlamaCppProvider as e, type LlamaGpu as f, type LlamaLoadedModel as g, type LlamaPromptOptions as h, type LlamaPromptResult as i, type LlamaSession as j, defaultLlamaRuntime as k, disposeLlamaModels as l, type WorkerBackend as m, type WorkerBackendHooks as n, type WorkerBackendOptions as o, type WorkerGpu as p, type WorkerGpuRequest as q, type WorkerMessage as r, type WorkerModel as s, type WorkerRequest as t, type WorkerSession as u, openBackend as v };
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export { W as WORKER_ENV_FLAG, m as WorkerBackend, n as WorkerBackendHooks, o as WorkerBackendOptions, p as WorkerGpu, q as WorkerGpuRequest, r as WorkerMessage, s as WorkerModel, t as WorkerRequest, u as WorkerSession, v as openBackend } from './llama-cpp-CdVKP63Z.js';
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":[],"sourcesContent":[],"mappings":"","names":[]}
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@hawkeyexl/inference",
|
|
3
|
-
"version": "0.
|
|
4
|
-
"description": "Shared TypeScript LLM inference layer: schema-constrained completion across Anthropic, OpenAI-compatible, Claude CLI, and
|
|
3
|
+
"version": "0.4.0",
|
|
4
|
+
"description": "Shared TypeScript LLM inference layer: schema-constrained completion across Anthropic, OpenAI-compatible, Claude CLI, and local llama.cpp providers, with caching, cost accounting, and an LLM-as-judge ensemble.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|
|
7
7
|
"module": "dist/index.js",
|