@hawkeyexl/inference 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +64 -138
- package/dist/index.d.ts +338 -7
- package/dist/index.js +810 -56
- package/dist/index.js.map +1 -1
- package/package.json +15 -3
package/README.md
CHANGED
|
@@ -1,46 +1,56 @@
|
|
|
1
1
|
# @hawkeyexl/inference
|
|
2
2
|
|
|
3
3
|
Shared LLM inference layer for the docs-as-tests toolchain: schema-constrained completion across
|
|
4
|
-
Anthropic, OpenAI-compatible,
|
|
5
|
-
an LLM-as-judge ensemble on top.
|
|
4
|
+
Anthropic, OpenAI-compatible, Claude CLI, and in-process local (llama.cpp) providers, with result
|
|
5
|
+
caching, cost accounting, and an LLM-as-judge ensemble on top.
|
|
6
6
|
|
|
7
7
|
Extracted from three projects that had each grown their own copy —
|
|
8
8
|
[docevals](https://github.com/hawkeyexl/docevals), [dockg](https://github.com/hawkeyexl/dockg), and
|
|
9
9
|
[agentevals](https://github.com/hawkeyexl/agentevals) — so a provider fix lands once instead of
|
|
10
10
|
three times.
|
|
11
11
|
|
|
12
|
+
**📖 [Documentation](https://hawkeyexl.github.io/inference/)**
|
|
13
|
+
|
|
12
14
|
## Install
|
|
13
15
|
|
|
14
16
|
```bash
|
|
15
17
|
npm install @hawkeyexl/inference
|
|
16
18
|
```
|
|
17
19
|
|
|
18
|
-
Requires Node 24+.
|
|
20
|
+
Requires Node 24+. Three runtime dependencies, plus one optional peer dependency for local models.
|
|
21
|
+
|
|
22
|
+
**ESM only.** The `exports` map has no `require` condition, so
|
|
23
|
+
`require("@hawkeyexl/inference")` fails with `ERR_PACKAGE_PATH_NOT_EXPORTED`. From CommonJS, use
|
|
24
|
+
`await import("@hawkeyexl/inference")`.
|
|
19
25
|
|
|
20
26
|
## What it does
|
|
21
27
|
|
|
22
|
-
Every consumer
|
|
23
|
-
|
|
24
|
-
|
|
28
|
+
Every consumer wants the same narrow thing: **send a system prompt, a user prompt, and a JSON
|
|
29
|
+
Schema; get back JSON that validates against that schema, or a recorded error.**
|
|
30
|
+
|
|
31
|
+
```ts
|
|
32
|
+
(system, user, schema, temperature) -> JSON
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
**No streaming, no multi-turn, no tool loops.** If you need a conversation, this is the wrong
|
|
36
|
+
package. Widening the provider contract requires an [ADR](adrs).
|
|
25
37
|
|
|
26
38
|
Two layers, one entry point:
|
|
27
39
|
|
|
28
|
-
- **Completion** — the provider contract,
|
|
29
|
-
and a validate-and-retry wrapper.
|
|
40
|
+
- **Completion** — the provider contract, five providers, a content-addressed cache, a price table,
|
|
41
|
+
and a validate-and-retry wrapper. All that structured extraction needs.
|
|
30
42
|
- **Judge** — the canonical verdict schema, an N-run ensemble, consensus math, and confidence-zone
|
|
31
43
|
routing. Built on the completion layer; ignore it if you do not need it.
|
|
32
44
|
|
|
33
45
|
## Quick start
|
|
34
46
|
|
|
35
|
-
|
|
47
|
+
No API key required — `MockProvider` is exported for exactly this.
|
|
36
48
|
|
|
37
49
|
```ts
|
|
38
|
-
import {
|
|
50
|
+
import { MockProvider, completeValidatedJSON } from "@hawkeyexl/inference";
|
|
39
51
|
|
|
40
|
-
const
|
|
41
|
-
|
|
42
|
-
const run = await completeValidatedJSON<{ summary: string }>({
|
|
43
|
-
provider,
|
|
52
|
+
const run = await completeValidatedJSON({
|
|
53
|
+
provider: new MockProvider([{ json: { summary: "Covers authentication." } }]),
|
|
44
54
|
system: "You summarize documentation pages.",
|
|
45
55
|
user: pageBody,
|
|
46
56
|
schema: {
|
|
@@ -55,145 +65,48 @@ if (run.error) console.error(run.error);
|
|
|
55
65
|
else console.log(run.result.summary, run.usage);
|
|
56
66
|
```
|
|
57
67
|
|
|
58
|
-
`completeValidatedJSON` never throws on a model failure and never coerces a bad response. It
|
|
59
|
-
|
|
68
|
+
`completeValidatedJSON` never throws on a model failure and never coerces a bad response. It retries
|
|
69
|
+
once, then returns a run with `error` set and `result` absent.
|
|
60
70
|
|
|
61
|
-
|
|
71
|
+
Point it at a real model by swapping the provider — or omit `provider` entirely and let the library
|
|
72
|
+
detect one this machine can use, ending at the free local model:
|
|
62
73
|
|
|
63
74
|
```ts
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
const consensus = await judge({
|
|
67
|
-
provider: makeProvider({ provider: "claude-cli" }),
|
|
68
|
-
system: "You evaluate whether a page satisfies an assertion.",
|
|
69
|
-
user: "# Assertion\nThe page documents authentication.\n\n# Page\n...",
|
|
70
|
-
runs: 3,
|
|
71
|
-
});
|
|
72
|
-
|
|
73
|
-
consensus.verdict; // "pass" | "fail" — partial counts as fail
|
|
74
|
-
consensus.zone; // "auto-pass" | "auto-fail" | "human-review"
|
|
75
|
-
consensus.agreement; // 0..1 across non-errored runs
|
|
75
|
+
const provider = await makeProviderAsync({});
|
|
76
76
|
```
|
|
77
77
|
|
|
78
|
-
Only a **unanimous, high-confidence** ensemble auto-resolves. Anything split, low-confidence, or
|
|
79
|
-
containing an errored run routes to `human-review` — an errored run can never produce a silent pass.
|
|
80
|
-
|
|
81
|
-
### Caching
|
|
82
|
-
|
|
83
|
-
Key composition stays with you, because each consumer has a different notion of what should
|
|
84
|
-
invalidate an entry (page body, prompt version, ensemble size, requested fields):
|
|
85
|
-
|
|
86
|
-
```ts
|
|
87
|
-
import { JsonCache, buildCacheKey, runEnsemble, sha256 } from "@hawkeyexl/inference";
|
|
88
|
-
|
|
89
|
-
const cache = new JsonCache(".mytool/cache", true, "mytool");
|
|
90
|
-
const cacheKey = buildCacheKey([
|
|
91
|
-
provider.provider(),
|
|
92
|
-
provider.modelName(),
|
|
93
|
-
`v${MY_PROMPT_VERSION}`,
|
|
94
|
-
`r${runs}`,
|
|
95
|
-
sha256(pageBody), // pre-hash long parts
|
|
96
|
-
]);
|
|
97
|
-
|
|
98
|
-
const judgeRuns = await runEnsemble({ provider, system, user, runs, cache, cacheKey });
|
|
99
|
-
```
|
|
100
|
-
|
|
101
|
-
Cached runs come back flagged `cached: true`, so `costOfRuns` correctly charges nothing for a
|
|
102
|
-
replay. Cache write failures warn once and continue — a read-only workspace must not abort a run
|
|
103
|
-
whose inference already succeeded and was already paid for.
|
|
104
|
-
|
|
105
|
-
### Cost
|
|
106
|
-
|
|
107
|
-
```ts
|
|
108
|
-
import { costOfRuns, pricingFor } from "@hawkeyexl/inference";
|
|
109
|
-
|
|
110
|
-
const pricing = pricingFor(provider.modelName(), configOverride);
|
|
111
|
-
const usd = costOfRuns(judgeRuns, pricing);
|
|
112
|
-
```
|
|
113
|
-
|
|
114
|
-
An unknown model returns `undefined` pricing and costs `0` — **unknown, never a guess**. A
|
|
115
|
-
fabricated price is worse than an absent one when a budget gate depends on it.
|
|
116
|
-
|
|
117
78
|
## Providers
|
|
118
79
|
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
[ADR 01000](adrs/01000-library-owned-provider-spec.md)).
|
|
122
|
-
|
|
123
|
-
| `provider` | Structured output via | Key | Reports usage |
|
|
124
|
-
|---|---|---|---|
|
|
80
|
+
| `provider` | Structured output via | Credential | Reports usage |
|
|
81
|
+
|---|---|---|:---:|
|
|
125
82
|
| `anthropic` | forced tool call | `ANTHROPIC_API_KEY` | yes |
|
|
126
83
|
| `openai` | strict `json_schema`, falls back to `json_object` | `OPENAI_API_KEY` | yes |
|
|
127
|
-
| `claude-cli` | schema in the prompt, `--output-format json` | local `claude` auth | no |
|
|
84
|
+
| `claude-cli` | schema in the prompt, `--output-format json` | local `claude` auth | **no** |
|
|
85
|
+
| `llama-cpp` | GBNF grammar compiled from the schema | — (runs locally) | yes |
|
|
128
86
|
| `mock` | scripted responses | — | synthetic |
|
|
129
87
|
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
apiKeyEnv?: string | null; // default ANTHROPIC_API_KEY / OPENAI_API_KEY
|
|
135
|
-
baseUrl?: string; // openai only, default https://api.openai.com/v1
|
|
136
|
-
command?: string; // claude-cli only, default "claude"
|
|
137
|
-
timeoutMs?: number; // claude-cli only, default 180000
|
|
138
|
-
pricing?: Pricing; // override the built-in table
|
|
139
|
-
anthropic?: AnthropicProviderOptions; // e.g. toolName, maxTokens
|
|
140
|
-
openai?: OpenAICompatProviderOptions; // e.g. schemaName
|
|
141
|
-
exec?: ExecFn; // test seam for claude-cli
|
|
142
|
-
mockResponses?: MockResponse[];
|
|
143
|
-
}
|
|
144
|
-
```
|
|
145
|
-
|
|
146
|
-
`resolveProviderIdentity(spec)` returns `{ provider, model }` **without constructing anything** —
|
|
147
|
-
cache keys and pricing need the identity, but a fully-cached run should not require an API key.
|
|
148
|
-
|
|
149
|
-
Notes on the non-obvious bits:
|
|
150
|
-
|
|
151
|
-
- **`openai`** targets any `/chat/completions` server (OpenAI, Azure, Ollama, Groq, Together). It
|
|
152
|
-
prefers strict `json_schema`, and `toStrictSchema` rewrites your schema into the strict subset
|
|
153
|
-
(every property in `required`, optionality as a `null` type union, unsupported keywords dropped);
|
|
154
|
-
nulls are stripped back out of the response. If the server rejects `response_format`, it
|
|
155
|
-
permanently falls back to `json_object` with the schema in the prompt. Keyless local servers are
|
|
156
|
-
allowed — only `api.openai.com` requires a key.
|
|
157
|
-
- **`claude-cli`** uses your local Claude CLI auth, so no API key. The prompt goes over **stdin**,
|
|
158
|
-
never argv: user content routinely exceeds the ~32K Windows command-line limit.
|
|
159
|
-
|
|
160
|
-
## Testing against this library
|
|
161
|
-
|
|
162
|
-
`MockProvider` is exported for exactly this. No network required:
|
|
163
|
-
|
|
164
|
-
```ts
|
|
165
|
-
import { MockProvider, mockVerdict, runEnsemble } from "@hawkeyexl/inference";
|
|
166
|
-
|
|
167
|
-
const provider = new MockProvider([mockVerdict("pass", 0.95)]); // cycles when exhausted
|
|
168
|
-
const runs = await runEnsemble({ provider, system, user, runs: 3 });
|
|
169
|
-
provider.requests; // every request seen, in order
|
|
170
|
-
```
|
|
171
|
-
|
|
172
|
-
Script an error with `{ error: "429 rate limited" }` to exercise your failure paths.
|
|
173
|
-
|
|
174
|
-
## API
|
|
175
|
-
|
|
176
|
-
Everything exports from the package root.
|
|
177
|
-
|
|
178
|
-
**Providers** — `makeProvider`, `resolveProviderIdentity`, `DEFAULT_MODELS`,
|
|
179
|
-
`DEFAULT_OPENAI_BASE_URL`, `AnthropicProvider`, `OpenAICompatProvider`, `ClaudeCliProvider`,
|
|
180
|
-
`MockProvider`, `mockVerdict`, `extractJson`, `toStrictSchema`, `stripNulls`, `realExec`
|
|
181
|
-
|
|
182
|
-
**Completion** — `completeValidatedJSON`, `validatorFor`
|
|
183
|
-
|
|
184
|
-
**Cache** — `JsonCache`, `buildCacheKey`, `sha256`
|
|
88
|
+
Omit `provider` and the highest-priority one this machine can actually use is detected, ending at
|
|
89
|
+
`llama-cpp` — which needs no key, and whose native binding is installed on demand into
|
|
90
|
+
`~/.hawkeyexl-inference/runtime` if it is missing. That install warns once and is refused by
|
|
91
|
+
`INFERENCE_NO_AUTO_INSTALL`; it never touches your `package.json`, lockfile or `node_modules`.
|
|
185
92
|
|
|
186
|
-
|
|
93
|
+
Usage reporting is the column that decides whether cost accounting works: a provider that reports no
|
|
94
|
+
tokens makes a budget gate inert. See
|
|
95
|
+
[Choose a provider](https://hawkeyexl.github.io/inference/get-started/choose-a-provider/).
|
|
187
96
|
|
|
188
|
-
|
|
189
|
-
`DEFAULT_ZONES`
|
|
97
|
+
## Documentation
|
|
190
98
|
|
|
191
|
-
|
|
99
|
+
| Track | What it covers |
|
|
100
|
+
|---|---|
|
|
101
|
+
| [Get started](https://hawkeyexl.github.io/inference/get-started/) | Install, one validated call with no key, choosing a provider |
|
|
102
|
+
| [Judge & consensus](https://hawkeyexl.github.io/inference/judge/) | Ensembles, consensus math, confidence zones, caching, budgets |
|
|
103
|
+
| [Structured extraction](https://hawkeyexl.github.io/inference/extract/) | One schema-constrained call, honest failures, the subprocess seam |
|
|
104
|
+
| [Run models locally](https://hawkeyexl.github.io/inference/local/) | GGUF weights in-process, model selection, managing weights on disk |
|
|
105
|
+
| [Keep it working](https://hawkeyexl.github.io/inference/keep-it-working/testing/) | Testing without a network, upgrading without losing a cache |
|
|
106
|
+
| [Reference](https://hawkeyexl.github.io/inference/reference/providers/) | Full signatures for every export |
|
|
192
107
|
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
`ConsensusResult`, `Match`, `Zone`, `ZoneThresholds`, `EnsembleOptions`, `ExecFn`, `ExecResult`,
|
|
196
|
-
`ExecOptions`, `MockResponse`.
|
|
108
|
+
Who the docs serve and why each page exists lives in
|
|
109
|
+
[docs/content-strategy/](docs/content-strategy).
|
|
197
110
|
|
|
198
111
|
## Design decisions
|
|
199
112
|
|
|
@@ -205,6 +118,19 @@ Recorded as ADRs in [adrs/](adrs):
|
|
|
205
118
|
canonical verdict schema with a per-consumer override seam
|
|
206
119
|
- [01002](adrs/01002-best-of-merge-of-three-forks.md) — which fork won for each merged file, so the
|
|
207
120
|
losing variants are not reintroduced
|
|
121
|
+
- [01003](adrs/01003-in-process-local-models-via-node-llama-cpp.md) — in-process local models via
|
|
122
|
+
node-llama-cpp, why selectors need an async factory, and why the catalog pins exact blob paths
|
|
123
|
+
- [01004](adrs/01004-provider-auto-detection.md) — detect an available provider when none is
|
|
124
|
+
specified, ending at the local model
|
|
125
|
+
- [01005](adrs/01005-docset-strategy-and-executable-examples.md) — a CUJ-first documentation set,
|
|
126
|
+
with samples that CI executes
|
|
127
|
+
- [01006](adrs/01006-documenting-failure-and-orchestration.md) — document failure and orchestration,
|
|
128
|
+
and gate both against the source
|
|
129
|
+
- [01007](adrs/01007-harden-two-operational-failure-paths.md) — harden two operational failure paths:
|
|
130
|
+
non-JSON CLI output, and an unsupported Node
|
|
131
|
+
- [01008](adrs/01008-auto-install-the-local-runtime.md) — auto-install the local runtime into a
|
|
132
|
+
library-owned prefix, why the shim beats `createRequire`, and why a model without a provider is
|
|
133
|
+
now an error
|
|
208
134
|
|
|
209
135
|
## License
|
|
210
136
|
|
package/dist/index.d.ts
CHANGED
|
@@ -9,6 +9,9 @@ declare class InferenceError extends Error {
|
|
|
9
9
|
constructor(message: string);
|
|
10
10
|
}
|
|
11
11
|
|
|
12
|
+
/** Test seam: reset the once-per-process Node version warning. */
|
|
13
|
+
declare function resetNodeVersionWarning(): void;
|
|
14
|
+
|
|
12
15
|
/**
|
|
13
16
|
* The provider contract. A provider turns a (system, user, schema) request
|
|
14
17
|
* into schema-conforming JSON. `provider()` and `modelName()` feed cache keys
|
|
@@ -173,9 +176,110 @@ declare function mockVerdict(match: "pass" | "fail" | "partial", confidence: num
|
|
|
173
176
|
json: unknown;
|
|
174
177
|
};
|
|
175
178
|
|
|
176
|
-
|
|
179
|
+
interface LlamaPromptOptions {
|
|
180
|
+
/** JSON Schema converted to a GBNF grammar by the runtime. */
|
|
181
|
+
schema: Record<string, unknown>;
|
|
182
|
+
temperature: number;
|
|
183
|
+
/** Thinking budget; 0 disables it. See the note in `completeJSON`. */
|
|
184
|
+
thoughtTokens: number;
|
|
185
|
+
maxTokens?: number;
|
|
186
|
+
}
|
|
187
|
+
interface LlamaPromptResult {
|
|
188
|
+
text: string;
|
|
189
|
+
usage?: TokenUsage;
|
|
190
|
+
/**
|
|
191
|
+
* Why generation stopped. `"maxTokens"` means the output was cut off, so the
|
|
192
|
+
* text is almost certainly truncated JSON — see the guard in `completeJSON`.
|
|
193
|
+
*/
|
|
194
|
+
stopReason?: string;
|
|
195
|
+
}
|
|
196
|
+
interface LlamaSession {
|
|
197
|
+
prompt(text: string, options: LlamaPromptOptions): Promise<LlamaPromptResult>;
|
|
198
|
+
dispose(): Promise<void>;
|
|
199
|
+
}
|
|
200
|
+
interface LlamaLoadedModel {
|
|
201
|
+
createSession(systemPrompt: string): Promise<LlamaSession>;
|
|
202
|
+
dispose(): Promise<void>;
|
|
203
|
+
}
|
|
204
|
+
/**
|
|
205
|
+
* The whole of `node-llama-cpp` that this provider uses. Kept this narrow so a
|
|
206
|
+
* test fake is a few lines and so the real adapter is the only place that
|
|
207
|
+
* knows the upstream API shape.
|
|
208
|
+
*/
|
|
209
|
+
interface LlamaRuntime {
|
|
210
|
+
/**
|
|
211
|
+
* Resolve an `hf:` URI or path to a local file inside `directory`,
|
|
212
|
+
* downloading if needed.
|
|
213
|
+
*/
|
|
214
|
+
resolveModelFile(uri: string, directory: string): Promise<string>;
|
|
215
|
+
loadModel(path: string): Promise<LlamaLoadedModel>;
|
|
216
|
+
/** Memory available for weights, in bytes — VRAM if there is a GPU, else RAM. */
|
|
217
|
+
getMemoryBudgetBytes(): Promise<number>;
|
|
218
|
+
}
|
|
219
|
+
interface LlamaCppProviderOptions {
|
|
220
|
+
/** Injected for tests; defaults to the real `node-llama-cpp` adapter. */
|
|
221
|
+
runtime?: LlamaRuntime;
|
|
222
|
+
/**
|
|
223
|
+
* Thinking budget in tokens, default 0.
|
|
224
|
+
*
|
|
225
|
+
* Gemma 4 has a thinking mode, but a grammar constrains generation from
|
|
226
|
+
* token 0 — so an unbudgeted model starts reasoning and gets cut off
|
|
227
|
+
* mid-thought. Zero is the deterministic choice for judging; raise it if you
|
|
228
|
+
* want reasoning before the JSON.
|
|
229
|
+
*/
|
|
230
|
+
thoughtTokens?: number;
|
|
231
|
+
maxTokens?: number;
|
|
232
|
+
/**
|
|
233
|
+
* Where to download and look for weights. Defaults to this library's own
|
|
234
|
+
* directory — see `defaultLlamaModelsDirectory`.
|
|
235
|
+
*/
|
|
236
|
+
modelsDirectory?: string;
|
|
237
|
+
}
|
|
238
|
+
/**
|
|
239
|
+
* Free every loaded model.
|
|
240
|
+
*
|
|
241
|
+
* A standalone function rather than a `dispose()` on `InferenceProvider`:
|
|
242
|
+
* adding one to the contract would make all five providers carry a lifecycle
|
|
243
|
+
* only this one has. Short-lived processes can skip it.
|
|
244
|
+
*/
|
|
245
|
+
declare function disposeLlamaModels(): Promise<void>;
|
|
246
|
+
declare class LlamaCppProvider implements InferenceProvider {
|
|
247
|
+
private readonly model;
|
|
248
|
+
private readonly uri;
|
|
249
|
+
private readonly runtime;
|
|
250
|
+
private readonly thoughtTokens;
|
|
251
|
+
private readonly maxTokens;
|
|
252
|
+
private readonly modelsDirectory;
|
|
253
|
+
/**
|
|
254
|
+
* Loaded-model key: the same URI in two directories is two different files.
|
|
255
|
+
* Built with `buildCacheKey` so its parts are length-prefixed — a plain join
|
|
256
|
+
* would let two different (directory, uri) pairs collide and hand a provider
|
|
257
|
+
* back the wrong weights.
|
|
258
|
+
*/
|
|
259
|
+
private readonly cacheKey;
|
|
260
|
+
constructor(model: string, options?: LlamaCppProviderOptions);
|
|
261
|
+
provider(): string;
|
|
262
|
+
modelName(): string;
|
|
263
|
+
completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
|
|
264
|
+
private load;
|
|
265
|
+
}
|
|
266
|
+
/**
|
|
267
|
+
* Lazy adapter over the real `node-llama-cpp`. Every method defers to the
|
|
268
|
+
* dynamic import, so constructing a provider for a fully-cached run never
|
|
269
|
+
* loads the native binary.
|
|
270
|
+
*/
|
|
271
|
+
declare function defaultLlamaRuntime(): LlamaRuntime;
|
|
272
|
+
|
|
273
|
+
type ProviderName = "anthropic" | "openai" | "claude-cli" | "mock" | "llama-cpp";
|
|
274
|
+
/** A concrete provider, or `"auto"` to detect one. */
|
|
275
|
+
type ProviderSelector = ProviderName | "auto";
|
|
177
276
|
interface ProviderSpec {
|
|
178
|
-
|
|
277
|
+
/**
|
|
278
|
+
* Omitting this is identical to `"auto"`: the highest-priority provider this
|
|
279
|
+
* machine can actually use is detected, ending at the free local model.
|
|
280
|
+
* Resolving it needs `makeProviderAsync`/`resolveProviderIdentityAsync`.
|
|
281
|
+
*/
|
|
282
|
+
provider?: ProviderSelector;
|
|
179
283
|
/** null/undefined selects the per-provider default. */
|
|
180
284
|
model?: string | null;
|
|
181
285
|
/** Env var NAME holding the API key; null/undefined selects the default. */
|
|
@@ -195,23 +299,77 @@ interface ProviderSpec {
|
|
|
195
299
|
/** Provider-specific tuning, ignored by the other providers. */
|
|
196
300
|
anthropic?: AnthropicProviderOptions;
|
|
197
301
|
openai?: OpenAICompatProviderOptions;
|
|
302
|
+
llamaCpp?: LlamaCppProviderOptions;
|
|
198
303
|
/** Test seam for the claude-cli provider. */
|
|
199
304
|
exec?: ExecFn;
|
|
305
|
+
/** Test seam for the llama-cpp provider. */
|
|
306
|
+
llamaRuntime?: LlamaRuntime;
|
|
200
307
|
/** Scripted responses for the mock provider; defaults to a single empty object. */
|
|
201
308
|
mockResponses?: MockResponse[];
|
|
202
309
|
}
|
|
203
310
|
declare const DEFAULT_MODELS: Record<ProviderName, string>;
|
|
204
311
|
declare const DEFAULT_OPENAI_BASE_URL = "https://api.openai.com/v1";
|
|
312
|
+
interface ProviderIdentity {
|
|
313
|
+
/** Always concrete — never `"auto"`, so it is safe as cache-key material. */
|
|
314
|
+
provider: ProviderName;
|
|
315
|
+
model: string;
|
|
316
|
+
}
|
|
205
317
|
/**
|
|
206
318
|
* Resolve the provider name and model WITHOUT constructing the provider —
|
|
207
319
|
* cache keys and pricing need the identity, but construction may require an
|
|
208
320
|
* API key that a fully-cached run never uses.
|
|
321
|
+
*
|
|
322
|
+
* Throws for an unresolved `llama-cpp` selector. That is deliberate: picking a
|
|
323
|
+
* tier weighs GPU VRAM, which needs an await, and returning the literal
|
|
324
|
+
* "auto" as cache-key material would let a 2 GB and a 12 GB model share cached
|
|
325
|
+
* verdicts — and give different results per machine under one key. Use
|
|
326
|
+
* `resolveProviderIdentityAsync` for selectors.
|
|
209
327
|
*/
|
|
210
|
-
declare function resolveProviderIdentity(spec: ProviderSpec):
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
328
|
+
declare function resolveProviderIdentity(spec: ProviderSpec): ProviderIdentity;
|
|
329
|
+
/**
|
|
330
|
+
* Selector-aware identity resolution. Returns the CONCRETE model a selector
|
|
331
|
+
* resolved to, so the cache key names the weights that actually ran.
|
|
332
|
+
*
|
|
333
|
+
* Every other provider delegates to the synchronous form, so a consumer can
|
|
334
|
+
* switch to this wholesale.
|
|
335
|
+
*/
|
|
336
|
+
declare function resolveProviderIdentityAsync(spec: ProviderSpec): Promise<ProviderIdentity>;
|
|
214
337
|
declare function makeProvider(spec: ProviderSpec): InferenceProvider;
|
|
338
|
+
/**
|
|
339
|
+
* Selector-aware provider construction. Resolves a `llama-cpp` selector
|
|
340
|
+
* against this machine first, so the returned provider's `modelName()` — and
|
|
341
|
+
* therefore the cache key — names the weights it will actually load.
|
|
342
|
+
*
|
|
343
|
+
* Every other provider delegates to `makeProvider`.
|
|
344
|
+
*/
|
|
345
|
+
declare function makeProviderAsync(spec: ProviderSpec): Promise<InferenceProvider>;
|
|
346
|
+
|
|
347
|
+
/**
|
|
348
|
+
* Priority order. `mock` is deliberately absent: it answers `{ json: {} }`
|
|
349
|
+
* unless scripted, which would sail through as a non-error result — the exact
|
|
350
|
+
* opposite of the "an errored run is recorded, never coerced" invariant the
|
|
351
|
+
* consuming eval tools depend on. It must always be asked for by name.
|
|
352
|
+
*/
|
|
353
|
+
declare const DETECTION_ORDER: readonly ProviderName[];
|
|
354
|
+
/** Test seam: forget the memoised Claude CLI probes. */
|
|
355
|
+
declare function resetClaudeCliProbe(): void;
|
|
356
|
+
/**
|
|
357
|
+
* Every provider this machine could use right now, in priority order.
|
|
358
|
+
*
|
|
359
|
+
* Useful for showing a picker or explaining a fallback; `detectProvider` is
|
|
360
|
+
* the same sweep with the first hit returned.
|
|
361
|
+
*/
|
|
362
|
+
declare function availableProviders(spec?: ProviderSpec): Promise<ProviderName[]>;
|
|
363
|
+
/**
|
|
364
|
+
* The highest-priority provider this machine can use.
|
|
365
|
+
*
|
|
366
|
+
* Throws an `InferenceError` naming every provider and why each was
|
|
367
|
+
* unavailable — far more actionable than the `Unknown provider "undefined"`
|
|
368
|
+
* this replaces.
|
|
369
|
+
*/
|
|
370
|
+
declare function detectProvider(spec?: ProviderSpec): Promise<ProviderName>;
|
|
371
|
+
/** Test seam: reset the once-per-process auto-detection warnings. */
|
|
372
|
+
declare function resetProviderDetectionWarning(): void;
|
|
215
373
|
|
|
216
374
|
declare class ClaudeCliProvider implements InferenceProvider {
|
|
217
375
|
private readonly model;
|
|
@@ -224,6 +382,179 @@ declare class ClaudeCliProvider implements InferenceProvider {
|
|
|
224
382
|
completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
|
|
225
383
|
}
|
|
226
384
|
|
|
385
|
+
/**
|
|
386
|
+
* Where this library downloads weights — its OWN directory, not
|
|
387
|
+
* node-llama-cpp's global `~/.node-llama-cpp/models`.
|
|
388
|
+
*
|
|
389
|
+
* That default is shared: node-llama-cpp's CLI writes there, as does anything
|
|
390
|
+
* else on the machine using it. Owning a directory outright means clearing it
|
|
391
|
+
* can never destroy a model this library did not download, and it keeps one
|
|
392
|
+
* copy shared across every consumer of this package on the machine.
|
|
393
|
+
*
|
|
394
|
+
* `INFERENCE_MODELS_DIR` overrides it — useful for CI or a shared volume.
|
|
395
|
+
*/
|
|
396
|
+
declare function defaultLlamaModelsDirectory(): string;
|
|
397
|
+
/** Size tiers, smallest first. Order is load-bearing for `tierForBudget`. */
|
|
398
|
+
declare const LLAMA_TIERS: readonly ["fast", "balanced", "quality"];
|
|
399
|
+
type LlamaTier = (typeof LLAMA_TIERS)[number];
|
|
400
|
+
/** Model selectors — resolved against hardware, never used as a cache key. */
|
|
401
|
+
declare const LLAMA_SELECTORS: readonly ["auto", "fast", "balanced", "quality"];
|
|
402
|
+
type LlamaSelector = (typeof LLAMA_SELECTORS)[number];
|
|
403
|
+
interface LlamaModelEntry {
|
|
404
|
+
/** `hf:` URI pinned to one blob, handed to `resolveModelFile` as-is. */
|
|
405
|
+
readonly uri: string;
|
|
406
|
+
/** Size of that blob in bytes, as reported by the Hugging Face API. */
|
|
407
|
+
readonly sizeBytes: number;
|
|
408
|
+
readonly license: string;
|
|
409
|
+
/** Absent for entries that are selectable by alias but not by tier. */
|
|
410
|
+
readonly tier?: LlamaTier;
|
|
411
|
+
/** Human note for `LLAMA_MODELS` readers deciding what to download. */
|
|
412
|
+
readonly notes: string;
|
|
413
|
+
}
|
|
414
|
+
/**
|
|
415
|
+
* Frozen per entry, not just at the top level.
|
|
416
|
+
*
|
|
417
|
+
* A shallow freeze leaves the entries writable, and this catalog is exported
|
|
418
|
+
* for consumers to read: a stray write to `sizeBytes` silently re-points
|
|
419
|
+
* `tierForBudget` process-wide, and a write to `uri` defeats the pinned-blob
|
|
420
|
+
* invariant the whole catalog exists to hold (ADR 01003).
|
|
421
|
+
*/
|
|
422
|
+
declare const LLAMA_MODELS: Readonly<Record<string, LlamaModelEntry>>;
|
|
423
|
+
declare function isLlamaSelector(model: string): model is LlamaSelector;
|
|
424
|
+
/**
|
|
425
|
+
* Largest tier whose weights fit the memory budget with headroom. Lands at
|
|
426
|
+
* roughly: >=24 GB -> quality, >=15 GB -> balanced, else fast.
|
|
427
|
+
*
|
|
428
|
+
* Sized off the catalog's recorded bytes rather than parameter counts: Gemma
|
|
429
|
+
* 4's E-series are per-layer-embedding models whose footprint does not track
|
|
430
|
+
* "effective params" (E4B is 4.5B effective but 15 GB at BF16).
|
|
431
|
+
*/
|
|
432
|
+
declare function tierForBudget(budgetBytes: number): LlamaTier;
|
|
433
|
+
/**
|
|
434
|
+
* Catalog alias backing a tier. Selectors resolve to this rather than to a raw
|
|
435
|
+
* URI so the identity — and therefore the cache key — stays human-readable.
|
|
436
|
+
*/
|
|
437
|
+
declare function aliasForTier(tier: LlamaTier): string;
|
|
438
|
+
/** The pinned URI backing a tier keyword. */
|
|
439
|
+
declare function uriForTier(tier: LlamaTier): string;
|
|
440
|
+
/**
|
|
441
|
+
* Turn a concrete model reference — a curated alias, an `hf:` URI, or a local
|
|
442
|
+
* path — into something `resolveModelFile` accepts.
|
|
443
|
+
*
|
|
444
|
+
* Selectors are rejected rather than guessed at: they need a hardware probe,
|
|
445
|
+
* which is async, and this runs on the synchronous cache-key path. Resolving
|
|
446
|
+
* one here from RAM alone would emit a key naming a model the provider then
|
|
447
|
+
* did not load.
|
|
448
|
+
*/
|
|
449
|
+
declare function resolveLlamaModelRef(model: string): string;
|
|
450
|
+
/**
|
|
451
|
+
* The catalog's pinned filename for a model reference.
|
|
452
|
+
*
|
|
453
|
+
* Strips a `#branch` fragment (node-llama-cpp accepts
|
|
454
|
+
* `hf:user/repo/file.gguf#branch`) and splits on both separators, so a Windows
|
|
455
|
+
* path resolves too. Getting either wrong makes callers silently match nothing.
|
|
456
|
+
*/
|
|
457
|
+
declare function blobNameFor(model: string): string;
|
|
458
|
+
/**
|
|
459
|
+
* Are this model's weights already on disk?
|
|
460
|
+
*
|
|
461
|
+
* A `.ipull` partial counts as NOT downloaded: it cannot be loaded, so
|
|
462
|
+
* treating it as present would skip the download warning and then stall on a
|
|
463
|
+
* download anyway.
|
|
464
|
+
*/
|
|
465
|
+
declare function isModelDownloaded(model: string, directory: string): boolean;
|
|
466
|
+
|
|
467
|
+
interface ClearedModelFile {
|
|
468
|
+
path: string;
|
|
469
|
+
sizeBytes: number;
|
|
470
|
+
}
|
|
471
|
+
interface ClearLlamaModelsResult {
|
|
472
|
+
/** Files removed, or that would be removed under `dryRun`. */
|
|
473
|
+
files: ClearedModelFile[];
|
|
474
|
+
freedBytes: number;
|
|
475
|
+
directory: string;
|
|
476
|
+
dryRun: boolean;
|
|
477
|
+
}
|
|
478
|
+
interface ClearLlamaModelsOptions {
|
|
479
|
+
/** Defaults to this library's own models directory. */
|
|
480
|
+
directory?: string;
|
|
481
|
+
/** Clear only these models (alias or `hf:` URI); default clears all. */
|
|
482
|
+
models?: string[];
|
|
483
|
+
/** Report what would be removed without deleting anything. */
|
|
484
|
+
dryRun?: boolean;
|
|
485
|
+
}
|
|
486
|
+
/**
|
|
487
|
+
* Delete downloaded GGUF weights and interrupted partial downloads.
|
|
488
|
+
*
|
|
489
|
+
* Loaded models are disposed first: on Windows the weights are memory-mapped
|
|
490
|
+
* while loaded, and deleting an open file fails with EBUSY/EPERM.
|
|
491
|
+
*/
|
|
492
|
+
declare function clearLlamaModels(options?: ClearLlamaModelsOptions): Promise<ClearLlamaModelsResult>;
|
|
493
|
+
|
|
494
|
+
interface RuntimeInstallOptions {
|
|
495
|
+
/** Defaults to this library's own runtime directory. */
|
|
496
|
+
directory?: string;
|
|
497
|
+
/** Injected for tests; defaults to the real process runner. */
|
|
498
|
+
exec?: ExecFn;
|
|
499
|
+
/** Injected for tests; defaults to `process.env`. */
|
|
500
|
+
env?: Record<string, string | undefined>;
|
|
501
|
+
timeoutMs?: number;
|
|
502
|
+
/** Test seam: how the shim is imported once it exists. */
|
|
503
|
+
importShim?: (url: string) => Promise<unknown>;
|
|
504
|
+
/**
|
|
505
|
+
* Test seam: how the consumer's own copy is looked for.
|
|
506
|
+
*
|
|
507
|
+
* Whether an optional peer is installed is a property of the machine, and the
|
|
508
|
+
* behaviour that matters most here — that probing installs nothing — is
|
|
509
|
+
* unobservable on a machine that already has it. Injecting the lookup is the
|
|
510
|
+
* only way to assert it deterministically.
|
|
511
|
+
*/
|
|
512
|
+
probeImport?: () => Promise<unknown>;
|
|
513
|
+
}
|
|
514
|
+
/**
|
|
515
|
+
* Where this library installs the binding — its OWN directory, beside the
|
|
516
|
+
* models directory and for the same reason.
|
|
517
|
+
*
|
|
518
|
+
* `INFERENCE_RUNTIME_DIR` overrides it, mirroring `INFERENCE_MODELS_DIR`.
|
|
519
|
+
*/
|
|
520
|
+
declare function defaultLlamaRuntimeDirectory(env?: Record<string, string | undefined>): string;
|
|
521
|
+
/** Whether the binding can be used, and at what cost, WITHOUT installing it. */
|
|
522
|
+
type RuntimeStatus =
|
|
523
|
+
/** Importable right now — either the consumer's own copy or a filled prefix. */
|
|
524
|
+
{
|
|
525
|
+
state: "present";
|
|
526
|
+
}
|
|
527
|
+
/** Absent, but an install is permitted. Using it will fetch a native module. */
|
|
528
|
+
| {
|
|
529
|
+
state: "installable";
|
|
530
|
+
directory: string;
|
|
531
|
+
}
|
|
532
|
+
/** Absent and installing is refused, so this provider cannot be used. */
|
|
533
|
+
| {
|
|
534
|
+
state: "refused";
|
|
535
|
+
reason: string;
|
|
536
|
+
};
|
|
537
|
+
/**
|
|
538
|
+
* Can the local runtime be used, and would using it install anything?
|
|
539
|
+
*
|
|
540
|
+
* Detection needs this rather than simply calling `getMemoryBudgetBytes`:
|
|
541
|
+
* that goes through `importNodeLlamaCpp` and would install, which turns
|
|
542
|
+
* `availableProviders()` — a function whose entire job is to REPORT what is
|
|
543
|
+
* usable — into one that changes what is usable. Probing must not have the side
|
|
544
|
+
* effect it is probing for.
|
|
545
|
+
*/
|
|
546
|
+
declare function nodeLlamaCppStatus(options?: RuntimeInstallOptions): Promise<RuntimeStatus>;
|
|
547
|
+
/** Test seam: forget in-flight installs and re-arm the one-time warning. */
|
|
548
|
+
declare function resetRuntimeInstall(): void;
|
|
549
|
+
/**
|
|
550
|
+
* Import `node-llama-cpp` from the library-owned prefix, installing it first if
|
|
551
|
+
* it is not there.
|
|
552
|
+
*
|
|
553
|
+
* Callers reach this only after a plain `import("node-llama-cpp")` has already
|
|
554
|
+
* failed — a consumer who installed the peer themselves never gets here.
|
|
555
|
+
*/
|
|
556
|
+
declare function importNodeLlamaCpp(options?: RuntimeInstallOptions): Promise<unknown>;
|
|
557
|
+
|
|
227
558
|
declare const realExec: ExecFn;
|
|
228
559
|
|
|
229
560
|
/** One attempt at a schema-constrained completion. */
|
|
@@ -375,4 +706,4 @@ declare function judge(options: EnsembleOptions & {
|
|
|
375
706
|
/** Test seam: reset the once-per-process temperature warning. */
|
|
376
707
|
declare function resetTemperatureWarning(): void;
|
|
377
708
|
|
|
378
|
-
export { AnthropicProvider, type AnthropicProviderOptions, ClaudeCliProvider, type CompleteJSONRequest, type CompleteJSONResponse, type CompleteValidatedOptions, type ConsensusResult, DEFAULT_MODELS, DEFAULT_OPENAI_BASE_URL, DEFAULT_ZONES, type EnsembleOptions, type ExecFn, type ExecOptions, type ExecResult, InferenceError, type InferenceProvider, type InferenceRun, JsonCache, type JudgeRun, type JudgeVerdict, type Match, MockProvider, type MockResponse, OpenAICompatProvider, type OpenAICompatProviderOptions, PRICE_TABLE, type Pricing, type ProviderName, type ProviderSpec, type TokenUsage, VERDICT_SCHEMA, type Zone, type ZoneThresholds, buildCacheKey, completeValidatedJSON, computeConsensus, costOfRuns, costOfUsage, extractJson, judge, makeProvider, mockVerdict, pricingFor, realExec, resetTemperatureWarning, resolveProviderIdentity, runEnsemble, sha256, stripNulls, toStrictSchema, validatorFor, zoneFor };
|
|
709
|
+
export { AnthropicProvider, type AnthropicProviderOptions, ClaudeCliProvider, type ClearLlamaModelsOptions, type ClearLlamaModelsResult, type ClearedModelFile, type CompleteJSONRequest, type CompleteJSONResponse, type CompleteValidatedOptions, type ConsensusResult, DEFAULT_MODELS, DEFAULT_OPENAI_BASE_URL, DEFAULT_ZONES, DETECTION_ORDER, type EnsembleOptions, type ExecFn, type ExecOptions, type ExecResult, InferenceError, type InferenceProvider, type InferenceRun, JsonCache, type JudgeRun, type JudgeVerdict, LLAMA_MODELS, LLAMA_SELECTORS, LLAMA_TIERS, LlamaCppProvider, type LlamaCppProviderOptions, type LlamaLoadedModel, type LlamaModelEntry, type LlamaPromptOptions, type LlamaPromptResult, type LlamaRuntime, type LlamaSelector, type LlamaSession, type LlamaTier, type Match, MockProvider, type MockResponse, OpenAICompatProvider, type OpenAICompatProviderOptions, PRICE_TABLE, type Pricing, type ProviderIdentity, type ProviderName, type ProviderSelector, type ProviderSpec, type RuntimeInstallOptions, type RuntimeStatus, type TokenUsage, VERDICT_SCHEMA, type Zone, type ZoneThresholds, aliasForTier, availableProviders, blobNameFor, buildCacheKey, clearLlamaModels, completeValidatedJSON, computeConsensus, costOfRuns, costOfUsage, defaultLlamaModelsDirectory, defaultLlamaRuntime, defaultLlamaRuntimeDirectory, detectProvider, disposeLlamaModels, extractJson, importNodeLlamaCpp, isLlamaSelector, isModelDownloaded, judge, makeProvider, makeProviderAsync, mockVerdict, nodeLlamaCppStatus, pricingFor, realExec, resetClaudeCliProbe, resetNodeVersionWarning, resetProviderDetectionWarning, resetRuntimeInstall, resetTemperatureWarning, resolveLlamaModelRef, resolveProviderIdentity, resolveProviderIdentityAsync, runEnsemble, sha256, stripNulls, tierForBudget, toStrictSchema, uriForTier, validatorFor, zoneFor };
|