@hawkeyexl/inference 0.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 inference contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,211 @@
1
+ # @hawkeyexl/inference
2
+
3
+ Shared LLM inference layer for the docs-as-tests toolchain: schema-constrained completion across
4
+ Anthropic, OpenAI-compatible, and Claude CLI providers, with result caching, cost accounting, and
5
+ an LLM-as-judge ensemble on top.
6
+
7
+ Extracted from three projects that had each grown their own copy —
8
+ [docevals](https://github.com/hawkeyexl/docevals), [dockg](https://github.com/hawkeyexl/dockg), and
9
+ [agentevals](https://github.com/hawkeyexl/agentevals) — so a provider fix lands once instead of
10
+ three times.
11
+
12
+ ## Install
13
+
14
+ ```bash
15
+ npm install @hawkeyexl/inference
16
+ ```
17
+
18
+ Requires Node 24+. ESM only.
19
+
20
+ ## What it does
21
+
22
+ Every consumer of this library wants the same narrow thing: **send a system prompt, a user prompt,
23
+ and a JSON Schema; get back JSON that validates against that schema, or a recorded error.** No
24
+ streaming, no multi-turn, no tool loops.
25
+
26
+ Two layers, one entry point:
27
+
28
+ - **Completion** — the provider contract, four providers, a content-addressed cache, a price table,
29
+ and a validate-and-retry wrapper. This is all dockg-style structured extraction needs.
30
+ - **Judge** — the canonical verdict schema, an N-run ensemble, consensus math, and confidence-zone
31
+ routing. Built on the completion layer; ignore it if you do not need it.
32
+
33
+ ## Quick start
34
+
35
+ ### Schema-constrained completion
36
+
37
+ ```ts
38
+ import { completeValidatedJSON, makeProvider } from "@hawkeyexl/inference";
39
+
40
+ const provider = makeProvider({ provider: "anthropic", model: "claude-sonnet-4-5" });
41
+
42
+ const run = await completeValidatedJSON<{ summary: string }>({
43
+ provider,
44
+ system: "You summarize documentation pages.",
45
+ user: pageBody,
46
+ schema: {
47
+ type: "object",
48
+ required: ["summary"],
49
+ properties: { summary: { type: "string" } },
50
+ additionalProperties: false,
51
+ },
52
+ });
53
+
54
+ if (run.error) console.error(run.error);
55
+ else console.log(run.result.summary, run.usage);
56
+ ```
57
+
58
+ `completeValidatedJSON` never throws on a model failure and never coerces a bad response. It
59
+ retries once, then returns a run with `error` set and `result` absent.
60
+
61
+ ### LLM-as-judge
62
+
63
+ ```ts
64
+ import { judge, makeProvider } from "@hawkeyexl/inference";
65
+
66
+ const consensus = await judge({
67
+ provider: makeProvider({ provider: "claude-cli" }),
68
+ system: "You evaluate whether a page satisfies an assertion.",
69
+ user: "# Assertion\nThe page documents authentication.\n\n# Page\n...",
70
+ runs: 3,
71
+ });
72
+
73
+ consensus.verdict; // "pass" | "fail" — partial counts as fail
74
+ consensus.zone; // "auto-pass" | "auto-fail" | "human-review"
75
+ consensus.agreement; // 0..1 across non-errored runs
76
+ ```
77
+
78
+ Only a **unanimous, high-confidence** ensemble auto-resolves. Anything split, low-confidence, or
79
+ containing an errored run routes to `human-review` — an errored run can never produce a silent pass.
80
+
81
+ ### Caching
82
+
83
+ Key composition stays with you, because each consumer has a different notion of what should
84
+ invalidate an entry (page body, prompt version, ensemble size, requested fields):
85
+
86
+ ```ts
87
+ import { JsonCache, buildCacheKey, runEnsemble, sha256 } from "@hawkeyexl/inference";
88
+
89
+ const cache = new JsonCache(".mytool/cache", true, "mytool");
90
+ const cacheKey = buildCacheKey([
91
+ provider.provider(),
92
+ provider.modelName(),
93
+ `v${MY_PROMPT_VERSION}`,
94
+ `r${runs}`,
95
+ sha256(pageBody), // pre-hash long parts
96
+ ]);
97
+
98
+ const judgeRuns = await runEnsemble({ provider, system, user, runs, cache, cacheKey });
99
+ ```
100
+
101
+ Cached runs come back flagged `cached: true`, so `costOfRuns` correctly charges nothing for a
102
+ replay. Cache write failures warn once and continue — a read-only workspace must not abort a run
103
+ whose inference already succeeded and was already paid for.
104
+
105
+ ### Cost
106
+
107
+ ```ts
108
+ import { costOfRuns, pricingFor } from "@hawkeyexl/inference";
109
+
110
+ const pricing = pricingFor(provider.modelName(), configOverride);
111
+ const usd = costOfRuns(judgeRuns, pricing);
112
+ ```
113
+
114
+ An unknown model returns `undefined` pricing and costs `0` — **unknown, never a guess**. A
115
+ fabricated price is worse than an absent one when a budget gate depends on it.
116
+
117
+ ## Providers
118
+
119
+ Constructed through `makeProvider(spec)`. The spec is a flat, library-owned shape — map your own
120
+ config into it rather than passing your config object (see
121
+ [ADR 01000](adrs/01000-library-owned-provider-spec.md)).
122
+
123
+ | `provider` | Structured output via | Key | Reports usage |
124
+ |---|---|---|---|
125
+ | `anthropic` | forced tool call | `ANTHROPIC_API_KEY` | yes |
126
+ | `openai` | strict `json_schema`, falls back to `json_object` | `OPENAI_API_KEY` | yes |
127
+ | `claude-cli` | schema in the prompt, `--output-format json` | local `claude` auth | no |
128
+ | `mock` | scripted responses | — | synthetic |
129
+
130
+ ```ts
131
+ interface ProviderSpec {
132
+ provider: "anthropic" | "openai" | "claude-cli" | "mock";
133
+ model?: string | null; // null/undefined -> per-provider default
134
+ apiKeyEnv?: string | null; // default ANTHROPIC_API_KEY / OPENAI_API_KEY
135
+ baseUrl?: string; // openai only, default https://api.openai.com/v1
136
+ command?: string; // claude-cli only, default "claude"
137
+ timeoutMs?: number; // claude-cli only, default 180000
138
+ pricing?: Pricing; // override the built-in table
139
+ anthropic?: AnthropicProviderOptions; // e.g. toolName, maxTokens
140
+ openai?: OpenAICompatProviderOptions; // e.g. schemaName
141
+ exec?: ExecFn; // test seam for claude-cli
142
+ mockResponses?: MockResponse[];
143
+ }
144
+ ```
145
+
146
+ `resolveProviderIdentity(spec)` returns `{ provider, model }` **without constructing anything** —
147
+ cache keys and pricing need the identity, but a fully-cached run should not require an API key.
148
+
149
+ Notes on the non-obvious bits:
150
+
151
+ - **`openai`** targets any `/chat/completions` server (OpenAI, Azure, Ollama, Groq, Together). It
152
+ prefers strict `json_schema`, and `toStrictSchema` rewrites your schema into the strict subset
153
+ (every property in `required`, optionality as a `null` type union, unsupported keywords dropped);
154
+ nulls are stripped back out of the response. If the server rejects `response_format`, it
155
+ permanently falls back to `json_object` with the schema in the prompt. Keyless local servers are
156
+ allowed — only `api.openai.com` requires a key.
157
+ - **`claude-cli`** uses your local Claude CLI auth, so no API key. The prompt goes over **stdin**,
158
+ never argv: user content routinely exceeds the ~32K Windows command-line limit.
159
+
160
+ ## Testing against this library
161
+
162
+ `MockProvider` is exported for exactly this. No network required:
163
+
164
+ ```ts
165
+ import { MockProvider, mockVerdict, runEnsemble } from "@hawkeyexl/inference";
166
+
167
+ const provider = new MockProvider([mockVerdict("pass", 0.95)]); // cycles when exhausted
168
+ const runs = await runEnsemble({ provider, system, user, runs: 3 });
169
+ provider.requests; // every request seen, in order
170
+ ```
171
+
172
+ Script an error with `{ error: "429 rate limited" }` to exercise your failure paths.
173
+
174
+ ## API
175
+
176
+ Everything exports from the package root.
177
+
178
+ **Providers** — `makeProvider`, `resolveProviderIdentity`, `DEFAULT_MODELS`,
179
+ `DEFAULT_OPENAI_BASE_URL`, `AnthropicProvider`, `OpenAICompatProvider`, `ClaudeCliProvider`,
180
+ `MockProvider`, `mockVerdict`, `extractJson`, `toStrictSchema`, `stripNulls`, `realExec`
181
+
182
+ **Completion** — `completeValidatedJSON`, `validatorFor`
183
+
184
+ **Cache** — `JsonCache`, `buildCacheKey`, `sha256`
185
+
186
+ **Cost** — `pricingFor`, `costOfUsage`, `costOfRuns`, `PRICE_TABLE`
187
+
188
+ **Judge** — `judge`, `runEnsemble`, `computeConsensus`, `zoneFor`, `VERDICT_SCHEMA`,
189
+ `DEFAULT_ZONES`
190
+
191
+ **Errors** — `InferenceError` (operational failures: missing key, unknown provider)
192
+
193
+ Types: `InferenceProvider`, `ProviderSpec`, `ProviderName`, `CompleteJSONRequest`,
194
+ `CompleteJSONResponse`, `InferenceRun`, `TokenUsage`, `Pricing`, `JudgeRun`, `JudgeVerdict`,
195
+ `ConsensusResult`, `Match`, `Zone`, `ZoneThresholds`, `EnsembleOptions`, `ExecFn`, `ExecResult`,
196
+ `ExecOptions`, `MockResponse`.
197
+
198
+ ## Design decisions
199
+
200
+ Recorded as ADRs in [adrs/](adrs):
201
+
202
+ - [01000](adrs/01000-library-owned-provider-spec.md) — a library-owned `ProviderSpec`, not consumer
203
+ config objects
204
+ - [01001](adrs/01001-single-entry-point-and-canonical-verdict-schema.md) — one entry point; a
205
+ canonical verdict schema with a per-consumer override seam
206
+ - [01002](adrs/01002-best-of-merge-of-three-forks.md) — which fork won for each merged file, so the
207
+ losing variants are not reintroduced
208
+
209
+ ## License
210
+
211
+ MIT
@@ -0,0 +1,371 @@
1
+ import { ValidateFunction } from 'ajv';
2
+
3
+ /**
4
+ * Shared error type. Consumers catch this to distinguish an inference-layer
5
+ * operational failure (missing API key, unknown provider) from their own
6
+ * domain errors, and typically map it to their own exit code.
7
+ */
8
+ declare class InferenceError extends Error {
9
+ constructor(message: string);
10
+ }
11
+
12
+ /**
13
+ * The provider contract. A provider turns a (system, user, schema) request
14
+ * into schema-conforming JSON. `provider()` and `modelName()` feed cache keys
15
+ * and pricing lookups, so two providers/models never share a cached result.
16
+ *
17
+ * This is deliberately the narrowest useful surface: no streaming, no
18
+ * multi-turn, no tool loops. Everything downstream of it — judging,
19
+ * extraction, classification — is schema-constrained single-shot completion.
20
+ */
21
+ interface CompleteJSONRequest {
22
+ system: string;
23
+ user: string;
24
+ /** JSON Schema the response must conform to. */
25
+ schema: Record<string, unknown>;
26
+ temperature: number;
27
+ }
28
+ interface TokenUsage {
29
+ inputTokens: number;
30
+ outputTokens: number;
31
+ }
32
+ interface CompleteJSONResponse {
33
+ json: unknown;
34
+ /** Absent when the provider does not report usage (e.g. the Claude CLI). */
35
+ usage?: TokenUsage;
36
+ }
37
+ interface InferenceProvider {
38
+ /** Stable provider id — feeds cache keys. */
39
+ provider(): string;
40
+ /** Model id — feeds cache keys and pricing. */
41
+ modelName(): string;
42
+ completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
43
+ }
44
+ interface ExecResult {
45
+ code: number | null;
46
+ stdout: string;
47
+ stderr: string;
48
+ timedOut: boolean;
49
+ /** Set when the process could not be spawned (e.g. binary not found). */
50
+ spawnError?: string;
51
+ }
52
+ interface ExecOptions {
53
+ cwd?: string;
54
+ timeoutMs?: number;
55
+ env?: Record<string, string>;
56
+ /** Text piped to the child's stdin (stdin is closed after writing). */
57
+ input?: string;
58
+ }
59
+ /** Injectable process-execution seam — subprocess providers take one for tests. */
60
+ type ExecFn = (cmd: string[], opts?: ExecOptions) => Promise<ExecResult>;
61
+
62
+ /**
63
+ * Cost tracking: token usage priced from a small static table, overridable per
64
+ * model by the caller. Unknown models cost 0 (unknown), never a guess — a
65
+ * fabricated price is worse than an absent one when a budget gate depends on it.
66
+ */
67
+
68
+ interface Pricing {
69
+ inputPerMTok: number;
70
+ outputPerMTok: number;
71
+ }
72
+ /**
73
+ * USD per million tokens. Entries are base names; pinned variants
74
+ * (`claude-sonnet-4-5-20250929`) resolve by prefix.
75
+ */
76
+ declare const PRICE_TABLE: Record<string, Pricing>;
77
+ declare function pricingFor(model: string, override?: Pricing): Pricing | undefined;
78
+ declare function costOfUsage(usage: TokenUsage | undefined, pricing: Pricing | undefined): number;
79
+ /** Sum the cost of a set of runs. Cached runs cost nothing — they made no call. */
80
+ declare function costOfRuns(runs: {
81
+ usage?: TokenUsage;
82
+ cached?: boolean;
83
+ }[], pricing: Pricing | undefined): number;
84
+
85
+ interface AnthropicProviderOptions {
86
+ /**
87
+ * Name of the forced tool. Purely cosmetic to the model, but a descriptive
88
+ * name ("record_verdict", "record_proposal") measurably steers output, so
89
+ * consumers may set their own.
90
+ */
91
+ toolName?: string;
92
+ /** Tool description shown to the model. */
93
+ toolDescription?: string;
94
+ maxTokens?: number;
95
+ }
96
+ declare class AnthropicProvider implements InferenceProvider {
97
+ private readonly model;
98
+ private readonly client;
99
+ private readonly toolName;
100
+ private readonly toolDescription;
101
+ private readonly maxTokens;
102
+ constructor(model: string, apiKeyEnv: string, options?: AnthropicProviderOptions);
103
+ provider(): string;
104
+ modelName(): string;
105
+ completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
106
+ }
107
+
108
+ /** Extract a JSON object from content that may carry markdown fences. */
109
+ declare function extractJson(content: string): unknown;
110
+ /**
111
+ * OpenAI strict mode requires `required` to list EVERY property (optionality
112
+ * is expressed as a `null` type union) and rejects keywords outside its
113
+ * subset (minLength, uniqueItems). Transform an all-optional schema into a
114
+ * strict-compatible equivalent; null values are stripped from the response.
115
+ */
116
+ declare function toStrictSchema(schema: Record<string, unknown>): Record<string, unknown>;
117
+ /** Remove null-valued keys (strict-mode "omitted" marker) from a response object. */
118
+ declare function stripNulls(value: unknown): unknown;
119
+ interface OpenAICompatProviderOptions {
120
+ /** `json_schema.name` sent to the server. Cosmetic; steers some models. */
121
+ schemaName?: string;
122
+ }
123
+ declare class OpenAICompatProvider implements InferenceProvider {
124
+ private readonly baseUrl;
125
+ private readonly model;
126
+ private readonly apiKey;
127
+ private supportsJsonSchema;
128
+ private readonly schemaName;
129
+ constructor(baseUrl: string, model: string, apiKeyEnv: string, apiKey?: string | undefined, options?: OpenAICompatProviderOptions);
130
+ provider(): string;
131
+ modelName(): string;
132
+ private chat;
133
+ completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
134
+ private jsonObjectFallback;
135
+ }
136
+
137
+ /**
138
+ * Mock provider for tests and offline development. Responds with scripted
139
+ * results in order, cycling when exhausted. Exported from the public API so
140
+ * downstream consumers can test their own pipelines without a live provider.
141
+ */
142
+
143
+ type MockResponse = {
144
+ json: unknown;
145
+ usage?: TokenUsage;
146
+ } | {
147
+ error: string;
148
+ };
149
+ declare class MockProvider implements InferenceProvider {
150
+ private readonly responses;
151
+ private readonly model;
152
+ private calls;
153
+ /** Every request seen, in order — assert against this in tests. */
154
+ readonly requests: CompleteJSONRequest[];
155
+ constructor(responses: MockResponse[], model?: string);
156
+ provider(): string;
157
+ modelName(): string;
158
+ completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
159
+ }
160
+ /** Convenience: a scripted response shaped like the canonical judge verdict. */
161
+ declare function mockVerdict(match: "pass" | "fail" | "partial", confidence: number, overrides?: Partial<{
162
+ claim: string;
163
+ observed: string;
164
+ reasoning: string;
165
+ }>): {
166
+ json: unknown;
167
+ };
168
+
169
+ type ProviderName = "anthropic" | "openai" | "claude-cli" | "mock";
170
+ interface ProviderSpec {
171
+ provider: ProviderName;
172
+ /** null/undefined selects the per-provider default. */
173
+ model?: string | null;
174
+ /** Env var NAME holding the API key; null/undefined selects the default. */
175
+ apiKeyEnv?: string | null;
176
+ /** openai only. */
177
+ baseUrl?: string;
178
+ /** claude-cli only: the executable to run. */
179
+ command?: string;
180
+ /** claude-cli only: subprocess timeout. */
181
+ timeoutMs?: number;
182
+ /**
183
+ * Pricing override for this model. Not used to construct the provider —
184
+ * carried here so a consumer passes one object to both `makeProvider` and
185
+ * `pricingFor`.
186
+ */
187
+ pricing?: Pricing;
188
+ /** Provider-specific tuning, ignored by the other providers. */
189
+ anthropic?: AnthropicProviderOptions;
190
+ openai?: OpenAICompatProviderOptions;
191
+ /** Test seam for the claude-cli provider. */
192
+ exec?: ExecFn;
193
+ /** Scripted responses for the mock provider; defaults to a single empty object. */
194
+ mockResponses?: MockResponse[];
195
+ }
196
+ declare const DEFAULT_MODELS: Record<ProviderName, string>;
197
+ declare const DEFAULT_OPENAI_BASE_URL = "https://api.openai.com/v1";
198
+ /**
199
+ * Resolve the provider name and model WITHOUT constructing the provider —
200
+ * cache keys and pricing need the identity, but construction may require an
201
+ * API key that a fully-cached run never uses.
202
+ */
203
+ declare function resolveProviderIdentity(spec: ProviderSpec): {
204
+ provider: string;
205
+ model: string;
206
+ };
207
+ declare function makeProvider(spec: ProviderSpec): InferenceProvider;
208
+
209
+ declare class ClaudeCliProvider implements InferenceProvider {
210
+ private readonly model;
211
+ private readonly command;
212
+ private readonly exec;
213
+ private readonly timeoutMs;
214
+ constructor(model: string, command?: string, exec?: ExecFn, timeoutMs?: number);
215
+ provider(): string;
216
+ modelName(): string;
217
+ completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
218
+ }
219
+
220
+ declare const realExec: ExecFn;
221
+
222
+ /** One attempt at a schema-constrained completion. */
223
+ interface InferenceRun<T = unknown> {
224
+ /** Absent when the run errored (invalid JSON after retry, API failure). */
225
+ result?: T;
226
+ error?: string;
227
+ provider: string;
228
+ model: string;
229
+ cached: boolean;
230
+ usage?: TokenUsage;
231
+ durationMs: number;
232
+ }
233
+ interface CompleteValidatedOptions {
234
+ provider: InferenceProvider;
235
+ system: string;
236
+ user: string;
237
+ schema: Record<string, unknown>;
238
+ temperature?: number;
239
+ /**
240
+ * Attempts before recording an error. Default 2 (one initial call plus one
241
+ * retry) — matches the behavior all three source projects shipped.
242
+ */
243
+ attempts?: number;
244
+ /**
245
+ * Pre-compiled validator. Compiling Ajv per call is wasteful in an ensemble
246
+ * loop, so `runEnsemble` compiles once and passes it down.
247
+ */
248
+ validate?: ValidateFunction;
249
+ }
250
+ /** Compile once per schema object identity — Ajv compilation is not cheap. */
251
+ declare function validatorFor(schema: Record<string, unknown>): ValidateFunction;
252
+ declare function completeValidatedJSON<T = unknown>(options: CompleteValidatedOptions): Promise<InferenceRun<T>>;
253
+
254
+ declare function sha256(text: string): string;
255
+ /**
256
+ * Hash an ordered list of key parts into a cache key. Long parts (page bodies,
257
+ * rendered traces) should be pre-hashed with `sha256` by the caller so the
258
+ * joined string stays small.
259
+ */
260
+ declare function buildCacheKey(parts: string[]): string;
261
+ declare class JsonCache<T> {
262
+ private readonly dir;
263
+ private readonly enabled;
264
+ /** Prefix for the one-time write-failure warning. */
265
+ private readonly label;
266
+ /** Cache-write failures warn once per process, not once per entry. */
267
+ private warned;
268
+ constructor(dir: string, enabled?: boolean,
269
+ /** Prefix for the one-time write-failure warning. */
270
+ label?: string);
271
+ get(key: string): T | undefined;
272
+ set(key: string, value: T): void;
273
+ }
274
+
275
+ declare const VERDICT_SCHEMA: Record<string, unknown>;
276
+ type Match = "pass" | "fail" | "partial";
277
+ /** Confidence-zone routing for LLM-judged evals. */
278
+ type Zone = "auto-pass" | "auto-fail" | "human-review";
279
+ interface JudgeVerdict {
280
+ /** The specific assertion under evaluation. */
281
+ claim: string;
282
+ /** What the judge actually observed. */
283
+ observed: string;
284
+ match: Match;
285
+ /** 0.0–1.0 self-reported confidence. */
286
+ confidence: number;
287
+ reasoning: string;
288
+ }
289
+ /** One run within an ensemble. */
290
+ interface JudgeRun {
291
+ /** Absent when the run errored (invalid JSON after retry, API failure). */
292
+ verdict?: JudgeVerdict;
293
+ error?: string;
294
+ provider: string;
295
+ model: string;
296
+ cached: boolean;
297
+ usage?: TokenUsage;
298
+ durationMs: number;
299
+ }
300
+ /** Aggregated outcome of an ensemble of judge runs for one subject. */
301
+ interface ConsensusResult {
302
+ runs: JudgeRun[];
303
+ votes: {
304
+ pass: number;
305
+ fail: number;
306
+ partial: number;
307
+ error: number;
308
+ };
309
+ /** Majority verdict; `partial` counts as fail for the binary outcome. */
310
+ verdict: Match;
311
+ /** Fraction of non-errored runs agreeing with the majority verdict. */
312
+ agreement: number;
313
+ /** Mean confidence across non-errored runs. */
314
+ meanConfidence: number;
315
+ zone: Zone;
316
+ }
317
+
318
+ /**
319
+ * Ensemble consensus math. `partial` counts as fail for the binary outcome
320
+ * but stays visible in the vote counts. Errored runs count against consensus —
321
+ * they can only push a result toward human review, never toward a silent pass.
322
+ */
323
+
324
+ declare function computeConsensus(runs: JudgeRun[]): Omit<ConsensusResult, "zone">;
325
+
326
+ /**
327
+ * Confidence-zone routing: only unanimous, high-confidence ensembles
328
+ * auto-resolve; everything else goes to a human.
329
+ */
330
+
331
+ interface ZoneThresholds {
332
+ autoPass: number;
333
+ autoFail: number;
334
+ }
335
+ declare const DEFAULT_ZONES: ZoneThresholds;
336
+ declare function zoneFor(consensus: Omit<ConsensusResult, "zone">, thresholds?: ZoneThresholds): Zone;
337
+
338
+ interface EnsembleOptions {
339
+ provider: InferenceProvider;
340
+ system: string;
341
+ user: string;
342
+ /** Ensemble size; default 3. */
343
+ runs?: number;
344
+ /** Default 0. Nonzero adds noise to verdicts and warns once. */
345
+ temperature?: number;
346
+ /**
347
+ * Verdict schema. Defaults to the canonical one; pass your own to keep
348
+ * domain-specific field descriptions (they measurably steer the model).
349
+ * Must still produce objects matching `JudgeVerdict`.
350
+ */
351
+ schema?: Record<string, unknown>;
352
+ /** Optional result cache. Requires `cacheKey`. */
353
+ cache?: JsonCache<JudgeRun[]>;
354
+ /** Content-addressed key; build it with `buildCacheKey`. */
355
+ cacheKey?: string;
356
+ /** Prefix for warnings, e.g. your tool's name. */
357
+ label?: string;
358
+ }
359
+ /**
360
+ * Run the ensemble and return every run. Cached ensembles replay identically,
361
+ * with each run flagged `cached: true` so cost accounting skips them.
362
+ */
363
+ declare function runEnsemble(options: EnsembleOptions): Promise<JudgeRun[]>;
364
+ /** Run the ensemble and aggregate it into a zoned consensus. */
365
+ declare function judge(options: EnsembleOptions & {
366
+ zones?: ZoneThresholds;
367
+ }): Promise<ConsensusResult>;
368
+ /** Test seam: reset the once-per-process temperature warning. */
369
+ declare function resetTemperatureWarning(): void;
370
+
371
+ export { AnthropicProvider, type AnthropicProviderOptions, ClaudeCliProvider, type CompleteJSONRequest, type CompleteJSONResponse, type CompleteValidatedOptions, type ConsensusResult, DEFAULT_MODELS, DEFAULT_OPENAI_BASE_URL, DEFAULT_ZONES, type EnsembleOptions, type ExecFn, type ExecOptions, type ExecResult, InferenceError, type InferenceProvider, type InferenceRun, JsonCache, type JudgeRun, type JudgeVerdict, type Match, MockProvider, type MockResponse, OpenAICompatProvider, type OpenAICompatProviderOptions, PRICE_TABLE, type Pricing, type ProviderName, type ProviderSpec, type TokenUsage, VERDICT_SCHEMA, type Zone, type ZoneThresholds, buildCacheKey, completeValidatedJSON, computeConsensus, costOfRuns, costOfUsage, extractJson, judge, makeProvider, mockVerdict, pricingFor, realExec, resetTemperatureWarning, resolveProviderIdentity, runEnsemble, sha256, stripNulls, toStrictSchema, validatorFor, zoneFor };