@hawkeyexl/inference 0.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +211 -0
- package/dist/index.d.ts +371 -0
- package/dist/index.js +722 -0
- package/dist/index.js.map +1 -0
- package/package.json +78 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 inference contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
# @hawkeyexl/inference
|
|
2
|
+
|
|
3
|
+
Shared LLM inference layer for the docs-as-tests toolchain: schema-constrained completion across
|
|
4
|
+
Anthropic, OpenAI-compatible, and Claude CLI providers, with result caching, cost accounting, and
|
|
5
|
+
an LLM-as-judge ensemble on top.
|
|
6
|
+
|
|
7
|
+
Extracted from three projects that had each grown their own copy —
|
|
8
|
+
[docevals](https://github.com/hawkeyexl/docevals), [dockg](https://github.com/hawkeyexl/dockg), and
|
|
9
|
+
[agentevals](https://github.com/hawkeyexl/agentevals) — so a provider fix lands once instead of
|
|
10
|
+
three times.
|
|
11
|
+
|
|
12
|
+
## Install
|
|
13
|
+
|
|
14
|
+
```bash
|
|
15
|
+
npm install @hawkeyexl/inference
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
Requires Node 24+. ESM only.
|
|
19
|
+
|
|
20
|
+
## What it does
|
|
21
|
+
|
|
22
|
+
Every consumer of this library wants the same narrow thing: **send a system prompt, a user prompt,
|
|
23
|
+
and a JSON Schema; get back JSON that validates against that schema, or a recorded error.** No
|
|
24
|
+
streaming, no multi-turn, no tool loops.
|
|
25
|
+
|
|
26
|
+
Two layers, one entry point:
|
|
27
|
+
|
|
28
|
+
- **Completion** — the provider contract, four providers, a content-addressed cache, a price table,
|
|
29
|
+
and a validate-and-retry wrapper. This is all dockg-style structured extraction needs.
|
|
30
|
+
- **Judge** — the canonical verdict schema, an N-run ensemble, consensus math, and confidence-zone
|
|
31
|
+
routing. Built on the completion layer; ignore it if you do not need it.
|
|
32
|
+
|
|
33
|
+
## Quick start
|
|
34
|
+
|
|
35
|
+
### Schema-constrained completion
|
|
36
|
+
|
|
37
|
+
```ts
|
|
38
|
+
import { completeValidatedJSON, makeProvider } from "@hawkeyexl/inference";
|
|
39
|
+
|
|
40
|
+
const provider = makeProvider({ provider: "anthropic", model: "claude-sonnet-4-5" });
|
|
41
|
+
|
|
42
|
+
const run = await completeValidatedJSON<{ summary: string }>({
|
|
43
|
+
provider,
|
|
44
|
+
system: "You summarize documentation pages.",
|
|
45
|
+
user: pageBody,
|
|
46
|
+
schema: {
|
|
47
|
+
type: "object",
|
|
48
|
+
required: ["summary"],
|
|
49
|
+
properties: { summary: { type: "string" } },
|
|
50
|
+
additionalProperties: false,
|
|
51
|
+
},
|
|
52
|
+
});
|
|
53
|
+
|
|
54
|
+
if (run.error) console.error(run.error);
|
|
55
|
+
else console.log(run.result.summary, run.usage);
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
`completeValidatedJSON` never throws on a model failure and never coerces a bad response. It
|
|
59
|
+
retries once, then returns a run with `error` set and `result` absent.
|
|
60
|
+
|
|
61
|
+
### LLM-as-judge
|
|
62
|
+
|
|
63
|
+
```ts
|
|
64
|
+
import { judge, makeProvider } from "@hawkeyexl/inference";
|
|
65
|
+
|
|
66
|
+
const consensus = await judge({
|
|
67
|
+
provider: makeProvider({ provider: "claude-cli" }),
|
|
68
|
+
system: "You evaluate whether a page satisfies an assertion.",
|
|
69
|
+
user: "# Assertion\nThe page documents authentication.\n\n# Page\n...",
|
|
70
|
+
runs: 3,
|
|
71
|
+
});
|
|
72
|
+
|
|
73
|
+
consensus.verdict; // "pass" | "fail" — partial counts as fail
|
|
74
|
+
consensus.zone; // "auto-pass" | "auto-fail" | "human-review"
|
|
75
|
+
consensus.agreement; // 0..1 across non-errored runs
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Only a **unanimous, high-confidence** ensemble auto-resolves. Anything split, low-confidence, or
|
|
79
|
+
containing an errored run routes to `human-review` — an errored run can never produce a silent pass.
|
|
80
|
+
|
|
81
|
+
### Caching
|
|
82
|
+
|
|
83
|
+
Key composition stays with you, because each consumer has a different notion of what should
|
|
84
|
+
invalidate an entry (page body, prompt version, ensemble size, requested fields):
|
|
85
|
+
|
|
86
|
+
```ts
|
|
87
|
+
import { JsonCache, buildCacheKey, runEnsemble, sha256 } from "@hawkeyexl/inference";
|
|
88
|
+
|
|
89
|
+
const cache = new JsonCache(".mytool/cache", true, "mytool");
|
|
90
|
+
const cacheKey = buildCacheKey([
|
|
91
|
+
provider.provider(),
|
|
92
|
+
provider.modelName(),
|
|
93
|
+
`v${MY_PROMPT_VERSION}`,
|
|
94
|
+
`r${runs}`,
|
|
95
|
+
sha256(pageBody), // pre-hash long parts
|
|
96
|
+
]);
|
|
97
|
+
|
|
98
|
+
const judgeRuns = await runEnsemble({ provider, system, user, runs, cache, cacheKey });
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
Cached runs come back flagged `cached: true`, so `costOfRuns` correctly charges nothing for a
|
|
102
|
+
replay. Cache write failures warn once and continue — a read-only workspace must not abort a run
|
|
103
|
+
whose inference already succeeded and was already paid for.
|
|
104
|
+
|
|
105
|
+
### Cost
|
|
106
|
+
|
|
107
|
+
```ts
|
|
108
|
+
import { costOfRuns, pricingFor } from "@hawkeyexl/inference";
|
|
109
|
+
|
|
110
|
+
const pricing = pricingFor(provider.modelName(), configOverride);
|
|
111
|
+
const usd = costOfRuns(judgeRuns, pricing);
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
An unknown model returns `undefined` pricing and costs `0` — **unknown, never a guess**. A
|
|
115
|
+
fabricated price is worse than an absent one when a budget gate depends on it.
|
|
116
|
+
|
|
117
|
+
## Providers
|
|
118
|
+
|
|
119
|
+
Constructed through `makeProvider(spec)`. The spec is a flat, library-owned shape — map your own
|
|
120
|
+
config into it rather than passing your config object (see
|
|
121
|
+
[ADR 01000](adrs/01000-library-owned-provider-spec.md)).
|
|
122
|
+
|
|
123
|
+
| `provider` | Structured output via | Key | Reports usage |
|
|
124
|
+
|---|---|---|---|
|
|
125
|
+
| `anthropic` | forced tool call | `ANTHROPIC_API_KEY` | yes |
|
|
126
|
+
| `openai` | strict `json_schema`, falls back to `json_object` | `OPENAI_API_KEY` | yes |
|
|
127
|
+
| `claude-cli` | schema in the prompt, `--output-format json` | local `claude` auth | no |
|
|
128
|
+
| `mock` | scripted responses | — | synthetic |
|
|
129
|
+
|
|
130
|
+
```ts
|
|
131
|
+
interface ProviderSpec {
|
|
132
|
+
provider: "anthropic" | "openai" | "claude-cli" | "mock";
|
|
133
|
+
model?: string | null; // null/undefined -> per-provider default
|
|
134
|
+
apiKeyEnv?: string | null; // default ANTHROPIC_API_KEY / OPENAI_API_KEY
|
|
135
|
+
baseUrl?: string; // openai only, default https://api.openai.com/v1
|
|
136
|
+
command?: string; // claude-cli only, default "claude"
|
|
137
|
+
timeoutMs?: number; // claude-cli only, default 180000
|
|
138
|
+
pricing?: Pricing; // override the built-in table
|
|
139
|
+
anthropic?: AnthropicProviderOptions; // e.g. toolName, maxTokens
|
|
140
|
+
openai?: OpenAICompatProviderOptions; // e.g. schemaName
|
|
141
|
+
exec?: ExecFn; // test seam for claude-cli
|
|
142
|
+
mockResponses?: MockResponse[];
|
|
143
|
+
}
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
`resolveProviderIdentity(spec)` returns `{ provider, model }` **without constructing anything** —
|
|
147
|
+
cache keys and pricing need the identity, but a fully-cached run should not require an API key.
|
|
148
|
+
|
|
149
|
+
Notes on the non-obvious bits:
|
|
150
|
+
|
|
151
|
+
- **`openai`** targets any `/chat/completions` server (OpenAI, Azure, Ollama, Groq, Together). It
|
|
152
|
+
prefers strict `json_schema`, and `toStrictSchema` rewrites your schema into the strict subset
|
|
153
|
+
(every property in `required`, optionality as a `null` type union, unsupported keywords dropped);
|
|
154
|
+
nulls are stripped back out of the response. If the server rejects `response_format`, it
|
|
155
|
+
permanently falls back to `json_object` with the schema in the prompt. Keyless local servers are
|
|
156
|
+
allowed — only `api.openai.com` requires a key.
|
|
157
|
+
- **`claude-cli`** uses your local Claude CLI auth, so no API key. The prompt goes over **stdin**,
|
|
158
|
+
never argv: user content routinely exceeds the ~32K Windows command-line limit.
|
|
159
|
+
|
|
160
|
+
## Testing against this library
|
|
161
|
+
|
|
162
|
+
`MockProvider` is exported for exactly this. No network required:
|
|
163
|
+
|
|
164
|
+
```ts
|
|
165
|
+
import { MockProvider, mockVerdict, runEnsemble } from "@hawkeyexl/inference";
|
|
166
|
+
|
|
167
|
+
const provider = new MockProvider([mockVerdict("pass", 0.95)]); // cycles when exhausted
|
|
168
|
+
const runs = await runEnsemble({ provider, system, user, runs: 3 });
|
|
169
|
+
provider.requests; // every request seen, in order
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
Script an error with `{ error: "429 rate limited" }` to exercise your failure paths.
|
|
173
|
+
|
|
174
|
+
## API
|
|
175
|
+
|
|
176
|
+
Everything exports from the package root.
|
|
177
|
+
|
|
178
|
+
**Providers** — `makeProvider`, `resolveProviderIdentity`, `DEFAULT_MODELS`,
|
|
179
|
+
`DEFAULT_OPENAI_BASE_URL`, `AnthropicProvider`, `OpenAICompatProvider`, `ClaudeCliProvider`,
|
|
180
|
+
`MockProvider`, `mockVerdict`, `extractJson`, `toStrictSchema`, `stripNulls`, `realExec`
|
|
181
|
+
|
|
182
|
+
**Completion** — `completeValidatedJSON`, `validatorFor`
|
|
183
|
+
|
|
184
|
+
**Cache** — `JsonCache`, `buildCacheKey`, `sha256`
|
|
185
|
+
|
|
186
|
+
**Cost** — `pricingFor`, `costOfUsage`, `costOfRuns`, `PRICE_TABLE`
|
|
187
|
+
|
|
188
|
+
**Judge** — `judge`, `runEnsemble`, `computeConsensus`, `zoneFor`, `VERDICT_SCHEMA`,
|
|
189
|
+
`DEFAULT_ZONES`
|
|
190
|
+
|
|
191
|
+
**Errors** — `InferenceError` (operational failures: missing key, unknown provider)
|
|
192
|
+
|
|
193
|
+
Types: `InferenceProvider`, `ProviderSpec`, `ProviderName`, `CompleteJSONRequest`,
|
|
194
|
+
`CompleteJSONResponse`, `InferenceRun`, `TokenUsage`, `Pricing`, `JudgeRun`, `JudgeVerdict`,
|
|
195
|
+
`ConsensusResult`, `Match`, `Zone`, `ZoneThresholds`, `EnsembleOptions`, `ExecFn`, `ExecResult`,
|
|
196
|
+
`ExecOptions`, `MockResponse`.
|
|
197
|
+
|
|
198
|
+
## Design decisions
|
|
199
|
+
|
|
200
|
+
Recorded as ADRs in [adrs/](adrs):
|
|
201
|
+
|
|
202
|
+
- [01000](adrs/01000-library-owned-provider-spec.md) — a library-owned `ProviderSpec`, not consumer
|
|
203
|
+
config objects
|
|
204
|
+
- [01001](adrs/01001-single-entry-point-and-canonical-verdict-schema.md) — one entry point; a
|
|
205
|
+
canonical verdict schema with a per-consumer override seam
|
|
206
|
+
- [01002](adrs/01002-best-of-merge-of-three-forks.md) — which fork won for each merged file, so the
|
|
207
|
+
losing variants are not reintroduced
|
|
208
|
+
|
|
209
|
+
## License
|
|
210
|
+
|
|
211
|
+
MIT
|
package/dist/index.d.ts
ADDED
|
@@ -0,0 +1,371 @@
|
|
|
1
|
+
import { ValidateFunction } from 'ajv';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Shared error type. Consumers catch this to distinguish an inference-layer
|
|
5
|
+
* operational failure (missing API key, unknown provider) from their own
|
|
6
|
+
* domain errors, and typically map it to their own exit code.
|
|
7
|
+
*/
|
|
8
|
+
declare class InferenceError extends Error {
|
|
9
|
+
constructor(message: string);
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* The provider contract. A provider turns a (system, user, schema) request
|
|
14
|
+
* into schema-conforming JSON. `provider()` and `modelName()` feed cache keys
|
|
15
|
+
* and pricing lookups, so two providers/models never share a cached result.
|
|
16
|
+
*
|
|
17
|
+
* This is deliberately the narrowest useful surface: no streaming, no
|
|
18
|
+
* multi-turn, no tool loops. Everything downstream of it — judging,
|
|
19
|
+
* extraction, classification — is schema-constrained single-shot completion.
|
|
20
|
+
*/
|
|
21
|
+
interface CompleteJSONRequest {
|
|
22
|
+
system: string;
|
|
23
|
+
user: string;
|
|
24
|
+
/** JSON Schema the response must conform to. */
|
|
25
|
+
schema: Record<string, unknown>;
|
|
26
|
+
temperature: number;
|
|
27
|
+
}
|
|
28
|
+
interface TokenUsage {
|
|
29
|
+
inputTokens: number;
|
|
30
|
+
outputTokens: number;
|
|
31
|
+
}
|
|
32
|
+
interface CompleteJSONResponse {
|
|
33
|
+
json: unknown;
|
|
34
|
+
/** Absent when the provider does not report usage (e.g. the Claude CLI). */
|
|
35
|
+
usage?: TokenUsage;
|
|
36
|
+
}
|
|
37
|
+
interface InferenceProvider {
|
|
38
|
+
/** Stable provider id — feeds cache keys. */
|
|
39
|
+
provider(): string;
|
|
40
|
+
/** Model id — feeds cache keys and pricing. */
|
|
41
|
+
modelName(): string;
|
|
42
|
+
completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
|
|
43
|
+
}
|
|
44
|
+
interface ExecResult {
|
|
45
|
+
code: number | null;
|
|
46
|
+
stdout: string;
|
|
47
|
+
stderr: string;
|
|
48
|
+
timedOut: boolean;
|
|
49
|
+
/** Set when the process could not be spawned (e.g. binary not found). */
|
|
50
|
+
spawnError?: string;
|
|
51
|
+
}
|
|
52
|
+
interface ExecOptions {
|
|
53
|
+
cwd?: string;
|
|
54
|
+
timeoutMs?: number;
|
|
55
|
+
env?: Record<string, string>;
|
|
56
|
+
/** Text piped to the child's stdin (stdin is closed after writing). */
|
|
57
|
+
input?: string;
|
|
58
|
+
}
|
|
59
|
+
/** Injectable process-execution seam — subprocess providers take one for tests. */
|
|
60
|
+
type ExecFn = (cmd: string[], opts?: ExecOptions) => Promise<ExecResult>;
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Cost tracking: token usage priced from a small static table, overridable per
|
|
64
|
+
* model by the caller. Unknown models cost 0 (unknown), never a guess — a
|
|
65
|
+
* fabricated price is worse than an absent one when a budget gate depends on it.
|
|
66
|
+
*/
|
|
67
|
+
|
|
68
|
+
interface Pricing {
|
|
69
|
+
inputPerMTok: number;
|
|
70
|
+
outputPerMTok: number;
|
|
71
|
+
}
|
|
72
|
+
/**
|
|
73
|
+
* USD per million tokens. Entries are base names; pinned variants
|
|
74
|
+
* (`claude-sonnet-4-5-20250929`) resolve by prefix.
|
|
75
|
+
*/
|
|
76
|
+
declare const PRICE_TABLE: Record<string, Pricing>;
|
|
77
|
+
declare function pricingFor(model: string, override?: Pricing): Pricing | undefined;
|
|
78
|
+
declare function costOfUsage(usage: TokenUsage | undefined, pricing: Pricing | undefined): number;
|
|
79
|
+
/** Sum the cost of a set of runs. Cached runs cost nothing — they made no call. */
|
|
80
|
+
declare function costOfRuns(runs: {
|
|
81
|
+
usage?: TokenUsage;
|
|
82
|
+
cached?: boolean;
|
|
83
|
+
}[], pricing: Pricing | undefined): number;
|
|
84
|
+
|
|
85
|
+
interface AnthropicProviderOptions {
|
|
86
|
+
/**
|
|
87
|
+
* Name of the forced tool. Purely cosmetic to the model, but a descriptive
|
|
88
|
+
* name ("record_verdict", "record_proposal") measurably steers output, so
|
|
89
|
+
* consumers may set their own.
|
|
90
|
+
*/
|
|
91
|
+
toolName?: string;
|
|
92
|
+
/** Tool description shown to the model. */
|
|
93
|
+
toolDescription?: string;
|
|
94
|
+
maxTokens?: number;
|
|
95
|
+
}
|
|
96
|
+
declare class AnthropicProvider implements InferenceProvider {
|
|
97
|
+
private readonly model;
|
|
98
|
+
private readonly client;
|
|
99
|
+
private readonly toolName;
|
|
100
|
+
private readonly toolDescription;
|
|
101
|
+
private readonly maxTokens;
|
|
102
|
+
constructor(model: string, apiKeyEnv: string, options?: AnthropicProviderOptions);
|
|
103
|
+
provider(): string;
|
|
104
|
+
modelName(): string;
|
|
105
|
+
completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/** Extract a JSON object from content that may carry markdown fences. */
|
|
109
|
+
declare function extractJson(content: string): unknown;
|
|
110
|
+
/**
|
|
111
|
+
* OpenAI strict mode requires `required` to list EVERY property (optionality
|
|
112
|
+
* is expressed as a `null` type union) and rejects keywords outside its
|
|
113
|
+
* subset (minLength, uniqueItems). Transform an all-optional schema into a
|
|
114
|
+
* strict-compatible equivalent; null values are stripped from the response.
|
|
115
|
+
*/
|
|
116
|
+
declare function toStrictSchema(schema: Record<string, unknown>): Record<string, unknown>;
|
|
117
|
+
/** Remove null-valued keys (strict-mode "omitted" marker) from a response object. */
|
|
118
|
+
declare function stripNulls(value: unknown): unknown;
|
|
119
|
+
interface OpenAICompatProviderOptions {
|
|
120
|
+
/** `json_schema.name` sent to the server. Cosmetic; steers some models. */
|
|
121
|
+
schemaName?: string;
|
|
122
|
+
}
|
|
123
|
+
declare class OpenAICompatProvider implements InferenceProvider {
|
|
124
|
+
private readonly baseUrl;
|
|
125
|
+
private readonly model;
|
|
126
|
+
private readonly apiKey;
|
|
127
|
+
private supportsJsonSchema;
|
|
128
|
+
private readonly schemaName;
|
|
129
|
+
constructor(baseUrl: string, model: string, apiKeyEnv: string, apiKey?: string | undefined, options?: OpenAICompatProviderOptions);
|
|
130
|
+
provider(): string;
|
|
131
|
+
modelName(): string;
|
|
132
|
+
private chat;
|
|
133
|
+
completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
|
|
134
|
+
private jsonObjectFallback;
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
/**
|
|
138
|
+
* Mock provider for tests and offline development. Responds with scripted
|
|
139
|
+
* results in order, cycling when exhausted. Exported from the public API so
|
|
140
|
+
* downstream consumers can test their own pipelines without a live provider.
|
|
141
|
+
*/
|
|
142
|
+
|
|
143
|
+
type MockResponse = {
|
|
144
|
+
json: unknown;
|
|
145
|
+
usage?: TokenUsage;
|
|
146
|
+
} | {
|
|
147
|
+
error: string;
|
|
148
|
+
};
|
|
149
|
+
declare class MockProvider implements InferenceProvider {
|
|
150
|
+
private readonly responses;
|
|
151
|
+
private readonly model;
|
|
152
|
+
private calls;
|
|
153
|
+
/** Every request seen, in order — assert against this in tests. */
|
|
154
|
+
readonly requests: CompleteJSONRequest[];
|
|
155
|
+
constructor(responses: MockResponse[], model?: string);
|
|
156
|
+
provider(): string;
|
|
157
|
+
modelName(): string;
|
|
158
|
+
completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
|
|
159
|
+
}
|
|
160
|
+
/** Convenience: a scripted response shaped like the canonical judge verdict. */
|
|
161
|
+
declare function mockVerdict(match: "pass" | "fail" | "partial", confidence: number, overrides?: Partial<{
|
|
162
|
+
claim: string;
|
|
163
|
+
observed: string;
|
|
164
|
+
reasoning: string;
|
|
165
|
+
}>): {
|
|
166
|
+
json: unknown;
|
|
167
|
+
};
|
|
168
|
+
|
|
169
|
+
type ProviderName = "anthropic" | "openai" | "claude-cli" | "mock";
|
|
170
|
+
interface ProviderSpec {
|
|
171
|
+
provider: ProviderName;
|
|
172
|
+
/** null/undefined selects the per-provider default. */
|
|
173
|
+
model?: string | null;
|
|
174
|
+
/** Env var NAME holding the API key; null/undefined selects the default. */
|
|
175
|
+
apiKeyEnv?: string | null;
|
|
176
|
+
/** openai only. */
|
|
177
|
+
baseUrl?: string;
|
|
178
|
+
/** claude-cli only: the executable to run. */
|
|
179
|
+
command?: string;
|
|
180
|
+
/** claude-cli only: subprocess timeout. */
|
|
181
|
+
timeoutMs?: number;
|
|
182
|
+
/**
|
|
183
|
+
* Pricing override for this model. Not used to construct the provider —
|
|
184
|
+
* carried here so a consumer passes one object to both `makeProvider` and
|
|
185
|
+
* `pricingFor`.
|
|
186
|
+
*/
|
|
187
|
+
pricing?: Pricing;
|
|
188
|
+
/** Provider-specific tuning, ignored by the other providers. */
|
|
189
|
+
anthropic?: AnthropicProviderOptions;
|
|
190
|
+
openai?: OpenAICompatProviderOptions;
|
|
191
|
+
/** Test seam for the claude-cli provider. */
|
|
192
|
+
exec?: ExecFn;
|
|
193
|
+
/** Scripted responses for the mock provider; defaults to a single empty object. */
|
|
194
|
+
mockResponses?: MockResponse[];
|
|
195
|
+
}
|
|
196
|
+
declare const DEFAULT_MODELS: Record<ProviderName, string>;
|
|
197
|
+
declare const DEFAULT_OPENAI_BASE_URL = "https://api.openai.com/v1";
|
|
198
|
+
/**
|
|
199
|
+
* Resolve the provider name and model WITHOUT constructing the provider —
|
|
200
|
+
* cache keys and pricing need the identity, but construction may require an
|
|
201
|
+
* API key that a fully-cached run never uses.
|
|
202
|
+
*/
|
|
203
|
+
declare function resolveProviderIdentity(spec: ProviderSpec): {
|
|
204
|
+
provider: string;
|
|
205
|
+
model: string;
|
|
206
|
+
};
|
|
207
|
+
declare function makeProvider(spec: ProviderSpec): InferenceProvider;
|
|
208
|
+
|
|
209
|
+
declare class ClaudeCliProvider implements InferenceProvider {
|
|
210
|
+
private readonly model;
|
|
211
|
+
private readonly command;
|
|
212
|
+
private readonly exec;
|
|
213
|
+
private readonly timeoutMs;
|
|
214
|
+
constructor(model: string, command?: string, exec?: ExecFn, timeoutMs?: number);
|
|
215
|
+
provider(): string;
|
|
216
|
+
modelName(): string;
|
|
217
|
+
completeJSON(req: CompleteJSONRequest): Promise<CompleteJSONResponse>;
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
declare const realExec: ExecFn;
|
|
221
|
+
|
|
222
|
+
/** One attempt at a schema-constrained completion. */
|
|
223
|
+
interface InferenceRun<T = unknown> {
|
|
224
|
+
/** Absent when the run errored (invalid JSON after retry, API failure). */
|
|
225
|
+
result?: T;
|
|
226
|
+
error?: string;
|
|
227
|
+
provider: string;
|
|
228
|
+
model: string;
|
|
229
|
+
cached: boolean;
|
|
230
|
+
usage?: TokenUsage;
|
|
231
|
+
durationMs: number;
|
|
232
|
+
}
|
|
233
|
+
interface CompleteValidatedOptions {
|
|
234
|
+
provider: InferenceProvider;
|
|
235
|
+
system: string;
|
|
236
|
+
user: string;
|
|
237
|
+
schema: Record<string, unknown>;
|
|
238
|
+
temperature?: number;
|
|
239
|
+
/**
|
|
240
|
+
* Attempts before recording an error. Default 2 (one initial call plus one
|
|
241
|
+
* retry) — matches the behavior all three source projects shipped.
|
|
242
|
+
*/
|
|
243
|
+
attempts?: number;
|
|
244
|
+
/**
|
|
245
|
+
* Pre-compiled validator. Compiling Ajv per call is wasteful in an ensemble
|
|
246
|
+
* loop, so `runEnsemble` compiles once and passes it down.
|
|
247
|
+
*/
|
|
248
|
+
validate?: ValidateFunction;
|
|
249
|
+
}
|
|
250
|
+
/** Compile once per schema object identity — Ajv compilation is not cheap. */
|
|
251
|
+
declare function validatorFor(schema: Record<string, unknown>): ValidateFunction;
|
|
252
|
+
declare function completeValidatedJSON<T = unknown>(options: CompleteValidatedOptions): Promise<InferenceRun<T>>;
|
|
253
|
+
|
|
254
|
+
declare function sha256(text: string): string;
|
|
255
|
+
/**
|
|
256
|
+
* Hash an ordered list of key parts into a cache key. Long parts (page bodies,
|
|
257
|
+
* rendered traces) should be pre-hashed with `sha256` by the caller so the
|
|
258
|
+
* joined string stays small.
|
|
259
|
+
*/
|
|
260
|
+
declare function buildCacheKey(parts: string[]): string;
|
|
261
|
+
declare class JsonCache<T> {
|
|
262
|
+
private readonly dir;
|
|
263
|
+
private readonly enabled;
|
|
264
|
+
/** Prefix for the one-time write-failure warning. */
|
|
265
|
+
private readonly label;
|
|
266
|
+
/** Cache-write failures warn once per process, not once per entry. */
|
|
267
|
+
private warned;
|
|
268
|
+
constructor(dir: string, enabled?: boolean,
|
|
269
|
+
/** Prefix for the one-time write-failure warning. */
|
|
270
|
+
label?: string);
|
|
271
|
+
get(key: string): T | undefined;
|
|
272
|
+
set(key: string, value: T): void;
|
|
273
|
+
}
|
|
274
|
+
|
|
275
|
+
declare const VERDICT_SCHEMA: Record<string, unknown>;
|
|
276
|
+
type Match = "pass" | "fail" | "partial";
|
|
277
|
+
/** Confidence-zone routing for LLM-judged evals. */
|
|
278
|
+
type Zone = "auto-pass" | "auto-fail" | "human-review";
|
|
279
|
+
interface JudgeVerdict {
|
|
280
|
+
/** The specific assertion under evaluation. */
|
|
281
|
+
claim: string;
|
|
282
|
+
/** What the judge actually observed. */
|
|
283
|
+
observed: string;
|
|
284
|
+
match: Match;
|
|
285
|
+
/** 0.0–1.0 self-reported confidence. */
|
|
286
|
+
confidence: number;
|
|
287
|
+
reasoning: string;
|
|
288
|
+
}
|
|
289
|
+
/** One run within an ensemble. */
|
|
290
|
+
interface JudgeRun {
|
|
291
|
+
/** Absent when the run errored (invalid JSON after retry, API failure). */
|
|
292
|
+
verdict?: JudgeVerdict;
|
|
293
|
+
error?: string;
|
|
294
|
+
provider: string;
|
|
295
|
+
model: string;
|
|
296
|
+
cached: boolean;
|
|
297
|
+
usage?: TokenUsage;
|
|
298
|
+
durationMs: number;
|
|
299
|
+
}
|
|
300
|
+
/** Aggregated outcome of an ensemble of judge runs for one subject. */
|
|
301
|
+
interface ConsensusResult {
|
|
302
|
+
runs: JudgeRun[];
|
|
303
|
+
votes: {
|
|
304
|
+
pass: number;
|
|
305
|
+
fail: number;
|
|
306
|
+
partial: number;
|
|
307
|
+
error: number;
|
|
308
|
+
};
|
|
309
|
+
/** Majority verdict; `partial` counts as fail for the binary outcome. */
|
|
310
|
+
verdict: Match;
|
|
311
|
+
/** Fraction of non-errored runs agreeing with the majority verdict. */
|
|
312
|
+
agreement: number;
|
|
313
|
+
/** Mean confidence across non-errored runs. */
|
|
314
|
+
meanConfidence: number;
|
|
315
|
+
zone: Zone;
|
|
316
|
+
}
|
|
317
|
+
|
|
318
|
+
/**
|
|
319
|
+
* Ensemble consensus math. `partial` counts as fail for the binary outcome
|
|
320
|
+
* but stays visible in the vote counts. Errored runs count against consensus —
|
|
321
|
+
* they can only push a result toward human review, never toward a silent pass.
|
|
322
|
+
*/
|
|
323
|
+
|
|
324
|
+
declare function computeConsensus(runs: JudgeRun[]): Omit<ConsensusResult, "zone">;
|
|
325
|
+
|
|
326
|
+
/**
|
|
327
|
+
* Confidence-zone routing: only unanimous, high-confidence ensembles
|
|
328
|
+
* auto-resolve; everything else goes to a human.
|
|
329
|
+
*/
|
|
330
|
+
|
|
331
|
+
interface ZoneThresholds {
|
|
332
|
+
autoPass: number;
|
|
333
|
+
autoFail: number;
|
|
334
|
+
}
|
|
335
|
+
declare const DEFAULT_ZONES: ZoneThresholds;
|
|
336
|
+
declare function zoneFor(consensus: Omit<ConsensusResult, "zone">, thresholds?: ZoneThresholds): Zone;
|
|
337
|
+
|
|
338
|
+
interface EnsembleOptions {
|
|
339
|
+
provider: InferenceProvider;
|
|
340
|
+
system: string;
|
|
341
|
+
user: string;
|
|
342
|
+
/** Ensemble size; default 3. */
|
|
343
|
+
runs?: number;
|
|
344
|
+
/** Default 0. Nonzero adds noise to verdicts and warns once. */
|
|
345
|
+
temperature?: number;
|
|
346
|
+
/**
|
|
347
|
+
* Verdict schema. Defaults to the canonical one; pass your own to keep
|
|
348
|
+
* domain-specific field descriptions (they measurably steer the model).
|
|
349
|
+
* Must still produce objects matching `JudgeVerdict`.
|
|
350
|
+
*/
|
|
351
|
+
schema?: Record<string, unknown>;
|
|
352
|
+
/** Optional result cache. Requires `cacheKey`. */
|
|
353
|
+
cache?: JsonCache<JudgeRun[]>;
|
|
354
|
+
/** Content-addressed key; build it with `buildCacheKey`. */
|
|
355
|
+
cacheKey?: string;
|
|
356
|
+
/** Prefix for warnings, e.g. your tool's name. */
|
|
357
|
+
label?: string;
|
|
358
|
+
}
|
|
359
|
+
/**
|
|
360
|
+
* Run the ensemble and return every run. Cached ensembles replay identically,
|
|
361
|
+
* with each run flagged `cached: true` so cost accounting skips them.
|
|
362
|
+
*/
|
|
363
|
+
declare function runEnsemble(options: EnsembleOptions): Promise<JudgeRun[]>;
|
|
364
|
+
/** Run the ensemble and aggregate it into a zoned consensus. */
|
|
365
|
+
declare function judge(options: EnsembleOptions & {
|
|
366
|
+
zones?: ZoneThresholds;
|
|
367
|
+
}): Promise<ConsensusResult>;
|
|
368
|
+
/** Test seam: reset the once-per-process temperature warning. */
|
|
369
|
+
declare function resetTemperatureWarning(): void;
|
|
370
|
+
|
|
371
|
+
export { AnthropicProvider, type AnthropicProviderOptions, ClaudeCliProvider, type CompleteJSONRequest, type CompleteJSONResponse, type CompleteValidatedOptions, type ConsensusResult, DEFAULT_MODELS, DEFAULT_OPENAI_BASE_URL, DEFAULT_ZONES, type EnsembleOptions, type ExecFn, type ExecOptions, type ExecResult, InferenceError, type InferenceProvider, type InferenceRun, JsonCache, type JudgeRun, type JudgeVerdict, type Match, MockProvider, type MockResponse, OpenAICompatProvider, type OpenAICompatProviderOptions, PRICE_TABLE, type Pricing, type ProviderName, type ProviderSpec, type TokenUsage, VERDICT_SCHEMA, type Zone, type ZoneThresholds, buildCacheKey, completeValidatedJSON, computeConsensus, costOfRuns, costOfUsage, extractJson, judge, makeProvider, mockVerdict, pricingFor, realExec, resetTemperatureWarning, resolveProviderIdentity, runEnsemble, sha256, stripNulls, toStrictSchema, validatorFor, zoneFor };
|