@plurnk/plurnk-providers 1.2.0 → 1.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,62 @@
1
+ import test from "node:test";
2
+ import { strict as assert } from "node:assert";
3
+ import { ProviderError, classifyProviderError, toProviderError, providerSource } from "./telemetry.ts";
4
+ import { OpenAiHttpError } from "./openaiStream.ts";
5
+
6
+ // Schema pattern for TelemetryEvent.source (from grammar's TelemetryEvent.json).
7
+ const SOURCE_PATTERN = /^[a-z]+(:[a-z][a-z0-9-]*)?$/;
8
+
9
+ test("providerSource produces a schema-valid colon-namespaced source", () => {
10
+ assert.equal(providerSource("openai"), "provider:openai");
11
+ assert.match(providerSource("openrouter"), SOURCE_PATTERN);
12
+ assert.match(providerSource("xai"), SOURCE_PATTERN); // contains no underscore — fine
13
+ });
14
+
15
+ test("classifyProviderError maps HTTP status to kind", () => {
16
+ const k = (status: number) => classifyProviderError(new OpenAiHttpError(status, "body", null)).kind;
17
+ assert.equal(k(401), "unauthorized");
18
+ assert.equal(k(403), "unauthorized");
19
+ assert.equal(k(402), "quota_exceeded");
20
+ assert.equal(k(429), "rate_limit");
21
+ assert.equal(k(500), "network_failure");
22
+ assert.equal(k(503), "network_failure");
23
+ assert.equal(k(400), "invalid_response");
24
+ assert.equal(k(404), "invalid_response");
25
+ });
26
+
27
+ test("classifyProviderError: a 422 flagged grammar_invalid is transient (#548); other 422s terminal", () => {
28
+ const rejected = new OpenAiHttpError(422, JSON.stringify({ error: { type: "grammar_invalid", message: "non-conforming emission rejected: ..." } }), null);
29
+ assert.equal(classifyProviderError(rejected).kind, "grammar_invalid");
30
+ // a 422 carrying any other error.type is a malformed request — terminal
31
+ assert.equal(classifyProviderError(new OpenAiHttpError(422, JSON.stringify({ error: { type: "invalid_request_error" } }), null)).kind, "invalid_response");
32
+ // a non-JSON 422 body (edge/proxy HTML) is terminal, and the sniff does not throw
33
+ assert.equal(classifyProviderError(new OpenAiHttpError(422, "<html>Bad</html>", null)).kind, "invalid_response");
34
+ });
35
+
36
+ test("classifyProviderError treats non-HTTP errors as network_failure", () => {
37
+ assert.equal(classifyProviderError(new TypeError("fetch failed")).kind, "network_failure");
38
+ const timeout = Object.assign(new Error("timed out"), { name: "TimeoutError" });
39
+ assert.equal(classifyProviderError(timeout).kind, "network_failure");
40
+ });
41
+
42
+ test("ProviderError.toTelemetryEvent emits the canonical envelope", () => {
43
+ const e = new ProviderError("provider:openai", "rate_limit", "OpenAI 429 - slow down", { status: 429 });
44
+ const ev = e.toTelemetryEvent();
45
+ assert.deepEqual(ev, { source: "provider:openai", kind: "rate_limit", message: "OpenAI 429 - slow down", position: null });
46
+ assert.match(ev.source, SOURCE_PATTERN);
47
+ assert.ok(e instanceof Error); // still catchable as a plain Error
48
+ assert.equal(e.status, 429);
49
+ });
50
+
51
+ test("toProviderError classifies and tags an HTTP error with the source + status", () => {
52
+ const pe = toProviderError(new OpenAiHttpError(401, "no key", null), "provider:groq");
53
+ assert.equal(pe.kind, "unauthorized");
54
+ assert.equal(pe.source, "provider:groq");
55
+ assert.equal(pe.status, 401);
56
+ assert.equal(pe.cause instanceof OpenAiHttpError, true); // original preserved
57
+ });
58
+
59
+ test("toProviderError passes an existing ProviderError through unchanged", () => {
60
+ const original = new ProviderError("provider:xai", "rate_limit", "429");
61
+ assert.equal(toProviderError(original, "provider:other"), original);
62
+ });
@@ -0,0 +1,108 @@
1
+ // Provider telemetry — the TelemetryEvent envelope for transport failures.
2
+ //
3
+ // TelemetryEvent is plurnk's cross-ecosystem error envelope (canonical schema:
4
+ // @plurnk/plurnk-grammar's schemas.plurnk.dev/v0/TelemetryEvent.json). It is
5
+ // mirrored here STRUCTURALLY rather than imported, so the framework keeps zero
6
+ // dependency on grammar (see SPEC §11). Consumers route provider events through
7
+ // the same `source` + `kind` discriminator as parse/rail events.
8
+
9
+ import { OpenAiHttpError } from "./openaiStream.ts";
10
+
11
+ // Required by the schema: source (producer id) + kind (discriminator). message
12
+ // and position are optional. A transport failure isn't localizable, so it carries
13
+ // a null position; a `grammar_unenforced` event (#24) carries the divergence
14
+ // code-point offset into the model's emission, so the consumer can render a
15
+ // snippet around it — same recovery affordance the consumer already gives DSL
16
+ // parse errors.
17
+ export type TelemetryEvent = {
18
+ source: string; // e.g. "provider:openai" — schema pattern ^[a-z]+(:[a-z][a-z0-9-]*)?$
19
+ kind: string; // open vocabulary; providers mint from ProviderTelemetryKind
20
+ message?: string | null;
21
+ position?: number | null;
22
+ };
23
+
24
+ // The kinds plurnk-service routes provider failures on.
25
+ export type ProviderTelemetryKind =
26
+ | "rate_limit"
27
+ | "network_failure"
28
+ | "model_refused"
29
+ | "invalid_response"
30
+ | "unauthorized"
31
+ | "quota_exceeded"
32
+ // A 422 whose error.type is "grammar_invalid" (#548): the backend served ONE
33
+ // generation and rejected it as non-conforming. Transient, not terminal — a
34
+ // fresh sample may conform — so it rides the retry budget. Distinct from
35
+ // grammar_unenforced below: that observes a SUCCEEDED exchange; this is a
36
+ // thrown transport failure the daemon retries, then fails as a leaf.
37
+ | "grammar_invalid"
38
+ // Output did not conform to the GBNF. ALWAYS an observation, never a throw:
39
+ // a completed exchange returns its bytes with this event on
40
+ // response.telemetry (message + divergence position), whether the grammar
41
+ // was transported or withheld (filter mode). Discard/retry/escalate is
42
+ // consumer policy (#24, SPEC §13).
43
+ | "grammar_unenforced";
44
+
45
+ // Build a provider source label (`provider:<vendor>`), schema-pattern-valid.
46
+ export const providerSource = (vendor: string): string => `provider:${vendor}`;
47
+
48
+ // A transport failure carrying its TelemetryEvent classification. IS-an Error
49
+ // (existing catchers keep working); telemetry-aware consumers call
50
+ // toTelemetryEvent() and route on source+kind.
51
+ export class ProviderError extends Error {
52
+ readonly source: string;
53
+ readonly kind: ProviderTelemetryKind;
54
+ readonly status: number | null;
55
+
56
+ constructor(source: string, kind: ProviderTelemetryKind, message: string, options: { status?: number | null; cause?: unknown } = {}) {
57
+ super(message, options.cause !== undefined ? { cause: options.cause } : undefined);
58
+ this.name = "ProviderError";
59
+ this.source = source;
60
+ this.kind = kind;
61
+ this.status = options.status ?? null;
62
+ }
63
+
64
+ toTelemetryEvent(): TelemetryEvent {
65
+ return { source: this.source, kind: this.kind, message: this.message, position: null };
66
+ }
67
+ }
68
+
69
+ // An OpenAI-shaped error body carries error.type. null when the body is absent,
70
+ // non-JSON (a proxy/CDN HTML page), or unshaped — none of which is a match.
71
+ const wireErrorType = (body: string): string | null => {
72
+ try {
73
+ const { error } = JSON.parse(body) as { error?: { type?: unknown } };
74
+ return typeof error?.type === "string" ? error.type : null;
75
+ } catch {
76
+ return null;
77
+ }
78
+ };
79
+
80
+ // Map a thrown transport error to a (kind, message). Conservative; the message
81
+ // is factual, no guidance prose (consumer SPEC §15.1 policy).
82
+ export const classifyProviderError = (err: unknown): { kind: ProviderTelemetryKind; message: string } => {
83
+ if (err instanceof OpenAiHttpError) {
84
+ const { status, message } = err;
85
+ if (status === 401 || status === 403) return { kind: "unauthorized", message };
86
+ if (status === 402) return { kind: "quota_exceeded", message };
87
+ if (status === 429) return { kind: "rate_limit", message };
88
+ if (status >= 500) return { kind: "network_failure", message };
89
+ // A 422 flagged grammar_invalid is a stochastic output reject (#548) — one
90
+ // generation served then rejected, so a fresh sample may conform: transient.
91
+ // Any other 422 is a malformed request: terminal (invalid_response).
92
+ if (status === 422 && wireErrorType(err.body) === "grammar_invalid") return { kind: "grammar_invalid", message };
93
+ return { kind: "invalid_response", message };
94
+ }
95
+ const e = err as { name?: string; message?: string };
96
+ const message = (e?.message ?? String(err)) || "request failed";
97
+ // Timeouts / fetch-level failures are transport, not a usable response.
98
+ return { kind: "network_failure", message };
99
+ };
100
+
101
+ // Wrap any thrown error as a ProviderError tagged with this provider's source.
102
+ // Already-classified errors pass through unchanged.
103
+ export const toProviderError = (err: unknown, source: string): ProviderError => {
104
+ if (err instanceof ProviderError) return err;
105
+ const { kind, message } = classifyProviderError(err);
106
+ const status = err instanceof OpenAiHttpError ? err.status : null;
107
+ return new ProviderError(source, kind, message, { status, cause: err });
108
+ };
package/src/types.ts ADDED
@@ -0,0 +1,219 @@
1
+ // Provider transport contract. Providers return raw wire-level output —
2
+ // content unparsed (consumer parses via @plurnk/plurnk-grammar), reasoning
3
+ // is the wire-reported CoT only.
4
+
5
+ import type { TelemetryEvent } from "./telemetry.ts";
6
+
7
+ export interface ChatMessage {
8
+ role: "system" | "user" | "assistant";
9
+ content: string;
10
+ }
11
+
12
+ // Normalized token accounting. Invariant (enforced by normalizeUsage at the
13
+ // provider boundary): total = prompt + completion + reasoning; cached is a
14
+ // subset of prompt. `completion` is visible output EXCLUDING reasoning; the
15
+ // billable output is `completion + reasoning` (frontier providers bill reasoning
16
+ // tokens at the output rate).
17
+ export interface ProviderUsage {
18
+ readonly prompt: number; // input tokens (cached ones included)
19
+ readonly completion: number; // visible output tokens, excluding reasoning
20
+ readonly reasoning: number; // reasoning tokens, billed as output
21
+ readonly cached: number; // subset of prompt served from cache
22
+ readonly total: number; // prompt + completion + reasoning
23
+ }
24
+
25
+ // Closed set per SPEC §2. Relay/aggregator providers MUST normalize wire
26
+ // values back to one of these at the provider boundary.
27
+ export type FinishReason = "stop" | "length" | "tool_calls" | "content_filter" | null;
28
+
29
+ // A per-token logprob (#36, SPEC §14). `logprob` is the backend's RAW model
30
+ // log-probability of the emitted token — the sampling-transform-invariant
31
+ // confidence, chosen over Fireworks' post-mask `sampling_logprob` (measured
32
+ // IDENTICAL under grammar, incl. an adversarial mask; the raw value is the honest
33
+ // model belief and the correct distillation target). `top` carries the top-N
34
+ // alternatives when top_logprobs was requested. The verbatim per-token record
35
+ // (sampling_logprob, token_id, bytes, mask fields) survives on
36
+ // ProviderResponse.rawBody — this structured view is the canonical signal only.
37
+ export interface TokenAlternative {
38
+ readonly token: string;
39
+ readonly logprob: number;
40
+ }
41
+ export interface TokenLogprob {
42
+ readonly token: string;
43
+ readonly logprob: number;
44
+ readonly top?: readonly TokenAlternative[];
45
+ }
46
+
47
+ export interface ProviderAssistant {
48
+ readonly content: string;
49
+ readonly reasoning: string | null;
50
+ // Sealed reasoning (#482): a relay backend (OpenRouter fronting OpenAI
51
+ // o-series) returns the chain-of-thought ENCRYPTED — surfaced verbatim as
52
+ // items { id, subtype, encrypted: [{data, format}] } (id from the wire,
53
+ // subtype from wire position), never decoded, never synthesized. Readable text
54
+ // stays on `reasoning`. Absent when the turn produced none; consumers (agui)
55
+ // project it as REASONING_ENCRYPTED_VALUE.
56
+ readonly reasoningEncrypted?: ReadonlyArray<{ id: string | null; subtype: string; encrypted: ReadonlyArray<{ data: string; format: string | null }> }>;
57
+ readonly usage: ProviderUsage;
58
+ readonly finishReason: FinishReason;
59
+ readonly model: string;
60
+ // Per-token logprobs (#36), present ONLY when PLURNK_PROVIDERS_TOP_LOGPROBS is set
61
+ // AND the backend returned them. Absent otherwise — NEVER synthesized. Opt-in,
62
+ // per-alias: a scraping alias enables it; serving turns carry nothing.
63
+ readonly logprobs?: readonly TokenLogprob[];
64
+ // Convenience: mean of logprobs[].logprob (natural log). Absent when logprobs is.
65
+ readonly meanLogprob?: number;
66
+ }
67
+
68
+ export interface ProviderResponse {
69
+ readonly assistant: ProviderAssistant;
70
+ readonly assistantRaw: unknown;
71
+ // Per-turn provider→client metadata bag: the backend's non-standard top-level
72
+ // response fields, passed through verbatim, PLUS validated known keys we hold a
73
+ // contract for (e.g. `balancePico` — a finite pico-USD number, from the plurnk
74
+ // endpoint). The consumer (service) merges this into its Turn metadata and
75
+ // filters what reaches the client; it reads `meta`, never mines `assistantRaw`.
76
+ // Absent when the backend reported no extra fields (#23, generalized).
77
+ readonly meta?: Record<string, unknown>;
78
+ // The VERBATIM backend response body (#36, SPEC §14) — the full wire JSON for
79
+ // a non-streamed turn, or the reassembled equivalent for a streamed one.
80
+ // `assistantRaw` is a normalized DIGEST (it drops choices[]); this is the
81
+ // capture-everything record for the endpoint's fine-tune corpus. Present ONLY
82
+ // when PLURNK_PROVIDERS_RAWBODY is on — off by default so serving turns never
83
+ // carry it. Absent otherwise.
84
+ readonly rawBody?: unknown;
85
+ // Observations attached to a COMPLETED exchange (#24, SPEC §13). The model's
86
+ // bytes always flow through `assistant`; these events annotate them. Today: a
87
+ // `grammar_unenforced` event whenever the output diverges from the grammar —
88
+ // transported OR withheld (filter mode) — carrying the divergence `position`.
89
+ // The provider never adjudicates conformance; discard/retry/escalate/
90
+ // self-correct is consumer policy. Absent when the turn produced no telemetry.
91
+ readonly telemetry?: readonly TelemetryEvent[];
92
+ }
93
+
94
+ export interface Provider {
95
+ // `grammar` is an optional GBNF string (canonically @plurnk/plurnk-grammar's
96
+ // plurnk.gbnf, possibly root-substituted by the consumer). Backends that
97
+ // support grammar-constrained sampling attach it verbatim; all others
98
+ // ignore it. The provider never chooses or modifies the grammar — whether
99
+ // to constrain and which root variant to send is consumer policy (SPEC §13).
100
+ //
101
+ // `maxTokens` is the consumer's per-call output ceiling (wire `max_tokens`).
102
+ // Without it, most servers generate UNBOUNDED (llama-server n_predict -1) —
103
+ // under a multi-op grammar that degenerates to the context wall (SPEC §13),
104
+ // so a constrained consumer is expected to pass it. Policy stays the
105
+ // consumer's; the provider only transports.
106
+ //
107
+ // `workerId` is the REQUIRED, opaque, stable identity of the consumer's work
108
+ // stream (loop/run). Providers MAY key backend affinity on it — e.g.
109
+ // llama-server slot pinning for KV-cache reuse — and MUST NOT interpret
110
+ // its content. The consumer never sees or chooses backend resources
111
+ // (slot integers, connections); the *mechanism* is the provider's (#11).
112
+ //
113
+ // `attributions` (per-turn, runtime-observed) and `client` (workspace-stable,
114
+ // self-identified) are first-party telemetry the consumer hands down: which
115
+ // installed plugin packages dispatched this turn, and which frontend
116
+ // originated the worker. They are forwarded ONLY by a provider whose spec opts
117
+ // in (the first-party `plurnk` endpoint, via `Plurnk-Attribution` /
118
+ // `Plurnk-Client` headers); every other provider DROPS them — the gate is
119
+ // structural so first-party metadata can never leak to a third-party backend.
120
+ //
121
+ // `sampling` is an optional bag of standard OpenAI-compat sampling params
122
+ // (temperature, top_p, top_k, min_p, penalties, stop, seed, …) forwarded into
123
+ // the request body UNDER the provider's managed fields — model/messages/grammar/
124
+ // reasoning/max_tokens/slot always win, and transport/protocol keys (stream,
125
+ // response_format, grammar, id_slot) are stripped, so it carries sampling intent
126
+ // only and can't bypass grammar transport (SPEC §8 holds). A PROXY consumer (the
127
+ // plurnk endpoint fronting its own backends) uses it to pass its caller's sampling
128
+ // knobs through; a direct consumer typically leaves it unset.
129
+ //
130
+ // `strikes` is the worker's CURRENT rail-strike streak at time-of-generate
131
+ // (0 = clean; a clean turn zeroes it; every loop starts at 0 — contract:
132
+ // plurnk-service#313). Forwarded as a `Plurnk-Strikes` header ONLY under the
133
+ // same firstPartyMetadata gate as attributions/client; dropped everywhere
134
+ // else. Headers only — the packet NEVER carries strike state (the model must
135
+ // not see engine accounting; it would become a metric to game).
136
+ //
137
+ // `workspaceId`/`loop`/`turn` (#404, per #391) are the turn COORDINATE — the
138
+ // daemon-side sequence of the turn being generated, which the endpoint can
139
+ // never scrape from the wire. Forwarded as `Plurnk-Workspace-Id`/`Plurnk-Loop`/
140
+ // `Plurnk-Turn` ONLY under the same firstPartyMetadata gate; dropped
141
+ // everywhere else. Coordinates are 1-based: absent/0 emits no header (no
142
+ // strikes-style zero exception). Headers only, never the packet.
143
+ generate(args: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse>;
144
+ // The model's context window in tokens. The provider RESOLVES it (operator pin
145
+ // -> live probe -> @plurnk/plurnk-models catalog). A CLOUD provider (no probe)
146
+ // FAILS AT CONSTRUCTION when it can't (#419/#417: never budget against a wrong
147
+ // number). A PROBING provider (openai/llama-server) instead DEGRADES to null on a
148
+ // probe miss - a blip must not crash it (#34) - and surfaces it once
149
+ // (PLURNK_CONTEXT_UNKNOWN). So null still means "window unknown -> no cap"; the
150
+ // consumer must NOT improvise a stand-in from it (#421). NOTE: under llama-server
151
+ // --parallel N, the window is PER SLOT (the server splits --ctx-size across slots
152
+ // and reports the divided value).
153
+ readonly contextWindow: number | null;
154
+ readonly model: string;
155
+ // OPTIONAL (#37): the backend's SELF-REPORTED served model id, from a
156
+ // /v1/models-shaped probe (llama-server today; any such backend). For a local
157
+ // alias, `model` is the alias but this is the real served name (the .gguf) the
158
+ // tokenizer seam maps exactly. Read-only, best-effort, no extra probing —
159
+ // absent when no probe ran. Consumers resolve `servedModel ?? model`.
160
+ readonly servedModel?: string;
161
+ // OPTIONAL resolved capability (#34): true when a transported grammar will
162
+ // actually constrain the decode (rails LIVE), false/undefined otherwise —
163
+ // introspectable so the consumer can fail hard on a dark-rails boot instead
164
+ // of discovering it from unconstrained emissions.
165
+ readonly constrainsOutput?: boolean;
166
+ // OPTIONAL resolved capability (#43): true when this backend decodes
167
+ // UNBOUNDED absent a caller cap — llama-server honors n_predict to the
168
+ // context wall (the 30,736-junk-token wall-run, providers#10), so a consumer
169
+ // MUST bring an output envelope (SPEC §13). Cloud backends that silently
170
+ // clamp an over-ask (fireworks/xai, verified live) never set this; undefined
171
+ // = no claim. Introspectable so a consumer can refuse AT BOOT a local alias
172
+ // with no declared envelope, instead of dying mid-turn in partition math.
173
+ readonly requiresMaxTokens?: boolean;
174
+ // OPTIONAL generation-envelope reserves (#507, owner-ruled) — the amounts OF
175
+ // the DETECTED window reserved for reasoning and completion: floor
176
+ // percentages of `contextWindow`, or absolute per-alias pins that win
177
+ // outright. The consumer's prompt budget is `contextWindow - reasoningReserve
178
+ // - completionReserve - <its own packing-safety margin>`; the generation cap
179
+ // is the two pooled. `null` = underivable (window unknown, no absolute pin) →
180
+ // the consumer's no-cap path. Absent = a bare sibling makes NO claim (treated
181
+ // as null). All first-party providers claim, so null means genuinely-unknown.
182
+ readonly reasoningReserve?: number | null;
183
+ readonly completionReserve?: number | null;
184
+ // Provider-owned tokenizer. Synchronous, non-negative integer. Without an
185
+ // exact family configured this is the chars/2 UPPER BOUND (surfaced at
186
+ // construction, never silent) — safe for refusal math, not an exact count.
187
+ countTokens(text: string): number;
188
+ // OPTIONAL capability: exact tokenization served by the backend's own vocab
189
+ // (llama-server /tokenize) — token ids in the model's real vocabulary.
190
+ // Present ONLY when the backend exposes such an endpoint (probe-gated);
191
+ // `tokenize === undefined` means the backend can't. Exact-counting
192
+ // consumers (the tokenizer seam) prefer this over any client-side data.
193
+ tokenize?(text: string): Promise<number[]>;
194
+ // Provider-owned cost calculation. Returns pico-USD (1e-12 USD).
195
+ // Returns 0 for siblings/models with no known rates.
196
+ costFor(usage: ProviderUsage): number;
197
+ }
198
+
199
+ // ProviderAlias moved to @plurnk/plurnk-aliases (the zero-dep parser, #27);
200
+ // index.ts re-exports it so the "." surface is unchanged.
201
+
202
+ // Per-alias instantiation overrides, threaded from the alias cascade into the
203
+ // factory. `baseUrl` lets two aliases on the SAME provider name (openai, ollama)
204
+ // target DIFFERENT endpoints — the only way to run N self-hosted boxes, since the
205
+ // provider's own base-URL env var binds one URL per name. Absent → the provider
206
+ // resolves its base from its env var as before.
207
+ export interface ProviderOptions {
208
+ readonly baseUrl?: string;
209
+ }
210
+
211
+ // Each provider package's default export MUST be a factory:
212
+ // static fromEnv(env, model, options?) → Provider | Promise<Provider>
213
+ // `model` is the second positional arg because PLURNK_MODEL_<alias>=<provider>/<model>
214
+ // is parsed by the registry; the resolved model id flows through. `options` is an
215
+ // optional third arg (per-alias overrides, e.g. baseUrl); a factory that ignores
216
+ // it keeps working unchanged.
217
+ export interface ProviderFactory {
218
+ fromEnv(env: NodeJS.ProcessEnv, model: string, options?: ProviderOptions): Provider | Promise<Provider>;
219
+ }
@@ -0,0 +1,136 @@
1
+ import test from "node:test";
2
+ import { strict as assert } from "node:assert";
3
+ import { normalizeUsage, computeCost } from "./usage.ts";
4
+
5
+ // — normalizeUsage —
6
+
7
+ test("normalizeUsage: Gemini-style — reasoning recovered from total gap", () => {
8
+ // Real Gemini OAI-compat shape: no details, reasoning hidden in total.
9
+ const u = normalizeUsage({ prompt_tokens: 19, completion_tokens: 285, total_tokens: 1165 });
10
+ assert.deepEqual(u, { prompt: 19, completion: 285, reasoning: 861, cached: 0, total: 1165 });
11
+ assert.equal(u.prompt + u.completion + u.reasoning, u.total); // invariant
12
+ });
13
+
14
+ test("normalizeUsage: OpenAI-style — reasoning split out of completion_tokens", () => {
15
+ // completion_tokens includes reasoning; total = prompt + completion.
16
+ const u = normalizeUsage({
17
+ prompt_tokens: 10,
18
+ completion_tokens: 100,
19
+ total_tokens: 110,
20
+ completion_tokens_details: { reasoning_tokens: 40 },
21
+ });
22
+ assert.deepEqual(u, { prompt: 10, completion: 60, reasoning: 40, cached: 0, total: 110 });
23
+ assert.equal(u.completion + u.reasoning, 100); // billable output unchanged
24
+ });
25
+
26
+ test("normalizeUsage: xAI/Grok-style — reasoning is ADDITIVE, not subtracted from completion (#28)", () => {
27
+ // Real grok-4.3 shape: completion_tokens is visible-only; reasoning_tokens is
28
+ // detailed but ADDITIVE — total = prompt + completion + reasoning.
29
+ const u = normalizeUsage({
30
+ prompt_tokens: 143,
31
+ completion_tokens: 1,
32
+ total_tokens: 441,
33
+ prompt_tokens_details: { cached_tokens: 128 },
34
+ completion_tokens_details: { reasoning_tokens: 297 },
35
+ });
36
+ assert.deepEqual(u, { prompt: 143, completion: 1, reasoning: 297, cached: 128, total: 441 });
37
+ assert.equal(u.completion, 1); // visible output preserved, NOT zeroed by the subtraction
38
+ assert.equal(u.completion + u.reasoning, 298); // billable output = visible + reasoning
39
+ });
40
+
41
+ test("normalizeUsage: cached read from prompt_tokens_details (OpenAI nesting)", () => {
42
+ const u = normalizeUsage({ prompt_tokens: 50, completion_tokens: 10, total_tokens: 60, prompt_tokens_details: { cached_tokens: 30 } });
43
+ assert.equal(u.cached, 30);
44
+ });
45
+
46
+ test("normalizeUsage: top-level cached_tokens still honored", () => {
47
+ const u = normalizeUsage({ prompt_tokens: 50, completion_tokens: 10, total_tokens: 60, cached_tokens: 12 });
48
+ assert.equal(u.cached, 12);
49
+ });
50
+
51
+ test("normalizeUsage: no reasoning — plain prompt+completion", () => {
52
+ const u = normalizeUsage({ prompt_tokens: 10, completion_tokens: 20, total_tokens: 30 });
53
+ assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
54
+ });
55
+
56
+ test("normalizeUsage: missing total is reconstructed, never negative reasoning", () => {
57
+ const u = normalizeUsage({ prompt_tokens: 10, completion_tokens: 20 });
58
+ assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
59
+ });
60
+
61
+ test("normalizeUsage: absent usage → all zeros", () => {
62
+ assert.deepEqual(normalizeUsage(null), { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 });
63
+ assert.deepEqual(normalizeUsage(undefined), { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 });
64
+ });
65
+
66
+ // -- Fireworks-style: reasoning shipped as TEXT, folded into completion, not itemized (#425) --
67
+
68
+ test("normalizeUsage: fireworks folds reasoning into completion -- re-split by text proportion, sum preserved (#425)", () => {
69
+ // total = prompt + completion (no gap), reasoning_tokens absent, but 750 vs 250
70
+ // chars of reasoning vs content came back. Split completion 75/25; cost base held.
71
+ const u = normalizeUsage(
72
+ { prompt_tokens: 100, completion_tokens: 1000, total_tokens: 1100 },
73
+ "r".repeat(750),
74
+ "c".repeat(250),
75
+ );
76
+ assert.deepEqual(u, { prompt: 100, completion: 250, reasoning: 750, cached: 0, total: 1100 });
77
+ assert.equal(u.completion + u.reasoning, 1000); // billable output byte-identical
78
+ assert.equal(u.prompt + u.completion + u.reasoning, u.total); // invariant
79
+ });
80
+
81
+ test("normalizeUsage: pure-reasoning turn (empty content) attributes all completion to reasoning (#425)", () => {
82
+ // The run52 runaway shape: 0 visible content, the whole budget spent reasoning.
83
+ const u = normalizeUsage(
84
+ { prompt_tokens: 100, completion_tokens: 500, total_tokens: 600 },
85
+ "t".repeat(9000),
86
+ "",
87
+ );
88
+ assert.deepEqual(u, { prompt: 100, completion: 0, reasoning: 500, cached: 0, total: 600 });
89
+ });
90
+
91
+ test("normalizeUsage: text args never perturb the itemized (reasoning_tokens) path", () => {
92
+ // OpenAI o-series reports reasoning_tokens -> that split wins, text is ignored.
93
+ const u = normalizeUsage(
94
+ { prompt_tokens: 10, completion_tokens: 100, total_tokens: 110, completion_tokens_details: { reasoning_tokens: 40 } },
95
+ "r".repeat(999), "c".repeat(1),
96
+ );
97
+ assert.deepEqual(u, { prompt: 10, completion: 60, reasoning: 40, cached: 0, total: 110 });
98
+ });
99
+
100
+ test("normalizeUsage: text args never perturb the Gemini gap path (gap already yields reasoning)", () => {
101
+ // A real total gap means reasoning is itemized-by-subtraction; do not re-split.
102
+ const u = normalizeUsage(
103
+ { prompt_tokens: 19, completion_tokens: 285, total_tokens: 1165 },
104
+ "r".repeat(500), "c".repeat(500),
105
+ );
106
+ assert.equal(u.reasoning, 861); // from the gap, NOT a text re-split
107
+ assert.equal(u.completion, 285);
108
+ });
109
+
110
+ test("normalizeUsage: no total reported -> re-split skipped, reasoning stays 0 (cannot split an unknown base)", () => {
111
+ const u = normalizeUsage(
112
+ { prompt_tokens: 10, completion_tokens: 20 },
113
+ "r".repeat(500), "c".repeat(500),
114
+ );
115
+ assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
116
+ });
117
+
118
+ // — computeCost —
119
+
120
+ test("computeCost: bills reasoning at the output rate", () => {
121
+ // 100 input, 0 cached, 50 completion + 200 reasoning = 250 output.
122
+ const usage = { prompt: 100, completion: 50, reasoning: 200, cached: 0, total: 350 };
123
+ // input 1 pico/tok, output 10 pico/tok → 100*1 + 250*10 = 2600
124
+ assert.equal(computeCost(usage, { input: 1, output: 10, cached: 0 }), 2600);
125
+ });
126
+
127
+ test("computeCost: cached prompt billed at the cache rate, remainder at input", () => {
128
+ const usage = { prompt: 1000, completion: 0, reasoning: 0, cached: 400, total: 1000 };
129
+ // 600 non-cached @5 + 400 cached @1 = 3000 + 400 = 3400
130
+ assert.equal(computeCost(usage, { input: 5, output: 99, cached: 1 }), 3400);
131
+ });
132
+
133
+ test("computeCost: zero rates → 0", () => {
134
+ const usage = { prompt: 9, completion: 9, reasoning: 9, cached: 9, total: 27 };
135
+ assert.equal(computeCost(usage, { input: 0, output: 0, cached: 0 }), 0);
136
+ });
package/src/usage.ts ADDED
@@ -0,0 +1,82 @@
1
+ // Usage normalization + cost — the shared token-accounting model.
2
+ //
3
+ // Providers report token usage in two incompatible ways:
4
+ // - OpenAI-style: reasoning is a SUBSET of completion_tokens, surfaced via
5
+ // completion_tokens_details.reasoning_tokens; total = prompt + completion.
6
+ // - Gemini-style: reasoning is OMITTED from completion_tokens and only
7
+ // recoverable as total - prompt - completion (no details field at all).
8
+ // - Fireworks-style: reasoning ships as TEXT (reasoning_content) but is folded
9
+ // into completion_tokens with NO reasoning_tokens itemization -- unrecoverable
10
+ // from the numbers alone, so it is re-split from the emitted text lengths.
11
+ // normalizeUsage collapses all three into one invariant (see ProviderUsage):
12
+ // total = prompt + completion + reasoning; cached ⊆ prompt;
13
+ // completion EXCLUDES reasoning; billable output = completion + reasoning.
14
+
15
+ import type { ProviderUsage } from "./types.ts";
16
+
17
+ // Raw OpenAI-compatible usage block — the superset of fields providers emit.
18
+ export type RawUsage = {
19
+ prompt_tokens?: number;
20
+ completion_tokens?: number;
21
+ total_tokens?: number;
22
+ cached_tokens?: number;
23
+ prompt_tokens_details?: { cached_tokens?: number };
24
+ completion_tokens_details?: { reasoning_tokens?: number };
25
+ };
26
+
27
+ export const normalizeUsage = (raw: RawUsage | null | undefined, reasoningText = "", contentText = ""): ProviderUsage => {
28
+ const prompt = raw?.prompt_tokens ?? 0;
29
+ const completionRaw = raw?.completion_tokens ?? 0;
30
+ const reportedTotal = raw?.total_tokens ?? 0;
31
+ // OpenAI nests cached under prompt_tokens_details; others put it top-level.
32
+ const cached = raw?.prompt_tokens_details?.cached_tokens ?? raw?.cached_tokens ?? 0;
33
+ const reasoningDetail = raw?.completion_tokens_details?.reasoning_tokens;
34
+
35
+ let completion: number;
36
+ let reasoning: number;
37
+ if (reasoningDetail !== undefined) {
38
+ reasoning = reasoningDetail;
39
+ // reasoning_tokens is reported two incompatible ways when detailed. OpenAI
40
+ // o-series folds it INTO completion_tokens (subset: total = prompt + completion,
41
+ // so completion must have reasoning subtracted out). xAI/Grok reports it
42
+ // ADDITIVE to a visible-only completion_tokens (total = prompt + completion +
43
+ // reasoning), where subtracting wrongly zeroes the visible output (#28). Tell
44
+ // them apart by the total identity; with no total reported, fall back on the
45
+ // impossible-subset signal — reasoning can't exceed the completion it's a
46
+ // subset of.
47
+ const additive = reportedTotal > 0
48
+ ? reportedTotal === prompt + completionRaw + reasoningDetail
49
+ : completionRaw < reasoningDetail;
50
+ completion = additive ? completionRaw : Math.max(0, completionRaw - reasoningDetail);
51
+ } else {
52
+ // Gemini-style (or no reasoning): tokens beyond prompt+completion are
53
+ // reasoning. Only trust the gap when a total was actually reported.
54
+ reasoning = reportedTotal > 0 ? Math.max(0, reportedTotal - prompt - completionRaw) : 0;
55
+ completion = completionRaw;
56
+ // Fireworks folds reasoning INTO completion_tokens and itemizes no
57
+ // reasoning_tokens, so a turn that shipped only reasoning reads reasoning=0
58
+ // though 300k chars of it arrived (#425). When reasoning TEXT came back but
59
+ // the reported totals leave no gap, re-split the reported completion by the
60
+ // emitted text proportions. Sum-preserving: billable output
61
+ // (completion+reasoning) and cost are byte-identical; only the
62
+ // visible/reasoning gauge is corrected (pure-reasoning turn -> completion 0).
63
+ if (reasoning === 0 && reasoningText.length > 0 && reportedTotal > 0 && completionRaw > 0) {
64
+ reasoning = Math.round(completionRaw * reasoningText.length / (reasoningText.length + contentText.length));
65
+ completion = completionRaw - reasoning;
66
+ }
67
+ }
68
+ const total = reportedTotal > 0 ? reportedTotal : prompt + completion + reasoning;
69
+ return { prompt, completion, reasoning, cached, total };
70
+ };
71
+
72
+ // Per-token rates in pico-USD (1e-12 USD).
73
+ export type TokenRates = { input: number; output: number; cached: number };
74
+
75
+ // The one cost formula every provider uses: non-cached prompt at the input
76
+ // rate, cached prompt at the cache rate, and billable output (completion +
77
+ // reasoning) at the output rate.
78
+ export const computeCost = (usage: ProviderUsage, rates: TokenRates): number => {
79
+ const nonCachedPrompt = Math.max(0, usage.prompt - usage.cached);
80
+ const output = usage.completion + usage.reasoning;
81
+ return Math.round(nonCachedPrompt * rates.input + usage.cached * rates.cached + output * rates.output);
82
+ };
@@ -0,0 +1,31 @@
1
+ import test, { mock } from "node:test";
2
+ import { strict as assert } from "node:assert";
3
+ import { emitWarningOnce, resetEmittedWarnings } from "./warnings.ts";
4
+
5
+ test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); });
6
+
7
+ test("#40: same (code, message) fires once per process", () => {
8
+ const seen: string[] = [];
9
+ mock.method(process, "emitWarning", (msg: string | Error) => { seen.push(String(msg)); });
10
+ emitWarningOnce("openai provider: heuristic", "PLURNK_TOKENIZER_HEURISTIC");
11
+ emitWarningOnce("openai provider: heuristic", "PLURNK_TOKENIZER_HEURISTIC");
12
+ emitWarningOnce("openai provider: heuristic", "PLURNK_TOKENIZER_HEURISTIC");
13
+ assert.deepEqual(seen, ["openai provider: heuristic"]);
14
+ });
15
+
16
+ test("#40: dedup is by (code, MESSAGE), not code — a second provider's surfacing is never suppressed", () => {
17
+ const seen: string[] = [];
18
+ mock.method(process, "emitWarning", (msg: string | Error) => { seen.push(String(msg)); });
19
+ emitWarningOnce("openai provider: heuristic", "PLURNK_TOKENIZER_HEURISTIC");
20
+ emitWarningOnce("groq provider: heuristic", "PLURNK_TOKENIZER_HEURISTIC"); // same code, different provider
21
+ assert.deepEqual(seen, ["openai provider: heuristic", "groq provider: heuristic"]);
22
+ });
23
+
24
+ test("#40: resetEmittedWarnings clears the set (test-order independence)", () => {
25
+ const seen: string[] = [];
26
+ mock.method(process, "emitWarning", (msg: string | Error) => { seen.push(String(msg)); });
27
+ emitWarningOnce("m", "C");
28
+ resetEmittedWarnings();
29
+ emitWarningOnce("m", "C");
30
+ assert.equal(seen.length, 2);
31
+ });
Binary file