@plurnk/plurnk-providers 1.1.1 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/.env.defaults +10 -0
  2. package/README.md +50 -60
  3. package/SPEC.md +7 -6
  4. package/dist/OpenAICompat.d.ts +4 -0
  5. package/dist/OpenAICompat.d.ts.map +1 -1
  6. package/dist/OpenAICompat.js +22 -3
  7. package/dist/OpenAICompat.js.map +1 -1
  8. package/dist/ProviderRegistry.js +2 -2
  9. package/dist/ProviderRegistry.js.map +1 -1
  10. package/dist/env.d.ts +1 -0
  11. package/dist/env.d.ts.map +1 -1
  12. package/dist/env.js +15 -3
  13. package/dist/env.js.map +1 -1
  14. package/dist/index.d.ts +1 -1
  15. package/dist/index.d.ts.map +1 -1
  16. package/dist/index.js +1 -1
  17. package/dist/index.js.map +1 -1
  18. package/dist/openaiStream.d.ts.map +1 -1
  19. package/dist/openaiStream.js +12 -0
  20. package/dist/openaiStream.js.map +1 -1
  21. package/dist/standardProviders.d.ts +1 -0
  22. package/dist/standardProviders.d.ts.map +1 -1
  23. package/dist/standardProviders.js +47 -7
  24. package/dist/standardProviders.js.map +1 -1
  25. package/package.json +8 -6
  26. package/src/Mock.test.ts +142 -0
  27. package/src/Mock.ts +95 -0
  28. package/src/OpenAICompat.test.ts +1107 -0
  29. package/src/OpenAICompat.ts +756 -0
  30. package/src/Pool.test.ts +155 -0
  31. package/src/Pool.ts +134 -0
  32. package/src/ProviderRegistry.test.ts +176 -0
  33. package/src/ProviderRegistry.ts +93 -0
  34. package/src/boundaries.test.ts +24 -0
  35. package/src/discover.test.ts +123 -0
  36. package/src/discover.ts +112 -0
  37. package/src/env.test.ts +190 -0
  38. package/src/env.ts +211 -0
  39. package/src/index.ts +51 -0
  40. package/src/lexicon-guard.test.ts +58 -0
  41. package/src/openaiStream.ts +279 -0
  42. package/src/standardProviders.test.ts +925 -0
  43. package/src/standardProviders.ts +618 -0
  44. package/src/telemetry.test.ts +62 -0
  45. package/src/telemetry.ts +108 -0
  46. package/src/types.ts +219 -0
  47. package/src/usage.test.ts +136 -0
  48. package/src/usage.ts +82 -0
  49. package/src/warnings.test.ts +31 -0
  50. package/src/warnings.ts +0 -0
@@ -0,0 +1,756 @@
1
+ // Shared OpenAI-compatible provider. Implements the universal generate()
2
+ // spine — signal merging, the SSE call, usage mapping, finishReason
3
+ // normalization, response assembly — that every sibling had duplicated.
4
+ //
5
+ // Composition, not inheritance: the per-provider deltas (resolved URL, auth
6
+ // headers, reasoning translation style, tokenizer, cost) arrive as config.
7
+ // A sibling's fromEnv probes whatever it needs (catalog, pricing, context
8
+ // window), builds the config, and returns `new OpenAICompatProvider(config)`.
9
+ // Pure-config providers come from ./standardProviders.ts with no sibling at all.
10
+
11
+ import type { ChatMessage, FinishReason, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
12
+ import type { Reasoning, ReserveSpec } from "./env.ts";
13
+ import { chatCompletionStream, chatCompletion, OpenAiHttpError, isEdgeStatus, type StreamResponse } from "./openaiStream.ts";
14
+ import { normalizeUsage } from "./usage.ts";
15
+ import { toProviderError, classifyProviderError, ProviderError, type TelemetryEvent } from "./telemetry.ts";
16
+ import { validateGbnf, type Verdict } from "@plurnk/gbnf";
17
+ import { emitWarningOnce } from "./warnings.ts";
18
+
19
+ // How the reasoning intent (PLURNK_PROVIDERS_REASONING: off | adaptive | on, plus
20
+ // REASONING_BUDGET iff on — #32/#33) translates to each backend's wire mechanism
21
+ // (SPEC §4); the per-style mapping lives in #reasoningBody. Non-obvious ones:
22
+ // "template" ALWAYS emits enable_thinking — the explicit false is llama-server's
23
+ // only working off-switch (§13); "anthropic" uses the `thinking` object and IGNORES
24
+ // reasoning_effort; "effort_explicit" (fireworks) sends the EXPLICIT "none" for OFF
25
+ // instead of omitting — reason-by-DEFAULT models (DeepSeek V4 defaults
26
+ // 'high') keep reasoning when the field is omitted, fatal under an active grammar
27
+ // (#30). Intent maps IDENTICALLY with or without a transported grammar — fireworks
28
+ // masks only the content channel, so reasoning and rails coexist in one call
29
+ // (canary-verified; the #32 clamp is lifted — it caused the plan-less service#331).
30
+ export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "template" | "anthropic";
31
+
32
+ // How a caller-supplied GBNF grammar is carried on the wire — backends accept
33
+ // different shapes for the SAME GBNF (probed/configured, never guessed; §13); the
34
+ // wire shape per style lives in #grammarBody. "none" means the grammar is NOT sent
35
+ // (never silently — so a constrained consumer can't mistake unconstrained output
36
+ // for enforced).
37
+ export type GrammarStyle = "none" | "llamacpp" | "response_format";
38
+
39
+ export type OpenAICompatConfig = {
40
+ model: string;
41
+ url: string; // fully-resolved chat-completions URL
42
+ fetchTimeoutMs: number;
43
+ headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
44
+ contextWindow?: number | null; // default null; caller resolves-or-fails (#419), narrows to required with the interface
45
+ reasoningStyle?: ReasoningStyle; // default "none"
46
+ countTokens?: (text: string) => number; // default chars/2 upper-bound heuristic
47
+ costFor?: (usage: ProviderUsage) => number; // default () => 0
48
+ source?: string; // telemetry source, e.g. "provider:openai"; default "provider"
49
+ grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
50
+ // #518: send the OpenAI-standard `prompt_cache_key` set to workerId, so a
51
+ // serverless backend with REPLICA-LOCAL prompt caching (fireworks, verified)
52
+ // pins a worker's turns to one replica and claims its stable prefix. Default
53
+ // false -- a backend that strict-validates unknown fields 400s, so enable only
54
+ // where the field is accepted. Same identity that already drives slot affinity.
55
+ promptCacheKey?: boolean;
56
+ gbnfDebug?: boolean; // PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar locally + throw on invalid, but DON'T transport it (run unconstrained); default false
57
+ streaming?: boolean; // SSE transport (default true); false → one non-streamed JSON
58
+ firstPartyMetadata?: boolean; // forward per-turn attributions + client as Plurnk-* headers (plurnk only); default false
59
+ apiKeyRejectedMessage?: string; // #537: friendly hint when a PRESENT key is 401/403-rejected (distinct from unset); default undefined
60
+ eosText?: string; // #539: server-reported eos_token, stripped from the content tail (--special renders it as text); default undefined
61
+ balanceMetaKey?: string; // top-level response field carrying account balance (pico-USD) → validated meta.balancePico (plurnk only, #23); default unset
62
+ // Slot affinity wiring (provider-INTERNAL — never consumer-facing, #11).
63
+ supportsSlotPinning?: boolean; // backend accepts an `id_slot` body field (llama-server); default false
64
+ slotCount?: number | null; // probed slot count for pinning backends; default null
65
+ // Backend-served exact tokenization (llama-server /tokenize). When set, the
66
+ // provider exposes the optional `tokenize()` capability — the model's OWN
67
+ // vocab, no client-side tokenizer data needed; default unset (capability absent).
68
+ tokenizeUrl?: string;
69
+ // #37: the backend's self-reported served model id (from the /v1/models probe),
70
+ // surfaced as Provider.servedModel. For a local llama-server the wire `model` is
71
+ // the alias; this is the real name (the .gguf) the tokenizer seam maps. Absent
72
+ // when no probe ran or it read no row.
73
+ servedModel?: string;
74
+ // #43: backend decodes unbounded without a caller cap (llama-server n_predict
75
+ // to the wall) — surfaced as Provider.requiresMaxTokens so consumers can
76
+ // boot-refuse an envelope-less local alias. Default unset (no claim).
77
+ requiresMaxTokens?: boolean;
78
+ // The side-channel reasoning intent — REQUIRED, no in-code default
79
+ // (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
80
+ // { mode: off|adaptive|on, budget: iff on }. The provider maps it to the
81
+ // backend's mechanism via reasoningStyle; budget is only ever a magnitude,
82
+ // never a hidden activation flag (#33).
83
+ reasoning: Reasoning;
84
+ // Decode tuning: no in-code defaults; the canonical measured values (0.2 /
85
+ // 1.15) ship as the floor in .env.defaults (alias-scopable). `temperature` is the
86
+ // DEFAULT for EVERY request, spread UNDER caller sampling (#30/endpoint#7).
87
+ // `repeatPenalty` is the FLOOR the provider manages wherever a grammar rides
88
+ // (greedy-under-mask loops without it, #9) — the VALUE is operator config;
89
+ // WHERE it applies stays mechanism. `retryDelayMs` is the transient-retry
90
+ // backoff base (attempt N waits retryDelayMs * 2^(N-1); Retry-After wins).
91
+ temperature: number;
92
+ repeatPenalty: number;
93
+ // #426: anti-degeneration guard on the CLOUD path (grammarStyle "none"), where the
94
+ // repeat_penalty multiplier isn't available - the OpenAI-standard frequency_penalty.
95
+ // Optional (default 0 = off) so an out-of-date plugin that omits it just runs unguarded
96
+ // rather than failing construction; the standard factory always supplies it.
97
+ frequencyPenalty?: number;
98
+ // #567: llama.cpp anti-repetition-LOOP controls, sent on the llamacpp path (DRY is a
99
+ // llama.cpp sampler). DRY penalizes repeated SEQUENCES with a penalty escalating in run
100
+ // length — the tool for a plan-restart loop a single-token repeat_penalty over a short
101
+ // window can't see. The GENERIC DRY defaults (0.8/1.75/2) ship as a floor in .env.defaults
102
+ // like repeatPenalty — applied for a detected llama.cpp backend, customer-overridable;
103
+ // absent (a plugin omitting them) = the box's own default. repeatLastN widens the window.
104
+ dryMultiplier?: number;
105
+ dryBase?: number;
106
+ dryAllowedLength?: number;
107
+ repeatLastN?: number;
108
+ retryDelayMs: number;
109
+ // Transient-failure retry budget — REQUIRED, no in-code default
110
+ // (PLURNK_PROVIDERS_RETRY_ATTEMPTS, a non-negative int): 0 = surface the
111
+ // first failure; N = up to N retries on a transient error (§4, #18).
112
+ retryAttempts: number;
113
+ // Data-capture knobs (#36), OFF by default — the flag IS the isolation, so a
114
+ // serving turn requests nothing and carries nothing. `topLogprobs`: when a
115
+ // non-negative int, request `logprobs:true, top_logprobs:<n>` and surface the
116
+ // per-token confidence on assistant.logprobs (PLURNK_PROVIDERS_TOP_LOGPROBS;
117
+ // null = off). `rawBody`: when true, attach the verbatim wire body to
118
+ // response.rawBody (PLURNK_PROVIDERS_RAWBODY). Both universal — any backend,
119
+ // gated per-alias.
120
+ topLogprobs?: number | null;
121
+ rawBody?: boolean;
122
+ // #507 (owner-ruled): the generation-envelope reserves, env-read via
123
+ // envelopeFromEnv — a percentage of the DETECTED window or an absolute token
124
+ // count. Optional so an out-of-date sibling keeps constructing (no claim);
125
+ // the standard factory always supplies them. Resolved against contextWindow
126
+ // at read time (getters), so a probe that lands after config assembly still
127
+ // derives correctly.
128
+ reasoningReserve?: ReserveSpec;
129
+ completionReserve?: ReserveSpec;
130
+ // #507: the plurnk.ai router owns tuning (SPEC §5) — false suppresses the
131
+ // client-side temperature/penalty FLOORS on this provider (caller `sampling`
132
+ // still passes through verbatim). Default true (floors ride).
133
+ tuningFloors?: boolean;
134
+ };
135
+
136
+ // Transient classifications worth retrying: rate_limit (429) and network_failure
137
+ // (5xx, timeout, connection reset) are transport; grammar_invalid (a 422 output
138
+ // reject a fresh sample may satisfy, #548) rides the same bounded budget.
139
+ // unauthorized, quota_exceeded, invalid_response, model_refused are terminal —
140
+ // retrying just burns time and budget.
141
+ const RETRYABLE: ReadonlySet<string> = new Set(["rate_limit", "network_failure", "grammar_invalid"]);
142
+
143
+ // #539: drop trailing occurrences of a server-rendered EOG marker. llama-server
144
+ // under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
145
+ // trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
146
+ // can never eat body content (a body ending in the literal marker isn't producible
147
+ // under the grammar, and is vanishingly rare unconstrained).
148
+ const stripTrailingSpecial = (content: string, marker: string): string => {
149
+ if (marker.length === 0) return content;
150
+ let out = content;
151
+ while (out.endsWith(marker)) out = out.slice(0, -marker.length);
152
+ return out;
153
+ };
154
+
155
+ // Sleep that rejects the moment `signal` aborts (caller cancellation must not
156
+ // wait out a backoff). Resolves normally on timeout.
157
+ const sleepWithAbort = (ms: number, signal: AbortSignal | undefined): Promise<void> =>
158
+ new Promise((resolve, reject) => {
159
+ if (signal?.aborted) { reject(signal.reason); return; }
160
+ const timer = setTimeout(resolve, ms);
161
+ signal?.addEventListener("abort", () => { clearTimeout(timer); reject(signal.reason); }, { once: true });
162
+ });
163
+
164
+ // SPEC §2 closed set. The four canonical values pass through; known per-backend
165
+ // synonyms translate INTO them (anthropic max_tokens/end_turn, gemini MAX_TOKENS/
166
+ // SAFETY/RECITATION) so a token-cap hit canonicalizes to "length" whatever the
167
+ // backend names it -- core's `finishReason === "length"` truncation check (#425)
168
+ // is then an invariant by construction, not a convention each backend must
169
+ // independently honor. A non-empty value outside both the set and the table
170
+ // collapses to null AND warns once, so a new backend's unmapped cap string
171
+ // surfaces instead of silently becoming "no signal" (which would make core miss
172
+ // the truncation entirely). Case-folded: gemini shouts its reasons.
173
+ const FINISH_SYNONYMS = new Map<string, Exclude<FinishReason, null>>([
174
+ ["stop", "stop"], ["length", "length"], ["tool_calls", "tool_calls"], ["content_filter", "content_filter"],
175
+ ["max_tokens", "length"], ["model_length", "length"], ["max_completion_tokens", "length"],
176
+ ["end_turn", "stop"], ["stop_sequence", "stop"], ["eos_token", "stop"],
177
+ ["tool_use", "tool_calls"],
178
+ ["safety", "content_filter"], ["recitation", "content_filter"],
179
+ ]);
180
+ const normalizeFinishReason = (raw: string | null): FinishReason => {
181
+ if (raw === null || raw.length === 0) return null;
182
+ const hit = FINISH_SYNONYMS.get(raw.toLowerCase());
183
+ if (hit !== undefined) return hit;
184
+ emitWarningOnce(
185
+ `unrecognized finish_reason "${raw}"; treated as no-signal (finishReason=null). If it denotes a token-cap hit, core's length-cap detection will miss it -- add it to FINISH_SYNONYMS.`,
186
+ "PLURNK_FINISH_REASON_UNKNOWN",
187
+ );
188
+ return null;
189
+ };
190
+
191
+ // Shared budget→effort breakpoints (xai and google had identical copies).
192
+ export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
193
+ if (budget <= 1000) return "low";
194
+ if (budget <= 4000) return "medium";
195
+ return "high";
196
+ };
197
+
198
+ // chars/2 upper bound (see ./tokenizers.ts) — overcounts safely, never under.
199
+ const heuristicTokens = (text: string): number => (text.length === 0 ? 0 : Math.ceil(text.length / 2));
200
+
201
+ // Body keys the provider owns — a caller's `sampling` passthrough may not set
202
+ // these. Two families (#477 audit):
203
+ // transport/managed — grammar transport, the stream/JSON choice, slot pinning,
204
+ // data capture (SPEC §8: backend-specific fields never cross the contract);
205
+ // contract invariants — `n` (atomic single completion: choices[0] is the
206
+ // response; n>1 = paid, dropped output), the tool-calling family (tools-in-
207
+ // body doctrine, §2: native tool_calls return null content = a broken turn),
208
+ // modalities/audio (text-only contract), prediction (decode semantics, not
209
+ // sampling), and the token caps (the envelope is the managed maxTokens —
210
+ // sampling must not bypass the consumer's #425 cap).
211
+ // Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
212
+ // platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
213
+ // metadata, store, verbosity) pass through; the managed floors spread UNDER
214
+ // sampling stay deliberately caller-overridable.
215
+ const RESERVED_BODY_KEYS: ReadonlySet<string> = new Set([
216
+ "model", "messages", "stream", "stream_options", "grammar", "response_format", "id_slot", "logprobs", "top_logprobs",
217
+ "n", "tools", "tool_choice", "functions", "function_call", "parallel_tool_calls",
218
+ "modalities", "audio", "prediction", "max_tokens", "max_completion_tokens",
219
+ "prompt_cache_key",
220
+ ]);
221
+
222
+ // Render a non-accept verdict into a terse, factual grammar_unenforced message
223
+ // (SPEC §12 message policy: no guidance prose). `reject` names the diverging code
224
+ // point + what the grammar would have accepted; `incomplete` names the valid-prefix
225
+ // length that never reached a terminal state.
226
+ const describeUnenforced = (v: Exclude<Verdict, { status: "accept" }>): string => {
227
+ if (v.status === "reject") {
228
+ const expected = v.expected.length > 0
229
+ ? v.expected.map((e) => `${e.rule} accepts ${e.accepts}`).join(", ")
230
+ : "end of input";
231
+ return `grammar not enforced: output rejected by the transported grammar at code point ${v.pos} (${JSON.stringify(v.char)}); expected ${expected}`;
232
+ }
233
+ return `grammar not enforced: output is an incomplete match of the transported grammar — a valid prefix of ${v.pos} code points that never terminated`;
234
+ };
235
+
236
+ export default class OpenAICompatProvider implements Provider {
237
+ #model: string;
238
+ #url: string;
239
+ #fetchTimeoutMs: number;
240
+ #headers: Record<string, string>;
241
+ #hasApiKey = false;
242
+ #apiKeyRejectedMessage: string | undefined;
243
+ #eosText: string | undefined;
244
+ #contextWindow: number | null;
245
+ #reasoning: Reasoning;
246
+ #temperature: number;
247
+ #repeatPenalty: number;
248
+ #frequencyPenalty: number;
249
+ #dryMultiplier: number | undefined;
250
+ #dryBase: number | undefined;
251
+ #dryAllowedLength: number | undefined;
252
+ #repeatLastN: number | undefined;
253
+ #retryDelayMs: number;
254
+ #reasoningStyle: ReasoningStyle;
255
+ #countTokens: (text: string) => number;
256
+ #costFor: (usage: ProviderUsage) => number;
257
+ #source: string;
258
+ #grammarStyle: GrammarStyle;
259
+ #promptCacheKey: boolean;
260
+ #gbnfDebug: boolean;
261
+ #streaming: boolean;
262
+ #firstPartyMetadata: boolean;
263
+ #balanceMetaKey: string | undefined;
264
+ #supportsSlotPinning: boolean;
265
+ #slotCount: number | null;
266
+ #retryAttempts: number;
267
+ #topLogprobs: number | null;
268
+ #reasoningReserve: ReserveSpec | undefined;
269
+ #completionReserve: ReserveSpec | undefined;
270
+ #tuningFloors: boolean;
271
+ #rawBody: boolean;
272
+ #servedModel: string | undefined;
273
+ #requiresMaxTokens: boolean | undefined;
274
+
275
+ // Optional capability (SPEC §2): exact tokenization served by the backend's
276
+ // own vocab. Assigned in the constructor ONLY when the config carries a
277
+ // tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
278
+ // the honest capability signal for every other backend.
279
+ tokenize?: (text: string) => Promise<number[]>;
280
+
281
+ constructor(config: OpenAICompatConfig) {
282
+ this.#model = config.model;
283
+ this.#url = config.url;
284
+ this.#fetchTimeoutMs = config.fetchTimeoutMs;
285
+ this.#headers = config.headers ?? {};
286
+ this.#contextWindow = config.contextWindow ?? null;
287
+ this.#reasoning = config.reasoning;
288
+ // Loud guard: an out-of-date consumer (stale plugin dist) omitting the
289
+ // required tuning fields must fail at construction, not silently send
290
+ // undefined sampling on every grammar request.
291
+ if (typeof config.temperature !== "number" || typeof config.repeatPenalty !== "number" || typeof config.retryDelayMs !== "number") {
292
+ throw new Error(`${config.source ?? "provider"}: OpenAICompatConfig requires temperature + repeatPenalty + retryDelayMs (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY / _RETRY_DELAY) — rebuild against providers >= 0.33.0`);
293
+ }
294
+ this.#temperature = config.temperature;
295
+ this.#repeatPenalty = config.repeatPenalty;
296
+ this.#frequencyPenalty = typeof config.frequencyPenalty === "number" ? config.frequencyPenalty : 0;
297
+ this.#dryMultiplier = config.dryMultiplier;
298
+ this.#dryBase = config.dryBase;
299
+ this.#dryAllowedLength = config.dryAllowedLength;
300
+ this.#repeatLastN = config.repeatLastN;
301
+ this.#retryDelayMs = config.retryDelayMs;
302
+ this.#retryAttempts = config.retryAttempts;
303
+ this.#reasoningStyle = config.reasoningStyle ?? "none";
304
+ this.#countTokens = config.countTokens ?? heuristicTokens;
305
+ this.#costFor = config.costFor ?? (() => 0);
306
+ this.#source = config.source ?? "provider";
307
+ this.#grammarStyle = config.grammarStyle ?? "none";
308
+ this.#promptCacheKey = config.promptCacheKey ?? false;
309
+ this.#gbnfDebug = config.gbnfDebug ?? false;
310
+ this.#streaming = config.streaming ?? true;
311
+ this.#firstPartyMetadata = config.firstPartyMetadata ?? false;
312
+ this.#apiKeyRejectedMessage = config.apiKeyRejectedMessage;
313
+ this.#eosText = config.eosText;
314
+ this.#hasApiKey = "Authorization" in this.#headers;
315
+ this.#balanceMetaKey = config.balanceMetaKey;
316
+ this.#supportsSlotPinning = config.supportsSlotPinning ?? false;
317
+ this.#slotCount = config.slotCount ?? null;
318
+ this.#topLogprobs = config.topLogprobs ?? null;
319
+ this.#reasoningReserve = config.reasoningReserve;
320
+ this.#completionReserve = config.completionReserve;
321
+ this.#tuningFloors = config.tuningFloors ?? true;
322
+ this.#rawBody = config.rawBody ?? false;
323
+ this.#servedModel = config.servedModel;
324
+ this.#requiresMaxTokens = config.requiresMaxTokens;
325
+ const { tokenizeUrl } = config;
326
+ if (tokenizeUrl !== undefined) {
327
+ this.tokenize = async (text: string): Promise<number[]> => {
328
+ const res = await fetch(tokenizeUrl, {
329
+ method: "POST",
330
+ headers: { "Content-Type": "application/json", ...this.#headers },
331
+ body: JSON.stringify({ content: text }),
332
+ signal: AbortSignal.timeout(this.#fetchTimeoutMs),
333
+ });
334
+ if (!res.ok) throw new Error(`${this.#source}: tokenize endpoint returned ${res.status}`);
335
+ const { tokens } = (await res.json()) as { tokens?: unknown };
336
+ if (!Array.isArray(tokens) || !tokens.every((t) => typeof t === "number")) {
337
+ throw new Error(`${this.#source}: tokenize endpoint returned no token array`);
338
+ }
339
+ return tokens;
340
+ };
341
+ }
342
+ }
343
+
344
+ get contextWindow(): number | null { return this.#contextWindow; }
345
+ // #507: envelope reserves — absolute pins stand alone; percentages need the
346
+ // detected window; null = underivable (no claim for core's no-cap path).
347
+ #resolveReserve(spec: ReserveSpec | undefined): number | null {
348
+ if (spec === undefined) return null;
349
+ if ("tokens" in spec) return spec.tokens;
350
+ return this.#contextWindow === null ? null : Math.round(spec.percent * this.#contextWindow);
351
+ }
352
+ get reasoningReserve(): number | null { return this.#resolveReserve(this.#reasoningReserve); }
353
+ get completionReserve(): number | null { return this.#resolveReserve(this.#completionReserve); }
354
+ get model(): string { return this.#model; }
355
+ // #37: backend's self-reported served id; undefined when unprobed/unknown.
356
+ get servedModel(): string | undefined { return this.#servedModel; }
357
+ // #43: resolved "decodes unbounded without a cap" fact; undefined = no claim.
358
+ get requiresMaxTokens(): boolean | undefined { return this.#requiresMaxTokens; }
359
+ // Resolved capability (#34): will a transported grammar actually constrain
360
+ // this backend's decode? Introspectable so a consumer can verify the rails
361
+ // are LIVE without spending a generation on a forcing-grammar probe.
362
+ get constrainsOutput(): boolean { return this.#grammarStyle !== "none"; }
363
+
364
+ countTokens(text: string): number { return this.#countTokens(text); }
365
+ costFor(usage: ProviderUsage): number { return this.#costFor(usage); }
366
+
367
+ // Maps the reasoning INTENT (off | adaptive | on+budget, #33) to the
368
+ // backend's wire mechanism — including under a transported grammar. The #32
369
+ // clamp (force reasoning_effort "none" under response_format) is LIFTED:
370
+ // canary-verified live that fireworks masks ONLY the content channel — the
371
+ // reasoning channel rides beside it unmasked, and the plurnk grammar's
372
+ // reasoning?/preplan regions absorb any in-band spillover. The old measured
373
+ // failures (low→cycles, high→spirals) were pre-max_tokens-cap; with a bounded
374
+ // cap the matrix ACCEPTs across efforts (reasoning-rails matrix
375
+ // F9). Clamping was the root of the plan-less regression (service#331).
376
+ #reasoningBody(): Record<string, unknown> {
377
+ const { mode, budget } = this.#reasoning;
378
+ const on = mode !== "off";
379
+ switch (this.#reasoningStyle) {
380
+ // Native-channel styles. "template" ALWAYS emits — the explicit
381
+ // enable_thinking:false is the only working off-switch on llama-server
382
+ // (§13). Activation only; budget is enforced by the box's
383
+ // --reasoning-budget launch flag (per-request numerics ignored, F7).
384
+ //
385
+ // #488 postmortem: intent maps IDENTICALLY under a transported
386
+ // grammar. The brief rails-win-the-channel clamp (enable_thinking
387
+ // forced false under a grammar) is REVERTED — specimens proved the
388
+ // SANCTIONED think block is the protection, not the hazard: the
389
+ // server auto-gates the grammar around it and content decodes
390
+ // constrained (26-run baseline green; zero grammar rejects across
391
+ // the #488 "railless" specimens). Closing the channel starved a
392
+ // reasoning-tuned model into ESCAPING mid-content into the raw
393
+ // thought channel — discarded server-side, decode unconstrained,
394
+ // 12,288 tokens billed for 1,033 visible chars. The escape is
395
+ // surfaced instead (vanished-token telemetry + meta rail state).
396
+ case "template": return { chat_template_kwargs: { enable_thinking: on } };
397
+ case "think": return on ? { think: true } : {};
398
+ case "include_reasoning": return on ? { include_reasoning: true } : {};
399
+ // effort tiers from the budget; off/adaptive omit the field (the
400
+ // API's default depth is its adaptive).
401
+ case "effort": return mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
402
+ // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
403
+ // reason-by-default model (DeepSeek V4: default 'high') reasoning (#30).
404
+ // ADAPTIVE omits the field: the backend's own default posture IS the
405
+ // adaptive semantics, and the literal "adaptive" is MiniMax-M3-only —
406
+ // fireworks 400s it for every other model (wire-verified, #403; the
407
+ // 1.0.2 adaptive default refused to boot on it). V4 gotcha: integer
408
+ // efforts 400.
409
+ case "effort_explicit": return mode === "off"
410
+ ? { reasoning_effort: "none" }
411
+ : mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
412
+ // Anthropic compat: explicit thinking object. off → disabled; on →
413
+ // enabled with budget_tokens; adaptive → omit (the API default).
414
+ case "anthropic": return mode === "off"
415
+ ? { thinking: { type: "disabled" } }
416
+ : mode === "on" ? { thinking: { type: "enabled", budget_tokens: budget } } : {};
417
+ case "none": return {};
418
+ }
419
+ }
420
+
421
+ // Per-run slot affinity (#11): the consumer passes WHICH run this is; the
422
+ // provider owns WHICH slot serves it. Sticky per workerId, round-robin across
423
+ // new runs (distinct runs → distinct slots while slots last), LRU-bounded
424
+ // bookkeeping so a long-lived daemon never grows the map unboundedly —
425
+ // an evicted-and-returning run simply re-pins, worst case one cold prefill.
426
+ #runSlots = new Map<string, number>();
427
+ #nextSlot = 0;
428
+
429
+ #slotBody(workerId: string): Record<string, unknown> {
430
+ if (!this.#supportsSlotPinning || this.#slotCount === null || this.#slotCount < 1) return {};
431
+ let slot = this.#runSlots.get(workerId);
432
+ if (slot === undefined) {
433
+ slot = this.#nextSlot++ % this.#slotCount;
434
+ if (this.#runSlots.size >= this.#slotCount * 8) {
435
+ this.#runSlots.delete(this.#runSlots.keys().next().value as string);
436
+ }
437
+ } else {
438
+ this.#runSlots.delete(workerId); // re-insert to refresh LRU recency
439
+ }
440
+ this.#runSlots.set(workerId, slot);
441
+ return { id_slot: slot };
442
+ }
443
+
444
+ // Grammar transport (SPEC §13): carry the caller-supplied GBNF in the shape
445
+ // the backend accepts. Same grammar, different wire field per backend; an
446
+ // unsupported/unknown backend sends NO field at all (cloud APIs 400 on
447
+ // unknowns, and a silent send would let a constrained consumer mistake
448
+ // unconstrained output for enforced).
449
+ #grammarBody(grammar: string | undefined): Record<string, unknown> {
450
+ if (grammar === undefined) return {};
451
+ switch (this.#grammarStyle) {
452
+ // Greedy decoding under hard constraint loops without a repeat-penalty
453
+ // floor (#9, SPEC §13) — every grammar path carries it. llama.cpp spells
454
+ // it `repeat_penalty`; the OpenAI-compat (Fireworks) shape is `repetition_penalty`
455
+ // (verified honored live, #20).
456
+ case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
457
+ case "response_format": return { response_format: { type: "grammar", grammar }, repetition_penalty: this.#repeatPenalty };
458
+ case "none": return {};
459
+ }
460
+ }
461
+
462
+ // Anti-degeneration DEFAULT on EVERY request (#426), keyed to the backend's wire
463
+ // convention - NOT grammar-bound. GBNF is a local rail (off for cloud), so a cloud
464
+ // alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
465
+ // straight to the token cap on pure looped repetition (run52). Ships next to
466
+ // temperature so caller `sampling` can tune it; the grammar path re-asserts it as a
467
+ // managed FLOOR in #grammarBody. Per backend: llama.cpp/response_format take the
468
+ // repeat_penalty MULTIPLIER; the plain cloud path ("none") can't, so it gets
469
+ // frequency_penalty - OpenAI-standard, accepted by every OpenAI-compat backend (verified
470
+ // live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
471
+ #repetitionPenaltyBody(): Record<string, unknown> {
472
+ switch (this.#grammarStyle) {
473
+ // #567: repeat_penalty + optional DRY (repeated-SEQUENCE penalty) + a wider
474
+ // repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
475
+ // Each rides only when its operator knob is set; absent = the box's default.
476
+ case "llamacpp": return {
477
+ repeat_penalty: this.#repeatPenalty,
478
+ ...(this.#repeatLastN !== undefined ? { repeat_last_n: this.#repeatLastN } : {}),
479
+ ...(this.#dryMultiplier !== undefined && this.#dryMultiplier > 0 ? {
480
+ dry_multiplier: this.#dryMultiplier,
481
+ ...(this.#dryBase !== undefined ? { dry_base: this.#dryBase } : {}),
482
+ ...(this.#dryAllowedLength !== undefined ? { dry_allowed_length: this.#dryAllowedLength } : {}),
483
+ } : {}),
484
+ };
485
+ case "response_format": return { repetition_penalty: this.#repeatPenalty };
486
+ case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
487
+ }
488
+ }
489
+
490
+ // First-party telemetry headers (SPEC §5): forwarded ONLY when the spec
491
+ // opted in (the plurnk endpoint). The gate is here, not at the call site, so
492
+ // attributions/client/strikes can never reach a third-party backend even if
493
+ // the consumer passes them to the wrong provider. Empty values emit no header
494
+ // — EXCEPT strikes, where 0 is a real value (clean streak) distinct from
495
+ // absent (consumer didn't report); contract per plurnk-service#313. Strikes
496
+ // ride HTTP headers only — the packet never carries them (the model must
497
+ // never see strike state; engine accounting is not a metric to game).
498
+ #metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined): Record<string, string> {
499
+ if (!this.#firstPartyMetadata) return {};
500
+ const h: Record<string, string> = {};
501
+ if (attributions !== undefined && attributions.length > 0) h["Plurnk-Attribution"] = JSON.stringify(attributions);
502
+ if (client !== undefined && client.length > 0) h["Plurnk-Client"] = client;
503
+ if (strikes !== undefined && Number.isInteger(strikes) && strikes >= 0) h["Plurnk-Strikes"] = String(strikes);
504
+ // Worker identity (#26, wire-name completed #486/#511): the opaque workerId
505
+ // the consumer already supplies, forwarded so the endpoint can key
506
+ // per-worker affinity/telemetry — same gate as every first-party signal.
507
+ h["Plurnk-Worker-Id"] = workerId;
508
+ // Root worker of the lineage (#522): the no-parent ancestor of this turn's
509
+ // worker tree. The consumer classifies primary-vs-spawned by equality
510
+ // (primaryWorkerId == workerId ⇒ the primary/root worker). The provider
511
+ // EMITS what the consumer supplies and never invents a primary; the
512
+ // consumer's contract is to stamp it EVERY turn (including the primary's
513
+ // own, where it equals workerId). Absence is the consumer's violation for
514
+ // the endpoint to surface, not a provider default.
515
+ if (primaryWorkerId !== undefined && primaryWorkerId.length > 0) h["Plurnk-Worker-Primary"] = primaryWorkerId;
516
+ // Turn coordinate (#404, extends #26 per #391): workspace/loop/turn, the
517
+ // daemon-side sequence the endpoint can never scrape from the wire.
518
+ // Coordinates are 1-based — 0 is not a real value, so no strikes-style
519
+ // zero exception; absent/empty/0 emits no header.
520
+ if (workspaceId !== undefined && workspaceId.length > 0) h["Plurnk-Workspace-Id"] = workspaceId;
521
+ if (loop !== undefined && Number.isInteger(loop) && loop >= 1) h["Plurnk-Loop"] = String(loop);
522
+ if (turn !== undefined && Number.isInteger(turn) && turn >= 1) h["Plurnk-Turn"] = String(turn);
523
+ return h;
524
+ }
525
+
526
+ // Enforcement verification (SPEC §13). When a grammar was actually transported
527
+ // (grammarStyle !== "none"), the backend MUST have constrained the output;
528
+ // some silently drop the grammar field or mislabel the channel, and without
529
+ // this check we would return unconstrained output as if enforced. STRICT: any
530
+ // non-accept verdict (reject, or an incomplete/never-terminated match) is a
531
+ // grammar_unenforced failure. A grammar our own validator can't parse — even
532
+ // though the backend accepted it (a port-vs-llama.cpp gap) — is a non-fatal
533
+ // verify gap: warn, don't fail a transport that may have worked. This is a
534
+ // conformance check against the grammar we already hold, NOT a plurnk-DSL
535
+ // parse (§8) — it stays grammar-generic and backend-agnostic.
536
+ // Validate output against the grammar. Returns the verdict, or null on the
537
+ // verify GAP — a grammar our own validator can't parse (a port-vs-llama.cpp
538
+ // gap): warn, don't manufacture a conflict from a check that didn't run.
539
+ #grammarVerdict(grammar: string, content: string): Verdict | null {
540
+ try {
541
+ return validateGbnf(grammar, content);
542
+ } catch (cause) {
543
+ // Once per (code, message) — #40: this fires PER TURN otherwise.
544
+ emitWarningOnce(
545
+ `${this.#source}: could not verify grammar enforcement — the transported grammar did not parse in @plurnk/gbnf (${(cause as Error).message})`,
546
+ "PLURNK_GRAMMAR_UNVERIFIABLE",
547
+ );
548
+ return null;
549
+ }
550
+ }
551
+
552
+ // PLURNK_PROVIDERS_GBNF_DEBUG (SPEC §13): validate the supplied GBNF locally and fail
553
+ // hard if it's malformed, BEFORE any wire call — and the grammar is NOT
554
+ // transported, so the request runs unconstrained. A debug aid to catch invalid
555
+ // grammars (e.g. while editing the plurnk grammar) without a model round-trip;
556
+ // off in production. `validateGbnf(grammar, "")` parses the grammar + resolves
557
+ // its root, throwing iff the grammar itself is invalid (the empty input's
558
+ // verdict is irrelevant — we only care that parsing succeeded).
559
+ #assertGrammarValid(grammar: string): void {
560
+ try {
561
+ validateGbnf(grammar, "");
562
+ } catch (cause) {
563
+ throw new Error(`grammar validation (PLURNK_PROVIDERS_GBNF_DEBUG): invalid GBNF — ${(cause as Error).message}`, { cause });
564
+ }
565
+ }
566
+
567
+ // Per-turn metadata bag (#23): pass the backend's non-standard top-level fields
568
+ // (the transport's `chunkMetadata`) through VERBATIM, then normalize the known
569
+ // keys we hold a contract for — the spec's balance field → a validated
570
+ // `balancePico` (finite pico-USD; dropped if non-numeric), renamed off its raw
571
+ // key so the consumer reads one canonical name. Undefined when nothing's there;
572
+ // the service merges this into its Turn metadata and filters what reaches clients.
573
+ #buildMeta(chunkMetadata: Record<string, unknown>): Record<string, unknown> | undefined {
574
+ const meta: Record<string, unknown> = { ...chunkMetadata };
575
+ if (this.#balanceMetaKey !== undefined) {
576
+ const raw = meta[this.#balanceMetaKey];
577
+ delete meta[this.#balanceMetaKey];
578
+ if (typeof raw === "number" && Number.isFinite(raw)) meta.balancePico = raw;
579
+ }
580
+ return Object.keys(meta).length > 0 ? meta : undefined;
581
+ }
582
+
583
+ // Caller-supplied OpenAI-compat sampling params (temperature, top_p, top_k,
584
+ // penalties, stop, seed, …) merged UNDER the managed body: model, messages,
585
+ // reasoning, grammar (+ its repeat-penalty floor), max_tokens and slot always
586
+ // win, and reserved transport/protocol keys are stripped so the passthrough
587
+ // can't smuggle a grammar, a stream toggle, or a backend slot (SPEC §8).
588
+ #samplingBody(sampling: Record<string, unknown> | undefined): Record<string, unknown> {
589
+ if (sampling === undefined) return {};
590
+ const out: Record<string, unknown> = {};
591
+ for (const [k, v] of Object.entries(sampling)) if (!RESERVED_BODY_KEYS.has(k)) out[k] = v;
592
+ return out;
593
+ }
594
+
595
+ async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse> {
596
+ // Boundary validation (SPEC §2): the worker identity is required.
597
+ if (workerId === undefined || workerId.length === 0) throw new Error("generate: workerId is required — the worker's stable, opaque identity");
598
+ // Reject before any wire call when already aborted (SPEC §10.8).
599
+ signal?.throwIfAborted();
600
+
601
+ // Grammar handling (SPEC §13). PLURNK_PROVIDERS_GBNF_DEBUG validates the supplied
602
+ // grammar locally and throws on a malformed one, then WITHHOLDS it so the
603
+ // model generates UNCONSTRAINED — and the free output is still verified
604
+ // against the grammar (below), surfacing exactly where the model's natural
605
+ // output and the grammar conflict. Otherwise the grammar is sent when the
606
+ // backend supports it (grammarStyle !== "none").
607
+ const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
608
+ if (wantGrammar && this.#gbnfDebug) this.#assertGrammarValid(grammar!);
609
+ const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
610
+
611
+ // Assembly order = precedence: the family's sampling DEFAULTS
612
+ // (PLURNK_PROVIDERS_TEMPERATURE — universal, #30 measured it on grammar
613
+ // paths and the name promises every request) < the caller's `sampling`
614
+ // < the managed fields, which always win.
615
+ const body: Record<string, unknown> = {
616
+ // #507: floors suppressed on router-owned-tuning providers (plurnk) —
617
+ // the router's per-model tuning must not be overridden by client floors.
618
+ ...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
619
+ ...this.#samplingBody(sampling),
620
+ model: this.#model,
621
+ messages,
622
+ ...this.#reasoningBody(),
623
+ ...this.#grammarBody(sendGrammar),
624
+ ...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
625
+ // #36: request per-token logprobs only when enabled (managed field —
626
+ // reserved from caller sampling; the env flag is the single control).
627
+ ...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
628
+ ...this.#slotBody(workerId),
629
+ // #518: prompt-cache affinity -- workerId as the OpenAI-standard
630
+ // prompt_cache_key routes a worker's turns to one serverless replica so
631
+ // its stable prefix caches (managed; reserved from caller sampling).
632
+ ...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
633
+ };
634
+
635
+ // Transient-failure retry (#18). Each attempt gets a FRESH fetch timeout
636
+ // (the budget is per-request, not shared across retries); the caller's
637
+ // signal spans them all. Retry only the transient classifications, prefer
638
+ // a server Retry-After over the backoff, and let the caller's abort cut
639
+ // through both the in-flight request and the backoff sleep.
640
+ // Stream by default, but fall back to one non-streamed JSON for the one
641
+ // case it breaks: a response_format grammar (fireworks) streams its
642
+ // constrained output mislabeled as reasoning_content, yet returns it as
643
+ // content non-streamed. The atomic dump is correct either way, so the
644
+ // demotion is scoped to exactly that request, not the whole provider.
645
+ const grammarBreaksStream = sendGrammar !== undefined && this.#grammarStyle === "response_format";
646
+ const transport = this.#streaming && !grammarBreaksStream ? chatCompletionStream : chatCompletion;
647
+
648
+ // Per-request headers = static auth/routing + any first-party telemetry.
649
+ const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
650
+ const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
651
+ let raw;
652
+ for (let attempt = 0; ; attempt++) {
653
+ const timeoutSignal = AbortSignal.timeout(this.#fetchTimeoutMs);
654
+ const effectiveSignal = signal !== undefined ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal;
655
+ try {
656
+ raw = await transport({ url: this.#url, headers, body, signal: effectiveSignal, captureRawBody: this.#rawBody });
657
+ break;
658
+ } catch (err) {
659
+ // Caller-initiated abort is cancellation — never retried or wrapped.
660
+ if (signal?.aborted) throw err;
661
+ const { kind } = classifyProviderError(err);
662
+ // #543: Cloudflare/CDN edge codes (520-527) classify as network_failure
663
+ // but fail-fast - a retry just re-incurs the same origin/edge timeout.
664
+ const edgeTimeout = err instanceof OpenAiHttpError && isEdgeStatus(err.status);
665
+ // Terminal kind, edge failure, or budget spent -> surface the failure.
666
+ if (!RETRYABLE.has(kind) || edgeTimeout || attempt >= this.#retryAttempts) {
667
+ const pe = toProviderError(err, this.#source);
668
+ // #537 case 2: a 401/403 with a key PRESENT is a rejected key, not a
669
+ // transport failure — surface the distinct, actionable hint (kind stays
670
+ // "unauthorized", so core's routing is unchanged) rather than the raw
671
+ // upstream JSON. Distinct from the unset-key throw at construction.
672
+ if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
673
+ throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
674
+ }
675
+ throw pe;
676
+ }
677
+ const retryAfter = err instanceof OpenAiHttpError ? err.retryAfter : null;
678
+ await sleepWithAbort(retryAfter ?? this.#retryDelayMs * 2 ** attempt, signal);
679
+ }
680
+ }
681
+
682
+ // #539: llama-server --special renders EOG tokens as text, so a turn ending
683
+ // via raw EOS carries a trailing <eos> the grammar never sanctioned - it both
684
+ // false-rejects the rail verdict and leaks a control token into the packet.
685
+ // Strip the server-reported eos_token from the tail ONCE, before the verdict
686
+ // grades it and before it reaches assistant/packet. rawBody keeps the verbatim
687
+ // wire text for forensics.
688
+ if (this.#eosText !== undefined) raw.content = stripTrailingSpecial(raw.content, this.#eosText);
689
+
690
+ // Grammar conformance (§13): bytes always flow; the verdict is an
691
+ // observation. Same check whether the grammar was transported
692
+ // (sendGrammar) or withheld (PLURNK_PROVIDERS_GBNF_DEBUG filter mode) — a
693
+ // non-accept verdict attaches a grammar_unenforced telemetry event
694
+ // (message + divergence position) and the response returns normally.
695
+ // Discard/retry/escalate/self-correct is the consumer's policy.
696
+ let telemetry: TelemetryEvent[] | undefined;
697
+ let railsMeta: Record<string, unknown> | undefined;
698
+ const usage = normalizeUsage(raw.usage, raw.reasoning_content, raw.content);
699
+ const observedGrammar = sendGrammar ?? (wantGrammar && this.#gbnfDebug ? grammar : undefined);
700
+ if (observedGrammar !== undefined) {
701
+ const verdict = this.#grammarVerdict(observedGrammar, raw.content);
702
+ if (verdict !== null && verdict.status !== "accept") {
703
+ telemetry = [{ source: this.#source, kind: "grammar_unenforced", message: describeUnenforced(verdict), position: verdict.pos }];
704
+ }
705
+ // #488 per-request loud state: rail attachment + conformance verdict
706
+ // ride `meta` into the consumer's turn row, so a drill reads rail
707
+ // presence PER TURN from the run db instead of inferring it from
708
+ // output shape (the #488 misdiagnosis, twice).
709
+ railsMeta = { railsAttached: sendGrammar !== undefined, railsVerdict: verdict?.status ?? "unverifiable" };
710
+ // #488 channel-escape detector (the run105 class): completion tokens
711
+ // billed far beyond every visible channel mean the decode ESCAPED into
712
+ // a server-discarded reasoning block mid-emission — unconstrained,
713
+ // invisible, billed (12,288 billed vs 1,033 chars visible, live).
714
+ // countTokens OVERCOUNTS text (chars/2 upper bound), so billed
715
+ // exceeding visible-plus-slack is real vanishing, not estimator noise.
716
+ const visible = this.#countTokens(raw.content) + this.#countTokens(raw.reasoning_content);
717
+ if (sendGrammar !== undefined && usage.completion > visible + 64) {
718
+ (telemetry ??= []).push({
719
+ source: this.#source,
720
+ kind: "grammar_unenforced",
721
+ message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ~${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
722
+ position: [...raw.content].length,
723
+ });
724
+ }
725
+ }
726
+
727
+ const builtMeta = this.#buildMeta(raw.chunkMetadata);
728
+ const meta = railsMeta !== undefined ? { ...builtMeta, ...railsMeta } : builtMeta;
729
+
730
+ // #36: surface per-token logprobs + their mean when the backend returned
731
+ // them (only possible when the flag requested them). Absent otherwise —
732
+ // never synthesized.
733
+ const logprobs = raw.logprobs !== null && raw.logprobs.length > 0 ? raw.logprobs : undefined;
734
+ const meanLogprob = logprobs !== undefined
735
+ ? logprobs.reduce((sum, t) => sum + t.logprob, 0) / logprobs.length
736
+ : undefined;
737
+
738
+ return {
739
+ assistant: {
740
+ content: raw.content,
741
+ reasoning: raw.reasoning_content.length > 0 ? raw.reasoning_content : null,
742
+ // #482: sealed relay blobs ride only when present — same absence
743
+ // discipline as logprobs (never synthesized, never empty-array).
744
+ ...(raw.reasoning_encrypted.length > 0 ? { reasoningEncrypted: raw.reasoning_encrypted } : {}),
745
+ usage,
746
+ finishReason: normalizeFinishReason(raw.finish_reason),
747
+ model: raw.model ?? this.#model,
748
+ ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
749
+ },
750
+ assistantRaw: raw,
751
+ ...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
752
+ ...(meta !== undefined ? { meta } : {}),
753
+ ...(telemetry !== undefined ? { telemetry } : {}),
754
+ };
755
+ }
756
+ }