@plurnk/plurnk-providers 1.2.0 → 1.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -60
- package/SPEC.md +6 -6
- package/dist/OpenAICompat.js +2 -2
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/ProviderRegistry.js +2 -2
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/env.js +3 -3
- package/dist/env.js.map +1 -1
- package/dist/openaiStream.d.ts.map +1 -1
- package/dist/openaiStream.js +12 -0
- package/dist/openaiStream.js.map +1 -1
- package/package.json +7 -6
- package/src/Mock.test.ts +142 -0
- package/src/Mock.ts +95 -0
- package/src/OpenAICompat.test.ts +1107 -0
- package/src/OpenAICompat.ts +756 -0
- package/src/Pool.test.ts +155 -0
- package/src/Pool.ts +134 -0
- package/src/ProviderRegistry.test.ts +176 -0
- package/src/ProviderRegistry.ts +93 -0
- package/src/boundaries.test.ts +24 -0
- package/src/discover.test.ts +123 -0
- package/src/discover.ts +112 -0
- package/src/env.test.ts +190 -0
- package/src/env.ts +211 -0
- package/src/index.ts +51 -0
- package/src/lexicon-guard.test.ts +58 -0
- package/src/openaiStream.ts +279 -0
- package/src/standardProviders.test.ts +925 -0
- package/src/standardProviders.ts +618 -0
- package/src/telemetry.test.ts +62 -0
- package/src/telemetry.ts +108 -0
- package/src/types.ts +219 -0
- package/src/usage.test.ts +136 -0
- package/src/usage.ts +82 -0
- package/src/warnings.test.ts +31 -0
- package/src/warnings.ts +0 -0
|
@@ -0,0 +1,756 @@
|
|
|
1
|
+
// Shared OpenAI-compatible provider. Implements the universal generate()
|
|
2
|
+
// spine — signal merging, the SSE call, usage mapping, finishReason
|
|
3
|
+
// normalization, response assembly — that every sibling had duplicated.
|
|
4
|
+
//
|
|
5
|
+
// Composition, not inheritance: the per-provider deltas (resolved URL, auth
|
|
6
|
+
// headers, reasoning translation style, tokenizer, cost) arrive as config.
|
|
7
|
+
// A sibling's fromEnv probes whatever it needs (catalog, pricing, context
|
|
8
|
+
// window), builds the config, and returns `new OpenAICompatProvider(config)`.
|
|
9
|
+
// Pure-config providers come from ./standardProviders.ts with no sibling at all.
|
|
10
|
+
|
|
11
|
+
import type { ChatMessage, FinishReason, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
12
|
+
import type { Reasoning, ReserveSpec } from "./env.ts";
|
|
13
|
+
import { chatCompletionStream, chatCompletion, OpenAiHttpError, isEdgeStatus, type StreamResponse } from "./openaiStream.ts";
|
|
14
|
+
import { normalizeUsage } from "./usage.ts";
|
|
15
|
+
import { toProviderError, classifyProviderError, ProviderError, type TelemetryEvent } from "./telemetry.ts";
|
|
16
|
+
import { validateGbnf, type Verdict } from "@plurnk/gbnf";
|
|
17
|
+
import { emitWarningOnce } from "./warnings.ts";
|
|
18
|
+
|
|
19
|
+
// How the reasoning intent (PLURNK_PROVIDERS_REASONING: off | adaptive | on, plus
|
|
20
|
+
// REASONING_BUDGET iff on — #32/#33) translates to each backend's wire mechanism
|
|
21
|
+
// (SPEC §4); the per-style mapping lives in #reasoningBody. Non-obvious ones:
|
|
22
|
+
// "template" ALWAYS emits enable_thinking — the explicit false is llama-server's
|
|
23
|
+
// only working off-switch (§13); "anthropic" uses the `thinking` object and IGNORES
|
|
24
|
+
// reasoning_effort; "effort_explicit" (fireworks) sends the EXPLICIT "none" for OFF
|
|
25
|
+
// instead of omitting — reason-by-DEFAULT models (DeepSeek V4 defaults
|
|
26
|
+
// 'high') keep reasoning when the field is omitted, fatal under an active grammar
|
|
27
|
+
// (#30). Intent maps IDENTICALLY with or without a transported grammar — fireworks
|
|
28
|
+
// masks only the content channel, so reasoning and rails coexist in one call
|
|
29
|
+
// (canary-verified; the #32 clamp is lifted — it caused the plan-less service#331).
|
|
30
|
+
export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "template" | "anthropic";
|
|
31
|
+
|
|
32
|
+
// How a caller-supplied GBNF grammar is carried on the wire — backends accept
|
|
33
|
+
// different shapes for the SAME GBNF (probed/configured, never guessed; §13); the
|
|
34
|
+
// wire shape per style lives in #grammarBody. "none" means the grammar is NOT sent
|
|
35
|
+
// (never silently — so a constrained consumer can't mistake unconstrained output
|
|
36
|
+
// for enforced).
|
|
37
|
+
export type GrammarStyle = "none" | "llamacpp" | "response_format";
|
|
38
|
+
|
|
39
|
+
export type OpenAICompatConfig = {
|
|
40
|
+
model: string;
|
|
41
|
+
url: string; // fully-resolved chat-completions URL
|
|
42
|
+
fetchTimeoutMs: number;
|
|
43
|
+
headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
|
|
44
|
+
contextWindow?: number | null; // default null; caller resolves-or-fails (#419), narrows to required with the interface
|
|
45
|
+
reasoningStyle?: ReasoningStyle; // default "none"
|
|
46
|
+
countTokens?: (text: string) => number; // default chars/2 upper-bound heuristic
|
|
47
|
+
costFor?: (usage: ProviderUsage) => number; // default () => 0
|
|
48
|
+
source?: string; // telemetry source, e.g. "provider:openai"; default "provider"
|
|
49
|
+
grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
|
|
50
|
+
// #518: send the OpenAI-standard `prompt_cache_key` set to workerId, so a
|
|
51
|
+
// serverless backend with REPLICA-LOCAL prompt caching (fireworks, verified)
|
|
52
|
+
// pins a worker's turns to one replica and claims its stable prefix. Default
|
|
53
|
+
// false -- a backend that strict-validates unknown fields 400s, so enable only
|
|
54
|
+
// where the field is accepted. Same identity that already drives slot affinity.
|
|
55
|
+
promptCacheKey?: boolean;
|
|
56
|
+
gbnfDebug?: boolean; // PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar locally + throw on invalid, but DON'T transport it (run unconstrained); default false
|
|
57
|
+
streaming?: boolean; // SSE transport (default true); false → one non-streamed JSON
|
|
58
|
+
firstPartyMetadata?: boolean; // forward per-turn attributions + client as Plurnk-* headers (plurnk only); default false
|
|
59
|
+
apiKeyRejectedMessage?: string; // #537: friendly hint when a PRESENT key is 401/403-rejected (distinct from unset); default undefined
|
|
60
|
+
eosText?: string; // #539: server-reported eos_token, stripped from the content tail (--special renders it as text); default undefined
|
|
61
|
+
balanceMetaKey?: string; // top-level response field carrying account balance (pico-USD) → validated meta.balancePico (plurnk only, #23); default unset
|
|
62
|
+
// Slot affinity wiring (provider-INTERNAL — never consumer-facing, #11).
|
|
63
|
+
supportsSlotPinning?: boolean; // backend accepts an `id_slot` body field (llama-server); default false
|
|
64
|
+
slotCount?: number | null; // probed slot count for pinning backends; default null
|
|
65
|
+
// Backend-served exact tokenization (llama-server /tokenize). When set, the
|
|
66
|
+
// provider exposes the optional `tokenize()` capability — the model's OWN
|
|
67
|
+
// vocab, no client-side tokenizer data needed; default unset (capability absent).
|
|
68
|
+
tokenizeUrl?: string;
|
|
69
|
+
// #37: the backend's self-reported served model id (from the /v1/models probe),
|
|
70
|
+
// surfaced as Provider.servedModel. For a local llama-server the wire `model` is
|
|
71
|
+
// the alias; this is the real name (the .gguf) the tokenizer seam maps. Absent
|
|
72
|
+
// when no probe ran or it read no row.
|
|
73
|
+
servedModel?: string;
|
|
74
|
+
// #43: backend decodes unbounded without a caller cap (llama-server n_predict
|
|
75
|
+
// to the wall) — surfaced as Provider.requiresMaxTokens so consumers can
|
|
76
|
+
// boot-refuse an envelope-less local alias. Default unset (no claim).
|
|
77
|
+
requiresMaxTokens?: boolean;
|
|
78
|
+
// The side-channel reasoning intent — REQUIRED, no in-code default
|
|
79
|
+
// (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
|
|
80
|
+
// { mode: off|adaptive|on, budget: iff on }. The provider maps it to the
|
|
81
|
+
// backend's mechanism via reasoningStyle; budget is only ever a magnitude,
|
|
82
|
+
// never a hidden activation flag (#33).
|
|
83
|
+
reasoning: Reasoning;
|
|
84
|
+
// Decode tuning: no in-code defaults; the canonical measured values (0.2 /
|
|
85
|
+
// 1.15) ship as the floor in .env.defaults (alias-scopable). `temperature` is the
|
|
86
|
+
// DEFAULT for EVERY request, spread UNDER caller sampling (#30/endpoint#7).
|
|
87
|
+
// `repeatPenalty` is the FLOOR the provider manages wherever a grammar rides
|
|
88
|
+
// (greedy-under-mask loops without it, #9) — the VALUE is operator config;
|
|
89
|
+
// WHERE it applies stays mechanism. `retryDelayMs` is the transient-retry
|
|
90
|
+
// backoff base (attempt N waits retryDelayMs * 2^(N-1); Retry-After wins).
|
|
91
|
+
temperature: number;
|
|
92
|
+
repeatPenalty: number;
|
|
93
|
+
// #426: anti-degeneration guard on the CLOUD path (grammarStyle "none"), where the
|
|
94
|
+
// repeat_penalty multiplier isn't available - the OpenAI-standard frequency_penalty.
|
|
95
|
+
// Optional (default 0 = off) so an out-of-date plugin that omits it just runs unguarded
|
|
96
|
+
// rather than failing construction; the standard factory always supplies it.
|
|
97
|
+
frequencyPenalty?: number;
|
|
98
|
+
// #567: llama.cpp anti-repetition-LOOP controls, sent on the llamacpp path (DRY is a
|
|
99
|
+
// llama.cpp sampler). DRY penalizes repeated SEQUENCES with a penalty escalating in run
|
|
100
|
+
// length — the tool for a plan-restart loop a single-token repeat_penalty over a short
|
|
101
|
+
// window can't see. The GENERIC DRY defaults (0.8/1.75/2) ship as a floor in .env.defaults
|
|
102
|
+
// like repeatPenalty — applied for a detected llama.cpp backend, customer-overridable;
|
|
103
|
+
// absent (a plugin omitting them) = the box's own default. repeatLastN widens the window.
|
|
104
|
+
dryMultiplier?: number;
|
|
105
|
+
dryBase?: number;
|
|
106
|
+
dryAllowedLength?: number;
|
|
107
|
+
repeatLastN?: number;
|
|
108
|
+
retryDelayMs: number;
|
|
109
|
+
// Transient-failure retry budget — REQUIRED, no in-code default
|
|
110
|
+
// (PLURNK_PROVIDERS_RETRY_ATTEMPTS, a non-negative int): 0 = surface the
|
|
111
|
+
// first failure; N = up to N retries on a transient error (§4, #18).
|
|
112
|
+
retryAttempts: number;
|
|
113
|
+
// Data-capture knobs (#36), OFF by default — the flag IS the isolation, so a
|
|
114
|
+
// serving turn requests nothing and carries nothing. `topLogprobs`: when a
|
|
115
|
+
// non-negative int, request `logprobs:true, top_logprobs:<n>` and surface the
|
|
116
|
+
// per-token confidence on assistant.logprobs (PLURNK_PROVIDERS_TOP_LOGPROBS;
|
|
117
|
+
// null = off). `rawBody`: when true, attach the verbatim wire body to
|
|
118
|
+
// response.rawBody (PLURNK_PROVIDERS_RAWBODY). Both universal — any backend,
|
|
119
|
+
// gated per-alias.
|
|
120
|
+
topLogprobs?: number | null;
|
|
121
|
+
rawBody?: boolean;
|
|
122
|
+
// #507 (owner-ruled): the generation-envelope reserves, env-read via
|
|
123
|
+
// envelopeFromEnv — a percentage of the DETECTED window or an absolute token
|
|
124
|
+
// count. Optional so an out-of-date sibling keeps constructing (no claim);
|
|
125
|
+
// the standard factory always supplies them. Resolved against contextWindow
|
|
126
|
+
// at read time (getters), so a probe that lands after config assembly still
|
|
127
|
+
// derives correctly.
|
|
128
|
+
reasoningReserve?: ReserveSpec;
|
|
129
|
+
completionReserve?: ReserveSpec;
|
|
130
|
+
// #507: the plurnk.ai router owns tuning (SPEC §5) — false suppresses the
|
|
131
|
+
// client-side temperature/penalty FLOORS on this provider (caller `sampling`
|
|
132
|
+
// still passes through verbatim). Default true (floors ride).
|
|
133
|
+
tuningFloors?: boolean;
|
|
134
|
+
};
|
|
135
|
+
|
|
136
|
+
// Transient classifications worth retrying: rate_limit (429) and network_failure
|
|
137
|
+
// (5xx, timeout, connection reset) are transport; grammar_invalid (a 422 output
|
|
138
|
+
// reject a fresh sample may satisfy, #548) rides the same bounded budget.
|
|
139
|
+
// unauthorized, quota_exceeded, invalid_response, model_refused are terminal —
|
|
140
|
+
// retrying just burns time and budget.
|
|
141
|
+
const RETRYABLE: ReadonlySet<string> = new Set(["rate_limit", "network_failure", "grammar_invalid"]);
|
|
142
|
+
|
|
143
|
+
// #539: drop trailing occurrences of a server-rendered EOG marker. llama-server
|
|
144
|
+
// under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
|
|
145
|
+
// trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
|
|
146
|
+
// can never eat body content (a body ending in the literal marker isn't producible
|
|
147
|
+
// under the grammar, and is vanishingly rare unconstrained).
|
|
148
|
+
const stripTrailingSpecial = (content: string, marker: string): string => {
|
|
149
|
+
if (marker.length === 0) return content;
|
|
150
|
+
let out = content;
|
|
151
|
+
while (out.endsWith(marker)) out = out.slice(0, -marker.length);
|
|
152
|
+
return out;
|
|
153
|
+
};
|
|
154
|
+
|
|
155
|
+
// Sleep that rejects the moment `signal` aborts (caller cancellation must not
|
|
156
|
+
// wait out a backoff). Resolves normally on timeout.
|
|
157
|
+
const sleepWithAbort = (ms: number, signal: AbortSignal | undefined): Promise<void> =>
|
|
158
|
+
new Promise((resolve, reject) => {
|
|
159
|
+
if (signal?.aborted) { reject(signal.reason); return; }
|
|
160
|
+
const timer = setTimeout(resolve, ms);
|
|
161
|
+
signal?.addEventListener("abort", () => { clearTimeout(timer); reject(signal.reason); }, { once: true });
|
|
162
|
+
});
|
|
163
|
+
|
|
164
|
+
// SPEC §2 closed set. The four canonical values pass through; known per-backend
|
|
165
|
+
// synonyms translate INTO them (anthropic max_tokens/end_turn, gemini MAX_TOKENS/
|
|
166
|
+
// SAFETY/RECITATION) so a token-cap hit canonicalizes to "length" whatever the
|
|
167
|
+
// backend names it -- core's `finishReason === "length"` truncation check (#425)
|
|
168
|
+
// is then an invariant by construction, not a convention each backend must
|
|
169
|
+
// independently honor. A non-empty value outside both the set and the table
|
|
170
|
+
// collapses to null AND warns once, so a new backend's unmapped cap string
|
|
171
|
+
// surfaces instead of silently becoming "no signal" (which would make core miss
|
|
172
|
+
// the truncation entirely). Case-folded: gemini shouts its reasons.
|
|
173
|
+
const FINISH_SYNONYMS = new Map<string, Exclude<FinishReason, null>>([
|
|
174
|
+
["stop", "stop"], ["length", "length"], ["tool_calls", "tool_calls"], ["content_filter", "content_filter"],
|
|
175
|
+
["max_tokens", "length"], ["model_length", "length"], ["max_completion_tokens", "length"],
|
|
176
|
+
["end_turn", "stop"], ["stop_sequence", "stop"], ["eos_token", "stop"],
|
|
177
|
+
["tool_use", "tool_calls"],
|
|
178
|
+
["safety", "content_filter"], ["recitation", "content_filter"],
|
|
179
|
+
]);
|
|
180
|
+
const normalizeFinishReason = (raw: string | null): FinishReason => {
|
|
181
|
+
if (raw === null || raw.length === 0) return null;
|
|
182
|
+
const hit = FINISH_SYNONYMS.get(raw.toLowerCase());
|
|
183
|
+
if (hit !== undefined) return hit;
|
|
184
|
+
emitWarningOnce(
|
|
185
|
+
`unrecognized finish_reason "${raw}"; treated as no-signal (finishReason=null). If it denotes a token-cap hit, core's length-cap detection will miss it -- add it to FINISH_SYNONYMS.`,
|
|
186
|
+
"PLURNK_FINISH_REASON_UNKNOWN",
|
|
187
|
+
);
|
|
188
|
+
return null;
|
|
189
|
+
};
|
|
190
|
+
|
|
191
|
+
// Shared budget→effort breakpoints (xai and google had identical copies).
|
|
192
|
+
export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
|
|
193
|
+
if (budget <= 1000) return "low";
|
|
194
|
+
if (budget <= 4000) return "medium";
|
|
195
|
+
return "high";
|
|
196
|
+
};
|
|
197
|
+
|
|
198
|
+
// chars/2 upper bound (see ./tokenizers.ts) — overcounts safely, never under.
|
|
199
|
+
const heuristicTokens = (text: string): number => (text.length === 0 ? 0 : Math.ceil(text.length / 2));
|
|
200
|
+
|
|
201
|
+
// Body keys the provider owns — a caller's `sampling` passthrough may not set
|
|
202
|
+
// these. Two families (#477 audit):
|
|
203
|
+
// transport/managed — grammar transport, the stream/JSON choice, slot pinning,
|
|
204
|
+
// data capture (SPEC §8: backend-specific fields never cross the contract);
|
|
205
|
+
// contract invariants — `n` (atomic single completion: choices[0] is the
|
|
206
|
+
// response; n>1 = paid, dropped output), the tool-calling family (tools-in-
|
|
207
|
+
// body doctrine, §2: native tool_calls return null content = a broken turn),
|
|
208
|
+
// modalities/audio (text-only contract), prediction (decode semantics, not
|
|
209
|
+
// sampling), and the token caps (the envelope is the managed maxTokens —
|
|
210
|
+
// sampling must not bypass the consumer's #425 cap).
|
|
211
|
+
// Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
|
|
212
|
+
// platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
|
|
213
|
+
// metadata, store, verbosity) pass through; the managed floors spread UNDER
|
|
214
|
+
// sampling stay deliberately caller-overridable.
|
|
215
|
+
const RESERVED_BODY_KEYS: ReadonlySet<string> = new Set([
|
|
216
|
+
"model", "messages", "stream", "stream_options", "grammar", "response_format", "id_slot", "logprobs", "top_logprobs",
|
|
217
|
+
"n", "tools", "tool_choice", "functions", "function_call", "parallel_tool_calls",
|
|
218
|
+
"modalities", "audio", "prediction", "max_tokens", "max_completion_tokens",
|
|
219
|
+
"prompt_cache_key",
|
|
220
|
+
]);
|
|
221
|
+
|
|
222
|
+
// Render a non-accept verdict into a terse, factual grammar_unenforced message
|
|
223
|
+
// (SPEC §12 message policy: no guidance prose). `reject` names the diverging code
|
|
224
|
+
// point + what the grammar would have accepted; `incomplete` names the valid-prefix
|
|
225
|
+
// length that never reached a terminal state.
|
|
226
|
+
const describeUnenforced = (v: Exclude<Verdict, { status: "accept" }>): string => {
|
|
227
|
+
if (v.status === "reject") {
|
|
228
|
+
const expected = v.expected.length > 0
|
|
229
|
+
? v.expected.map((e) => `${e.rule} accepts ${e.accepts}`).join(", ")
|
|
230
|
+
: "end of input";
|
|
231
|
+
return `grammar not enforced: output rejected by the transported grammar at code point ${v.pos} (${JSON.stringify(v.char)}); expected ${expected}`;
|
|
232
|
+
}
|
|
233
|
+
return `grammar not enforced: output is an incomplete match of the transported grammar — a valid prefix of ${v.pos} code points that never terminated`;
|
|
234
|
+
};
|
|
235
|
+
|
|
236
|
+
export default class OpenAICompatProvider implements Provider {
|
|
237
|
+
#model: string;
|
|
238
|
+
#url: string;
|
|
239
|
+
#fetchTimeoutMs: number;
|
|
240
|
+
#headers: Record<string, string>;
|
|
241
|
+
#hasApiKey = false;
|
|
242
|
+
#apiKeyRejectedMessage: string | undefined;
|
|
243
|
+
#eosText: string | undefined;
|
|
244
|
+
#contextWindow: number | null;
|
|
245
|
+
#reasoning: Reasoning;
|
|
246
|
+
#temperature: number;
|
|
247
|
+
#repeatPenalty: number;
|
|
248
|
+
#frequencyPenalty: number;
|
|
249
|
+
#dryMultiplier: number | undefined;
|
|
250
|
+
#dryBase: number | undefined;
|
|
251
|
+
#dryAllowedLength: number | undefined;
|
|
252
|
+
#repeatLastN: number | undefined;
|
|
253
|
+
#retryDelayMs: number;
|
|
254
|
+
#reasoningStyle: ReasoningStyle;
|
|
255
|
+
#countTokens: (text: string) => number;
|
|
256
|
+
#costFor: (usage: ProviderUsage) => number;
|
|
257
|
+
#source: string;
|
|
258
|
+
#grammarStyle: GrammarStyle;
|
|
259
|
+
#promptCacheKey: boolean;
|
|
260
|
+
#gbnfDebug: boolean;
|
|
261
|
+
#streaming: boolean;
|
|
262
|
+
#firstPartyMetadata: boolean;
|
|
263
|
+
#balanceMetaKey: string | undefined;
|
|
264
|
+
#supportsSlotPinning: boolean;
|
|
265
|
+
#slotCount: number | null;
|
|
266
|
+
#retryAttempts: number;
|
|
267
|
+
#topLogprobs: number | null;
|
|
268
|
+
#reasoningReserve: ReserveSpec | undefined;
|
|
269
|
+
#completionReserve: ReserveSpec | undefined;
|
|
270
|
+
#tuningFloors: boolean;
|
|
271
|
+
#rawBody: boolean;
|
|
272
|
+
#servedModel: string | undefined;
|
|
273
|
+
#requiresMaxTokens: boolean | undefined;
|
|
274
|
+
|
|
275
|
+
// Optional capability (SPEC §2): exact tokenization served by the backend's
|
|
276
|
+
// own vocab. Assigned in the constructor ONLY when the config carries a
|
|
277
|
+
// tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
|
|
278
|
+
// the honest capability signal for every other backend.
|
|
279
|
+
tokenize?: (text: string) => Promise<number[]>;
|
|
280
|
+
|
|
281
|
+
constructor(config: OpenAICompatConfig) {
|
|
282
|
+
this.#model = config.model;
|
|
283
|
+
this.#url = config.url;
|
|
284
|
+
this.#fetchTimeoutMs = config.fetchTimeoutMs;
|
|
285
|
+
this.#headers = config.headers ?? {};
|
|
286
|
+
this.#contextWindow = config.contextWindow ?? null;
|
|
287
|
+
this.#reasoning = config.reasoning;
|
|
288
|
+
// Loud guard: an out-of-date consumer (stale plugin dist) omitting the
|
|
289
|
+
// required tuning fields must fail at construction, not silently send
|
|
290
|
+
// undefined sampling on every grammar request.
|
|
291
|
+
if (typeof config.temperature !== "number" || typeof config.repeatPenalty !== "number" || typeof config.retryDelayMs !== "number") {
|
|
292
|
+
throw new Error(`${config.source ?? "provider"}: OpenAICompatConfig requires temperature + repeatPenalty + retryDelayMs (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY / _RETRY_DELAY) — rebuild against providers >= 0.33.0`);
|
|
293
|
+
}
|
|
294
|
+
this.#temperature = config.temperature;
|
|
295
|
+
this.#repeatPenalty = config.repeatPenalty;
|
|
296
|
+
this.#frequencyPenalty = typeof config.frequencyPenalty === "number" ? config.frequencyPenalty : 0;
|
|
297
|
+
this.#dryMultiplier = config.dryMultiplier;
|
|
298
|
+
this.#dryBase = config.dryBase;
|
|
299
|
+
this.#dryAllowedLength = config.dryAllowedLength;
|
|
300
|
+
this.#repeatLastN = config.repeatLastN;
|
|
301
|
+
this.#retryDelayMs = config.retryDelayMs;
|
|
302
|
+
this.#retryAttempts = config.retryAttempts;
|
|
303
|
+
this.#reasoningStyle = config.reasoningStyle ?? "none";
|
|
304
|
+
this.#countTokens = config.countTokens ?? heuristicTokens;
|
|
305
|
+
this.#costFor = config.costFor ?? (() => 0);
|
|
306
|
+
this.#source = config.source ?? "provider";
|
|
307
|
+
this.#grammarStyle = config.grammarStyle ?? "none";
|
|
308
|
+
this.#promptCacheKey = config.promptCacheKey ?? false;
|
|
309
|
+
this.#gbnfDebug = config.gbnfDebug ?? false;
|
|
310
|
+
this.#streaming = config.streaming ?? true;
|
|
311
|
+
this.#firstPartyMetadata = config.firstPartyMetadata ?? false;
|
|
312
|
+
this.#apiKeyRejectedMessage = config.apiKeyRejectedMessage;
|
|
313
|
+
this.#eosText = config.eosText;
|
|
314
|
+
this.#hasApiKey = "Authorization" in this.#headers;
|
|
315
|
+
this.#balanceMetaKey = config.balanceMetaKey;
|
|
316
|
+
this.#supportsSlotPinning = config.supportsSlotPinning ?? false;
|
|
317
|
+
this.#slotCount = config.slotCount ?? null;
|
|
318
|
+
this.#topLogprobs = config.topLogprobs ?? null;
|
|
319
|
+
this.#reasoningReserve = config.reasoningReserve;
|
|
320
|
+
this.#completionReserve = config.completionReserve;
|
|
321
|
+
this.#tuningFloors = config.tuningFloors ?? true;
|
|
322
|
+
this.#rawBody = config.rawBody ?? false;
|
|
323
|
+
this.#servedModel = config.servedModel;
|
|
324
|
+
this.#requiresMaxTokens = config.requiresMaxTokens;
|
|
325
|
+
const { tokenizeUrl } = config;
|
|
326
|
+
if (tokenizeUrl !== undefined) {
|
|
327
|
+
this.tokenize = async (text: string): Promise<number[]> => {
|
|
328
|
+
const res = await fetch(tokenizeUrl, {
|
|
329
|
+
method: "POST",
|
|
330
|
+
headers: { "Content-Type": "application/json", ...this.#headers },
|
|
331
|
+
body: JSON.stringify({ content: text }),
|
|
332
|
+
signal: AbortSignal.timeout(this.#fetchTimeoutMs),
|
|
333
|
+
});
|
|
334
|
+
if (!res.ok) throw new Error(`${this.#source}: tokenize endpoint returned ${res.status}`);
|
|
335
|
+
const { tokens } = (await res.json()) as { tokens?: unknown };
|
|
336
|
+
if (!Array.isArray(tokens) || !tokens.every((t) => typeof t === "number")) {
|
|
337
|
+
throw new Error(`${this.#source}: tokenize endpoint returned no token array`);
|
|
338
|
+
}
|
|
339
|
+
return tokens;
|
|
340
|
+
};
|
|
341
|
+
}
|
|
342
|
+
}
|
|
343
|
+
|
|
344
|
+
get contextWindow(): number | null { return this.#contextWindow; }
|
|
345
|
+
// #507: envelope reserves — absolute pins stand alone; percentages need the
|
|
346
|
+
// detected window; null = underivable (no claim for core's no-cap path).
|
|
347
|
+
#resolveReserve(spec: ReserveSpec | undefined): number | null {
|
|
348
|
+
if (spec === undefined) return null;
|
|
349
|
+
if ("tokens" in spec) return spec.tokens;
|
|
350
|
+
return this.#contextWindow === null ? null : Math.round(spec.percent * this.#contextWindow);
|
|
351
|
+
}
|
|
352
|
+
get reasoningReserve(): number | null { return this.#resolveReserve(this.#reasoningReserve); }
|
|
353
|
+
get completionReserve(): number | null { return this.#resolveReserve(this.#completionReserve); }
|
|
354
|
+
get model(): string { return this.#model; }
|
|
355
|
+
// #37: backend's self-reported served id; undefined when unprobed/unknown.
|
|
356
|
+
get servedModel(): string | undefined { return this.#servedModel; }
|
|
357
|
+
// #43: resolved "decodes unbounded without a cap" fact; undefined = no claim.
|
|
358
|
+
get requiresMaxTokens(): boolean | undefined { return this.#requiresMaxTokens; }
|
|
359
|
+
// Resolved capability (#34): will a transported grammar actually constrain
|
|
360
|
+
// this backend's decode? Introspectable so a consumer can verify the rails
|
|
361
|
+
// are LIVE without spending a generation on a forcing-grammar probe.
|
|
362
|
+
get constrainsOutput(): boolean { return this.#grammarStyle !== "none"; }
|
|
363
|
+
|
|
364
|
+
countTokens(text: string): number { return this.#countTokens(text); }
|
|
365
|
+
costFor(usage: ProviderUsage): number { return this.#costFor(usage); }
|
|
366
|
+
|
|
367
|
+
// Maps the reasoning INTENT (off | adaptive | on+budget, #33) to the
|
|
368
|
+
// backend's wire mechanism — including under a transported grammar. The #32
|
|
369
|
+
// clamp (force reasoning_effort "none" under response_format) is LIFTED:
|
|
370
|
+
// canary-verified live that fireworks masks ONLY the content channel — the
|
|
371
|
+
// reasoning channel rides beside it unmasked, and the plurnk grammar's
|
|
372
|
+
// reasoning?/preplan regions absorb any in-band spillover. The old measured
|
|
373
|
+
// failures (low→cycles, high→spirals) were pre-max_tokens-cap; with a bounded
|
|
374
|
+
// cap the matrix ACCEPTs across efforts (reasoning-rails matrix
|
|
375
|
+
// F9). Clamping was the root of the plan-less regression (service#331).
|
|
376
|
+
#reasoningBody(): Record<string, unknown> {
|
|
377
|
+
const { mode, budget } = this.#reasoning;
|
|
378
|
+
const on = mode !== "off";
|
|
379
|
+
switch (this.#reasoningStyle) {
|
|
380
|
+
// Native-channel styles. "template" ALWAYS emits — the explicit
|
|
381
|
+
// enable_thinking:false is the only working off-switch on llama-server
|
|
382
|
+
// (§13). Activation only; budget is enforced by the box's
|
|
383
|
+
// --reasoning-budget launch flag (per-request numerics ignored, F7).
|
|
384
|
+
//
|
|
385
|
+
// #488 postmortem: intent maps IDENTICALLY under a transported
|
|
386
|
+
// grammar. The brief rails-win-the-channel clamp (enable_thinking
|
|
387
|
+
// forced false under a grammar) is REVERTED — specimens proved the
|
|
388
|
+
// SANCTIONED think block is the protection, not the hazard: the
|
|
389
|
+
// server auto-gates the grammar around it and content decodes
|
|
390
|
+
// constrained (26-run baseline green; zero grammar rejects across
|
|
391
|
+
// the #488 "railless" specimens). Closing the channel starved a
|
|
392
|
+
// reasoning-tuned model into ESCAPING mid-content into the raw
|
|
393
|
+
// thought channel — discarded server-side, decode unconstrained,
|
|
394
|
+
// 12,288 tokens billed for 1,033 visible chars. The escape is
|
|
395
|
+
// surfaced instead (vanished-token telemetry + meta rail state).
|
|
396
|
+
case "template": return { chat_template_kwargs: { enable_thinking: on } };
|
|
397
|
+
case "think": return on ? { think: true } : {};
|
|
398
|
+
case "include_reasoning": return on ? { include_reasoning: true } : {};
|
|
399
|
+
// effort tiers from the budget; off/adaptive omit the field (the
|
|
400
|
+
// API's default depth is its adaptive).
|
|
401
|
+
case "effort": return mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
|
|
402
|
+
// Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
|
|
403
|
+
// reason-by-default model (DeepSeek V4: default 'high') reasoning (#30).
|
|
404
|
+
// ADAPTIVE omits the field: the backend's own default posture IS the
|
|
405
|
+
// adaptive semantics, and the literal "adaptive" is MiniMax-M3-only —
|
|
406
|
+
// fireworks 400s it for every other model (wire-verified, #403; the
|
|
407
|
+
// 1.0.2 adaptive default refused to boot on it). V4 gotcha: integer
|
|
408
|
+
// efforts 400.
|
|
409
|
+
case "effort_explicit": return mode === "off"
|
|
410
|
+
? { reasoning_effort: "none" }
|
|
411
|
+
: mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
|
|
412
|
+
// Anthropic compat: explicit thinking object. off → disabled; on →
|
|
413
|
+
// enabled with budget_tokens; adaptive → omit (the API default).
|
|
414
|
+
case "anthropic": return mode === "off"
|
|
415
|
+
? { thinking: { type: "disabled" } }
|
|
416
|
+
: mode === "on" ? { thinking: { type: "enabled", budget_tokens: budget } } : {};
|
|
417
|
+
case "none": return {};
|
|
418
|
+
}
|
|
419
|
+
}
|
|
420
|
+
|
|
421
|
+
// Per-run slot affinity (#11): the consumer passes WHICH run this is; the
|
|
422
|
+
// provider owns WHICH slot serves it. Sticky per workerId, round-robin across
|
|
423
|
+
// new runs (distinct runs → distinct slots while slots last), LRU-bounded
|
|
424
|
+
// bookkeeping so a long-lived daemon never grows the map unboundedly —
|
|
425
|
+
// an evicted-and-returning run simply re-pins, worst case one cold prefill.
|
|
426
|
+
#runSlots = new Map<string, number>();
|
|
427
|
+
#nextSlot = 0;
|
|
428
|
+
|
|
429
|
+
#slotBody(workerId: string): Record<string, unknown> {
|
|
430
|
+
if (!this.#supportsSlotPinning || this.#slotCount === null || this.#slotCount < 1) return {};
|
|
431
|
+
let slot = this.#runSlots.get(workerId);
|
|
432
|
+
if (slot === undefined) {
|
|
433
|
+
slot = this.#nextSlot++ % this.#slotCount;
|
|
434
|
+
if (this.#runSlots.size >= this.#slotCount * 8) {
|
|
435
|
+
this.#runSlots.delete(this.#runSlots.keys().next().value as string);
|
|
436
|
+
}
|
|
437
|
+
} else {
|
|
438
|
+
this.#runSlots.delete(workerId); // re-insert to refresh LRU recency
|
|
439
|
+
}
|
|
440
|
+
this.#runSlots.set(workerId, slot);
|
|
441
|
+
return { id_slot: slot };
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
// Grammar transport (SPEC §13): carry the caller-supplied GBNF in the shape
|
|
445
|
+
// the backend accepts. Same grammar, different wire field per backend; an
|
|
446
|
+
// unsupported/unknown backend sends NO field at all (cloud APIs 400 on
|
|
447
|
+
// unknowns, and a silent send would let a constrained consumer mistake
|
|
448
|
+
// unconstrained output for enforced).
|
|
449
|
+
#grammarBody(grammar: string | undefined): Record<string, unknown> {
|
|
450
|
+
if (grammar === undefined) return {};
|
|
451
|
+
switch (this.#grammarStyle) {
|
|
452
|
+
// Greedy decoding under hard constraint loops without a repeat-penalty
|
|
453
|
+
// floor (#9, SPEC §13) — every grammar path carries it. llama.cpp spells
|
|
454
|
+
// it `repeat_penalty`; the OpenAI-compat (Fireworks) shape is `repetition_penalty`
|
|
455
|
+
// (verified honored live, #20).
|
|
456
|
+
case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
|
|
457
|
+
case "response_format": return { response_format: { type: "grammar", grammar }, repetition_penalty: this.#repeatPenalty };
|
|
458
|
+
case "none": return {};
|
|
459
|
+
}
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
// Anti-degeneration DEFAULT on EVERY request (#426), keyed to the backend's wire
|
|
463
|
+
// convention - NOT grammar-bound. GBNF is a local rail (off for cloud), so a cloud
|
|
464
|
+
// alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
|
|
465
|
+
// straight to the token cap on pure looped repetition (run52). Ships next to
|
|
466
|
+
// temperature so caller `sampling` can tune it; the grammar path re-asserts it as a
|
|
467
|
+
// managed FLOOR in #grammarBody. Per backend: llama.cpp/response_format take the
|
|
468
|
+
// repeat_penalty MULTIPLIER; the plain cloud path ("none") can't, so it gets
|
|
469
|
+
// frequency_penalty - OpenAI-standard, accepted by every OpenAI-compat backend (verified
|
|
470
|
+
// live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
|
|
471
|
+
#repetitionPenaltyBody(): Record<string, unknown> {
|
|
472
|
+
switch (this.#grammarStyle) {
|
|
473
|
+
// #567: repeat_penalty + optional DRY (repeated-SEQUENCE penalty) + a wider
|
|
474
|
+
// repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
|
|
475
|
+
// Each rides only when its operator knob is set; absent = the box's default.
|
|
476
|
+
case "llamacpp": return {
|
|
477
|
+
repeat_penalty: this.#repeatPenalty,
|
|
478
|
+
...(this.#repeatLastN !== undefined ? { repeat_last_n: this.#repeatLastN } : {}),
|
|
479
|
+
...(this.#dryMultiplier !== undefined && this.#dryMultiplier > 0 ? {
|
|
480
|
+
dry_multiplier: this.#dryMultiplier,
|
|
481
|
+
...(this.#dryBase !== undefined ? { dry_base: this.#dryBase } : {}),
|
|
482
|
+
...(this.#dryAllowedLength !== undefined ? { dry_allowed_length: this.#dryAllowedLength } : {}),
|
|
483
|
+
} : {}),
|
|
484
|
+
};
|
|
485
|
+
case "response_format": return { repetition_penalty: this.#repeatPenalty };
|
|
486
|
+
case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
|
|
487
|
+
}
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
// First-party telemetry headers (SPEC §5): forwarded ONLY when the spec
|
|
491
|
+
// opted in (the plurnk endpoint). The gate is here, not at the call site, so
|
|
492
|
+
// attributions/client/strikes can never reach a third-party backend even if
|
|
493
|
+
// the consumer passes them to the wrong provider. Empty values emit no header
|
|
494
|
+
// — EXCEPT strikes, where 0 is a real value (clean streak) distinct from
|
|
495
|
+
// absent (consumer didn't report); contract per plurnk-service#313. Strikes
|
|
496
|
+
// ride HTTP headers only — the packet never carries them (the model must
|
|
497
|
+
// never see strike state; engine accounting is not a metric to game).
|
|
498
|
+
#metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined): Record<string, string> {
|
|
499
|
+
if (!this.#firstPartyMetadata) return {};
|
|
500
|
+
const h: Record<string, string> = {};
|
|
501
|
+
if (attributions !== undefined && attributions.length > 0) h["Plurnk-Attribution"] = JSON.stringify(attributions);
|
|
502
|
+
if (client !== undefined && client.length > 0) h["Plurnk-Client"] = client;
|
|
503
|
+
if (strikes !== undefined && Number.isInteger(strikes) && strikes >= 0) h["Plurnk-Strikes"] = String(strikes);
|
|
504
|
+
// Worker identity (#26, wire-name completed #486/#511): the opaque workerId
|
|
505
|
+
// the consumer already supplies, forwarded so the endpoint can key
|
|
506
|
+
// per-worker affinity/telemetry — same gate as every first-party signal.
|
|
507
|
+
h["Plurnk-Worker-Id"] = workerId;
|
|
508
|
+
// Root worker of the lineage (#522): the no-parent ancestor of this turn's
|
|
509
|
+
// worker tree. The consumer classifies primary-vs-spawned by equality
|
|
510
|
+
// (primaryWorkerId == workerId ⇒ the primary/root worker). The provider
|
|
511
|
+
// EMITS what the consumer supplies and never invents a primary; the
|
|
512
|
+
// consumer's contract is to stamp it EVERY turn (including the primary's
|
|
513
|
+
// own, where it equals workerId). Absence is the consumer's violation for
|
|
514
|
+
// the endpoint to surface, not a provider default.
|
|
515
|
+
if (primaryWorkerId !== undefined && primaryWorkerId.length > 0) h["Plurnk-Worker-Primary"] = primaryWorkerId;
|
|
516
|
+
// Turn coordinate (#404, extends #26 per #391): workspace/loop/turn, the
|
|
517
|
+
// daemon-side sequence the endpoint can never scrape from the wire.
|
|
518
|
+
// Coordinates are 1-based — 0 is not a real value, so no strikes-style
|
|
519
|
+
// zero exception; absent/empty/0 emits no header.
|
|
520
|
+
if (workspaceId !== undefined && workspaceId.length > 0) h["Plurnk-Workspace-Id"] = workspaceId;
|
|
521
|
+
if (loop !== undefined && Number.isInteger(loop) && loop >= 1) h["Plurnk-Loop"] = String(loop);
|
|
522
|
+
if (turn !== undefined && Number.isInteger(turn) && turn >= 1) h["Plurnk-Turn"] = String(turn);
|
|
523
|
+
return h;
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
// Enforcement verification (SPEC §13). When a grammar was actually transported
|
|
527
|
+
// (grammarStyle !== "none"), the backend MUST have constrained the output;
|
|
528
|
+
// some silently drop the grammar field or mislabel the channel, and without
|
|
529
|
+
// this check we would return unconstrained output as if enforced. STRICT: any
|
|
530
|
+
// non-accept verdict (reject, or an incomplete/never-terminated match) is a
|
|
531
|
+
// grammar_unenforced failure. A grammar our own validator can't parse — even
|
|
532
|
+
// though the backend accepted it (a port-vs-llama.cpp gap) — is a non-fatal
|
|
533
|
+
// verify gap: warn, don't fail a transport that may have worked. This is a
|
|
534
|
+
// conformance check against the grammar we already hold, NOT a plurnk-DSL
|
|
535
|
+
// parse (§8) — it stays grammar-generic and backend-agnostic.
|
|
536
|
+
// Validate output against the grammar. Returns the verdict, or null on the
|
|
537
|
+
// verify GAP — a grammar our own validator can't parse (a port-vs-llama.cpp
|
|
538
|
+
// gap): warn, don't manufacture a conflict from a check that didn't run.
|
|
539
|
+
#grammarVerdict(grammar: string, content: string): Verdict | null {
|
|
540
|
+
try {
|
|
541
|
+
return validateGbnf(grammar, content);
|
|
542
|
+
} catch (cause) {
|
|
543
|
+
// Once per (code, message) — #40: this fires PER TURN otherwise.
|
|
544
|
+
emitWarningOnce(
|
|
545
|
+
`${this.#source}: could not verify grammar enforcement — the transported grammar did not parse in @plurnk/gbnf (${(cause as Error).message})`,
|
|
546
|
+
"PLURNK_GRAMMAR_UNVERIFIABLE",
|
|
547
|
+
);
|
|
548
|
+
return null;
|
|
549
|
+
}
|
|
550
|
+
}
|
|
551
|
+
|
|
552
|
+
// PLURNK_PROVIDERS_GBNF_DEBUG (SPEC §13): validate the supplied GBNF locally and fail
|
|
553
|
+
// hard if it's malformed, BEFORE any wire call — and the grammar is NOT
|
|
554
|
+
// transported, so the request runs unconstrained. A debug aid to catch invalid
|
|
555
|
+
// grammars (e.g. while editing the plurnk grammar) without a model round-trip;
|
|
556
|
+
// off in production. `validateGbnf(grammar, "")` parses the grammar + resolves
|
|
557
|
+
// its root, throwing iff the grammar itself is invalid (the empty input's
|
|
558
|
+
// verdict is irrelevant — we only care that parsing succeeded).
|
|
559
|
+
#assertGrammarValid(grammar: string): void {
|
|
560
|
+
try {
|
|
561
|
+
validateGbnf(grammar, "");
|
|
562
|
+
} catch (cause) {
|
|
563
|
+
throw new Error(`grammar validation (PLURNK_PROVIDERS_GBNF_DEBUG): invalid GBNF — ${(cause as Error).message}`, { cause });
|
|
564
|
+
}
|
|
565
|
+
}
|
|
566
|
+
|
|
567
|
+
// Per-turn metadata bag (#23): pass the backend's non-standard top-level fields
|
|
568
|
+
// (the transport's `chunkMetadata`) through VERBATIM, then normalize the known
|
|
569
|
+
// keys we hold a contract for — the spec's balance field → a validated
|
|
570
|
+
// `balancePico` (finite pico-USD; dropped if non-numeric), renamed off its raw
|
|
571
|
+
// key so the consumer reads one canonical name. Undefined when nothing's there;
|
|
572
|
+
// the service merges this into its Turn metadata and filters what reaches clients.
|
|
573
|
+
#buildMeta(chunkMetadata: Record<string, unknown>): Record<string, unknown> | undefined {
|
|
574
|
+
const meta: Record<string, unknown> = { ...chunkMetadata };
|
|
575
|
+
if (this.#balanceMetaKey !== undefined) {
|
|
576
|
+
const raw = meta[this.#balanceMetaKey];
|
|
577
|
+
delete meta[this.#balanceMetaKey];
|
|
578
|
+
if (typeof raw === "number" && Number.isFinite(raw)) meta.balancePico = raw;
|
|
579
|
+
}
|
|
580
|
+
return Object.keys(meta).length > 0 ? meta : undefined;
|
|
581
|
+
}
|
|
582
|
+
|
|
583
|
+
// Caller-supplied OpenAI-compat sampling params (temperature, top_p, top_k,
|
|
584
|
+
// penalties, stop, seed, …) merged UNDER the managed body: model, messages,
|
|
585
|
+
// reasoning, grammar (+ its repeat-penalty floor), max_tokens and slot always
|
|
586
|
+
// win, and reserved transport/protocol keys are stripped so the passthrough
|
|
587
|
+
// can't smuggle a grammar, a stream toggle, or a backend slot (SPEC §8).
|
|
588
|
+
#samplingBody(sampling: Record<string, unknown> | undefined): Record<string, unknown> {
|
|
589
|
+
if (sampling === undefined) return {};
|
|
590
|
+
const out: Record<string, unknown> = {};
|
|
591
|
+
for (const [k, v] of Object.entries(sampling)) if (!RESERVED_BODY_KEYS.has(k)) out[k] = v;
|
|
592
|
+
return out;
|
|
593
|
+
}
|
|
594
|
+
|
|
595
|
+
async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse> {
|
|
596
|
+
// Boundary validation (SPEC §2): the worker identity is required.
|
|
597
|
+
if (workerId === undefined || workerId.length === 0) throw new Error("generate: workerId is required — the worker's stable, opaque identity");
|
|
598
|
+
// Reject before any wire call when already aborted (SPEC §10.8).
|
|
599
|
+
signal?.throwIfAborted();
|
|
600
|
+
|
|
601
|
+
// Grammar handling (SPEC §13). PLURNK_PROVIDERS_GBNF_DEBUG validates the supplied
|
|
602
|
+
// grammar locally and throws on a malformed one, then WITHHOLDS it so the
|
|
603
|
+
// model generates UNCONSTRAINED — and the free output is still verified
|
|
604
|
+
// against the grammar (below), surfacing exactly where the model's natural
|
|
605
|
+
// output and the grammar conflict. Otherwise the grammar is sent when the
|
|
606
|
+
// backend supports it (grammarStyle !== "none").
|
|
607
|
+
const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
|
|
608
|
+
if (wantGrammar && this.#gbnfDebug) this.#assertGrammarValid(grammar!);
|
|
609
|
+
const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
|
|
610
|
+
|
|
611
|
+
// Assembly order = precedence: the family's sampling DEFAULTS
|
|
612
|
+
// (PLURNK_PROVIDERS_TEMPERATURE — universal, #30 measured it on grammar
|
|
613
|
+
// paths and the name promises every request) < the caller's `sampling`
|
|
614
|
+
// < the managed fields, which always win.
|
|
615
|
+
const body: Record<string, unknown> = {
|
|
616
|
+
// #507: floors suppressed on router-owned-tuning providers (plurnk) —
|
|
617
|
+
// the router's per-model tuning must not be overridden by client floors.
|
|
618
|
+
...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
|
|
619
|
+
...this.#samplingBody(sampling),
|
|
620
|
+
model: this.#model,
|
|
621
|
+
messages,
|
|
622
|
+
...this.#reasoningBody(),
|
|
623
|
+
...this.#grammarBody(sendGrammar),
|
|
624
|
+
...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
|
|
625
|
+
// #36: request per-token logprobs only when enabled (managed field —
|
|
626
|
+
// reserved from caller sampling; the env flag is the single control).
|
|
627
|
+
...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
|
|
628
|
+
...this.#slotBody(workerId),
|
|
629
|
+
// #518: prompt-cache affinity -- workerId as the OpenAI-standard
|
|
630
|
+
// prompt_cache_key routes a worker's turns to one serverless replica so
|
|
631
|
+
// its stable prefix caches (managed; reserved from caller sampling).
|
|
632
|
+
...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
|
|
633
|
+
};
|
|
634
|
+
|
|
635
|
+
// Transient-failure retry (#18). Each attempt gets a FRESH fetch timeout
|
|
636
|
+
// (the budget is per-request, not shared across retries); the caller's
|
|
637
|
+
// signal spans them all. Retry only the transient classifications, prefer
|
|
638
|
+
// a server Retry-After over the backoff, and let the caller's abort cut
|
|
639
|
+
// through both the in-flight request and the backoff sleep.
|
|
640
|
+
// Stream by default, but fall back to one non-streamed JSON for the one
|
|
641
|
+
// case it breaks: a response_format grammar (fireworks) streams its
|
|
642
|
+
// constrained output mislabeled as reasoning_content, yet returns it as
|
|
643
|
+
// content non-streamed. The atomic dump is correct either way, so the
|
|
644
|
+
// demotion is scoped to exactly that request, not the whole provider.
|
|
645
|
+
const grammarBreaksStream = sendGrammar !== undefined && this.#grammarStyle === "response_format";
|
|
646
|
+
const transport = this.#streaming && !grammarBreaksStream ? chatCompletionStream : chatCompletion;
|
|
647
|
+
|
|
648
|
+
// Per-request headers = static auth/routing + any first-party telemetry.
|
|
649
|
+
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
|
|
650
|
+
const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
|
|
651
|
+
let raw;
|
|
652
|
+
for (let attempt = 0; ; attempt++) {
|
|
653
|
+
const timeoutSignal = AbortSignal.timeout(this.#fetchTimeoutMs);
|
|
654
|
+
const effectiveSignal = signal !== undefined ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal;
|
|
655
|
+
try {
|
|
656
|
+
raw = await transport({ url: this.#url, headers, body, signal: effectiveSignal, captureRawBody: this.#rawBody });
|
|
657
|
+
break;
|
|
658
|
+
} catch (err) {
|
|
659
|
+
// Caller-initiated abort is cancellation — never retried or wrapped.
|
|
660
|
+
if (signal?.aborted) throw err;
|
|
661
|
+
const { kind } = classifyProviderError(err);
|
|
662
|
+
// #543: Cloudflare/CDN edge codes (520-527) classify as network_failure
|
|
663
|
+
// but fail-fast - a retry just re-incurs the same origin/edge timeout.
|
|
664
|
+
const edgeTimeout = err instanceof OpenAiHttpError && isEdgeStatus(err.status);
|
|
665
|
+
// Terminal kind, edge failure, or budget spent -> surface the failure.
|
|
666
|
+
if (!RETRYABLE.has(kind) || edgeTimeout || attempt >= this.#retryAttempts) {
|
|
667
|
+
const pe = toProviderError(err, this.#source);
|
|
668
|
+
// #537 case 2: a 401/403 with a key PRESENT is a rejected key, not a
|
|
669
|
+
// transport failure — surface the distinct, actionable hint (kind stays
|
|
670
|
+
// "unauthorized", so core's routing is unchanged) rather than the raw
|
|
671
|
+
// upstream JSON. Distinct from the unset-key throw at construction.
|
|
672
|
+
if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
|
|
673
|
+
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
|
|
674
|
+
}
|
|
675
|
+
throw pe;
|
|
676
|
+
}
|
|
677
|
+
const retryAfter = err instanceof OpenAiHttpError ? err.retryAfter : null;
|
|
678
|
+
await sleepWithAbort(retryAfter ?? this.#retryDelayMs * 2 ** attempt, signal);
|
|
679
|
+
}
|
|
680
|
+
}
|
|
681
|
+
|
|
682
|
+
// #539: llama-server --special renders EOG tokens as text, so a turn ending
|
|
683
|
+
// via raw EOS carries a trailing <eos> the grammar never sanctioned - it both
|
|
684
|
+
// false-rejects the rail verdict and leaks a control token into the packet.
|
|
685
|
+
// Strip the server-reported eos_token from the tail ONCE, before the verdict
|
|
686
|
+
// grades it and before it reaches assistant/packet. rawBody keeps the verbatim
|
|
687
|
+
// wire text for forensics.
|
|
688
|
+
if (this.#eosText !== undefined) raw.content = stripTrailingSpecial(raw.content, this.#eosText);
|
|
689
|
+
|
|
690
|
+
// Grammar conformance (§13): bytes always flow; the verdict is an
|
|
691
|
+
// observation. Same check whether the grammar was transported
|
|
692
|
+
// (sendGrammar) or withheld (PLURNK_PROVIDERS_GBNF_DEBUG filter mode) — a
|
|
693
|
+
// non-accept verdict attaches a grammar_unenforced telemetry event
|
|
694
|
+
// (message + divergence position) and the response returns normally.
|
|
695
|
+
// Discard/retry/escalate/self-correct is the consumer's policy.
|
|
696
|
+
let telemetry: TelemetryEvent[] | undefined;
|
|
697
|
+
let railsMeta: Record<string, unknown> | undefined;
|
|
698
|
+
const usage = normalizeUsage(raw.usage, raw.reasoning_content, raw.content);
|
|
699
|
+
const observedGrammar = sendGrammar ?? (wantGrammar && this.#gbnfDebug ? grammar : undefined);
|
|
700
|
+
if (observedGrammar !== undefined) {
|
|
701
|
+
const verdict = this.#grammarVerdict(observedGrammar, raw.content);
|
|
702
|
+
if (verdict !== null && verdict.status !== "accept") {
|
|
703
|
+
telemetry = [{ source: this.#source, kind: "grammar_unenforced", message: describeUnenforced(verdict), position: verdict.pos }];
|
|
704
|
+
}
|
|
705
|
+
// #488 per-request loud state: rail attachment + conformance verdict
|
|
706
|
+
// ride `meta` into the consumer's turn row, so a drill reads rail
|
|
707
|
+
// presence PER TURN from the run db instead of inferring it from
|
|
708
|
+
// output shape (the #488 misdiagnosis, twice).
|
|
709
|
+
railsMeta = { railsAttached: sendGrammar !== undefined, railsVerdict: verdict?.status ?? "unverifiable" };
|
|
710
|
+
// #488 channel-escape detector (the run105 class): completion tokens
|
|
711
|
+
// billed far beyond every visible channel mean the decode ESCAPED into
|
|
712
|
+
// a server-discarded reasoning block mid-emission — unconstrained,
|
|
713
|
+
// invisible, billed (12,288 billed vs 1,033 chars visible, live).
|
|
714
|
+
// countTokens OVERCOUNTS text (chars/2 upper bound), so billed
|
|
715
|
+
// exceeding visible-plus-slack is real vanishing, not estimator noise.
|
|
716
|
+
const visible = this.#countTokens(raw.content) + this.#countTokens(raw.reasoning_content);
|
|
717
|
+
if (sendGrammar !== undefined && usage.completion > visible + 64) {
|
|
718
|
+
(telemetry ??= []).push({
|
|
719
|
+
source: this.#source,
|
|
720
|
+
kind: "grammar_unenforced",
|
|
721
|
+
message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ~${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
|
|
722
|
+
position: [...raw.content].length,
|
|
723
|
+
});
|
|
724
|
+
}
|
|
725
|
+
}
|
|
726
|
+
|
|
727
|
+
const builtMeta = this.#buildMeta(raw.chunkMetadata);
|
|
728
|
+
const meta = railsMeta !== undefined ? { ...builtMeta, ...railsMeta } : builtMeta;
|
|
729
|
+
|
|
730
|
+
// #36: surface per-token logprobs + their mean when the backend returned
|
|
731
|
+
// them (only possible when the flag requested them). Absent otherwise —
|
|
732
|
+
// never synthesized.
|
|
733
|
+
const logprobs = raw.logprobs !== null && raw.logprobs.length > 0 ? raw.logprobs : undefined;
|
|
734
|
+
const meanLogprob = logprobs !== undefined
|
|
735
|
+
? logprobs.reduce((sum, t) => sum + t.logprob, 0) / logprobs.length
|
|
736
|
+
: undefined;
|
|
737
|
+
|
|
738
|
+
return {
|
|
739
|
+
assistant: {
|
|
740
|
+
content: raw.content,
|
|
741
|
+
reasoning: raw.reasoning_content.length > 0 ? raw.reasoning_content : null,
|
|
742
|
+
// #482: sealed relay blobs ride only when present — same absence
|
|
743
|
+
// discipline as logprobs (never synthesized, never empty-array).
|
|
744
|
+
...(raw.reasoning_encrypted.length > 0 ? { reasoningEncrypted: raw.reasoning_encrypted } : {}),
|
|
745
|
+
usage,
|
|
746
|
+
finishReason: normalizeFinishReason(raw.finish_reason),
|
|
747
|
+
model: raw.model ?? this.#model,
|
|
748
|
+
...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
|
|
749
|
+
},
|
|
750
|
+
assistantRaw: raw,
|
|
751
|
+
...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
|
|
752
|
+
...(meta !== undefined ? { meta } : {}),
|
|
753
|
+
...(telemetry !== undefined ? { telemetry } : {}),
|
|
754
|
+
};
|
|
755
|
+
}
|
|
756
|
+
}
|