@plurnk/plurnk-providers 1.2.0 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/index.ts ADDED
@@ -0,0 +1,51 @@
1
+ export type {
2
+ ChatMessage,
3
+ FinishReason,
4
+ Provider,
5
+ ProviderAssistant,
6
+ ProviderFactory,
7
+ ProviderOptions,
8
+ ProviderResponse,
9
+ ProviderUsage,
10
+ TokenLogprob,
11
+ TokenAlternative,
12
+ } from "./types.ts";
13
+
14
+ // Alias cascade — re-exported from the zero-dep @plurnk/plurnk-aliases (#27), so
15
+ // the "." surface is unchanged for existing importers and there's one source of
16
+ // truth for the parser (thin clients depend on that package directly).
17
+ export type { ProviderAlias } from "@plurnk/plurnk-aliases";
18
+ export { parseAliasesFromEnv, resolveActiveAlias } from "@plurnk/plurnk-aliases";
19
+
20
+ export {
21
+ instantiateProvider,
22
+ loadActiveProvider,
23
+ resetDiscoveryCache,
24
+ } from "./ProviderRegistry.ts";
25
+
26
+ // Scope-agnostic tier-2 discovery (SPEC §5) — exported so a consumer can list
27
+ // installed providers (first-party + third-party) without instantiating them.
28
+ export { discover } from "./discover.ts";
29
+ export type { DiscoverOptions, Discovery } from "./discover.ts";
30
+
31
+ // Shared OpenAI-compatible transport machinery — the spine every sibling
32
+ // extends and the basis for ./standardProviders.ts.
33
+ export { default as OpenAICompatProvider, effortFromBudget } from "./OpenAICompat.ts";
34
+ export type { OpenAICompatConfig, ReasoningStyle, GrammarStyle } from "./OpenAICompat.ts";
35
+ // Capacity pool (SPEC §15): front N interchangeable backends as one Provider -
36
+ // worker-sticky for KV-cache reuse, overflow to a healthy sibling; the blend
37
+ // DECISION stays the consumer's, by choosing which pool to call.
38
+ export { default as Pool } from "./Pool.ts";
39
+ export { chatCompletionStream, chatCompletion, OpenAiHttpError } from "./openaiStream.ts";
40
+ export type { StreamResponse, EncryptedReasoningItem } from "./openaiStream.ts";
41
+ export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, envelopeFromEnv, resolveReserve } from "./env.ts";
42
+ export type { Reasoning, ReasoningMode, ReserveSpec } from "./env.ts";
43
+ export { normalizeUsage, computeCost } from "./usage.ts";
44
+ export type { RawUsage, TokenRates } from "./usage.ts";
45
+ export { ProviderError, classifyProviderError, toProviderError, providerSource } from "./telemetry.ts";
46
+ export type { TelemetryEvent, ProviderTelemetryKind } from "./telemetry.ts";
47
+ export { STANDARD_PROVIDERS, isStandardProvider, standardProviderFromEnv } from "./standardProviders.ts";
48
+
49
+ export { default as Mock } from "./Mock.ts";
50
+ export type { MockAssistant, MockResponse, MockReturnedAssistant } from "./Mock.ts";
51
+ export { mockDefaultUsage } from "./Mock.ts";
@@ -0,0 +1,58 @@
1
+ // [§lexicon] the providers-lane standing guard (#472/#477, owner-ruled OpenAI
2
+ // lexicon). Retired terms fail CI here, not at the next audit — the mirror of
3
+ // core's plurnk-core guard, tuned to what PROVIDERS retired. Scope: src/ non-test
4
+ // + SPEC.md. A `lexicon-allow` line marker exempts the shed's own call sites
5
+ // (shedRenamed) that must NAME a retired knob to point off it. PLURNK_SERVICE_*
6
+ // knobs are CORE's — its own guard polices them; this one stays in-lane.
7
+ //
8
+ // The `thinking` rule differs from core's on purpose: in the PROVIDER (wire)
9
+ // layer, `thinking` is legitimate backend VOCABULARY — anthropic's `thinking`
10
+ // wire object, a backend's `thinking` SSE field, gemini's "thinking models"
11
+ // brand. So the rule catches OUR-VOICE prose drift while exempting the wire
12
+ // forms (a `thinking` object key, a `.thinking` field read, a `thinking` object/
13
+ // model reference). Core, a consumer that never speaks the wire, bans it flat.
14
+ import test from "node:test";
15
+ import { strict as assert } from "node:assert";
16
+ import { readdirSync, readFileSync, statSync } from "node:fs";
17
+ import { join, resolve } from "node:path";
18
+ import { fileURLToPath } from "node:url";
19
+
20
+ const ROOT = resolve(fileURLToPath(import.meta.url), "..", "..");
21
+ const walk = (d: string, a: string[] = []): string[] => {
22
+ for (const e of readdirSync(d)) {
23
+ const p = join(d, e);
24
+ if (statSync(p).isDirectory()) walk(p, a);
25
+ else if (/\.ts$/.test(e) && !/\.test\./.test(e)) a.push(p);
26
+ }
27
+ return a;
28
+ };
29
+
30
+ // `thinking` as backend wire vocabulary (never our-voice): a wire object key,
31
+ // a field read, a backtick-named param, the "thinking object/model" references.
32
+ const WIRE_THINKING = /enable_thinking|`thinking`|thinking\s*:|\.thinking\b|thinking (?:model|object)/i;
33
+
34
+ // Each entry: the banned pattern and the canonical term the violation must become.
35
+ const BANNED: Array<{ label: string; re: RegExp; canon: string; exempt?: RegExp }> = [
36
+ { label: "thinking (our-voice)", re: /\bthinking\b/i, canon: "reasoning (the #472 lexicon ruling)", exempt: WIRE_THINKING },
37
+ { label: "contextSize", re: /\bcontextSize\b/, canon: "contextWindow — the provider window (#472)" },
38
+ { label: "retired providers knob", re: /PLURNK_PROVIDERS_(THINKING|LOGPROB\b|CONTEXT_SIZE\b)/, canon: "PLURNK_PROVIDERS_{REASONING,TOP_LOGPROBS,CONTEXT_WINDOW} (#399/#472) — only the shed may name these" },
39
+ // #511: catch the retired run/session noun in the WIRE-HEADER form too (a
40
+ // quoted string, not an identifier — the hole the old `Plurnk-Run-Id` hid in),
41
+ // alongside the coordinate identifiers.
42
+ { label: "run/session (retired noun — coordinate or wire header)", re: /\b(sessionId|runId)\b|Plurnk-(Run|Session)-Id/, canon: "workerId/workspaceId, Plurnk-Worker-Id/Plurnk-Workspace-Id (#486/#511)" },
43
+ ];
44
+
45
+ test("retired provider terms never reappear in src or SPEC — drift fails CI, not the next audit", () => {
46
+ const files = [...walk(join(ROOT, "src")), join(ROOT, "SPEC.md")];
47
+ const violations: string[] = [];
48
+ for (const f of files) {
49
+ readFileSync(f, "utf8").split("\n").forEach((line, i) => {
50
+ if (line.includes("lexicon-allow")) return; // the shed's own regex/error strings + migration notes carry the marker
51
+ for (const { label, re, canon, exempt } of BANNED) {
52
+ if (exempt !== undefined && exempt.test(line)) continue;
53
+ if (re.test(line)) violations.push(`${f.slice(ROOT.length + 1)}:${i + 1} [${label}] → ${canon}`);
54
+ }
55
+ });
56
+ }
57
+ assert.deepEqual(violations, [], `retired lexicon found:\n${violations.join("\n")}`);
58
+ });
@@ -0,0 +1,279 @@
1
+ // SSE client for OpenAI-compatible /chat/completions. Streaming keeps long
2
+ // completions alive through CDN proxies; the aggregated result is returned as
3
+ // one StreamResponse (the Provider contract is atomic — no partial resolves).
4
+ // Adapted from rummy's proven implementation; previously copy-pasted byte-for-
5
+ // byte into every @plurnk/plurnk-providers-* sibling, now shared from here.
6
+
7
+ type StreamRequest = {
8
+ url: string;
9
+ headers: Record<string, string>;
10
+ body: Record<string, unknown>;
11
+ signal: AbortSignal;
12
+ // #36: assemble the verbatim wire body onto StreamResponse.rawBody. Off by
13
+ // default so a serving turn never pays the reassembly/retention cost.
14
+ captureRawBody?: boolean;
15
+ };
16
+
17
+ import type { RawUsage } from "./usage.ts";
18
+ import type { TokenLogprob } from "./types.ts";
19
+
20
+ // Sealed reasoning (#482, widened per client). A relay backend (OpenRouter
21
+ // fronting OpenAI o-series) returns the chain-of-thought ENCRYPTED as
22
+ // reasoning_details entries ({ type: "reasoning.encrypted", id, data, format,
23
+ // index }) while readable text still rides reasoning/reasoning_content (verified
24
+ // live: o4-mini via OpenRouter — reasoning null, one encrypted entry, format
25
+ // "openai-responses-v1"). The ITEM shape preserves the wire's `id` (a flat blob
26
+ // list dropped item identity — the widening's whole point) and a `subtype` from
27
+ // wire POSITION: we parse message.reasoning_details, so it is message-attached.
28
+ // plurnk is tools-in-body (SPEC §2), so reasoning is never tool-call-attached and
29
+ // subtype is constant here; the field is structural, future-proofing the seam.
30
+ // An ARRAY of items (not a single object) so N distinct reasoning ids never
31
+ // re-collide the identity this fixes. Blobs verbatim, never decoded.
32
+ export type EncryptedReasoningItem = { id: string | null; subtype: string; encrypted: Array<{ data: string; format: string | null }> };
33
+
34
+ type RawEncrypted = { id: string | null; data: string; format: string | null };
35
+
36
+ // Group accumulated encrypted entries into items by wire `id` (id-less entries
37
+ // stand alone, never merged). Order preserved.
38
+ const groupEncrypted = (entries: Iterable<RawEncrypted>): EncryptedReasoningItem[] => {
39
+ const items: EncryptedReasoningItem[] = [];
40
+ const byId = new Map<string, EncryptedReasoningItem>();
41
+ for (const e of entries) {
42
+ const blob = { data: e.data, format: e.format };
43
+ if (e.id === null) { items.push({ id: null, subtype: "message", encrypted: [blob] }); continue; }
44
+ let item = byId.get(e.id);
45
+ if (item === undefined) { item = { id: e.id, subtype: "message", encrypted: [] }; byId.set(e.id, item); items.push(item); }
46
+ item.encrypted.push(blob);
47
+ }
48
+ return items;
49
+ };
50
+
51
+ const encryptedFromDetails = (details: unknown): EncryptedReasoningItem[] => {
52
+ if (!Array.isArray(details)) return [];
53
+ const raw: RawEncrypted[] = [];
54
+ for (const e of details) {
55
+ const entry = e as { type?: unknown; id?: unknown; data?: unknown; format?: unknown };
56
+ if (entry?.type !== "reasoning.encrypted" || typeof entry.data !== "string") continue;
57
+ raw.push({ id: typeof entry.id === "string" ? entry.id : null, data: entry.data, format: typeof entry.format === "string" ? entry.format : null });
58
+ }
59
+ return groupEncrypted(raw);
60
+ };
61
+
62
+ export type StreamResponse = {
63
+ model: string | null;
64
+ content: string;
65
+ reasoning_content: string;
66
+ // Sealed relay reasoning (#482) — empty for the open-reasoning backends.
67
+ reasoning_encrypted: EncryptedReasoningItem[];
68
+ finish_reason: string | null;
69
+ usage: RawUsage | null;
70
+ chunkMetadata: Record<string, unknown>;
71
+ // #36: per-token logprobs parsed from choices[0].logprobs.content[], present
72
+ // only when the request asked for them (else the field is absent → null).
73
+ logprobs: TokenLogprob[] | null;
74
+ // #36: the verbatim response body, populated only when captureRawBody is set.
75
+ rawBody: unknown;
76
+ };
77
+
78
+ // Map an OpenAI-style `logprobs.content[]` array to the canonical structured view
79
+ // (#36). Reads the RAW `logprob` (not `sampling_logprob`); `top_logprobs` → `top`.
80
+ // Returns null when the shape is absent — never synthesizes.
81
+ const parseLogprobs = (raw: unknown): TokenLogprob[] | null => {
82
+ const content = (raw as { content?: unknown } | null | undefined)?.content;
83
+ if (!Array.isArray(content)) return null;
84
+ return content.map((entry) => {
85
+ const { token, logprob, top_logprobs } = entry as { token: string; logprob: number; top_logprobs?: unknown };
86
+ const top = Array.isArray(top_logprobs)
87
+ ? top_logprobs.map((a) => ({ token: (a as TokenLogprob).token, logprob: (a as TokenLogprob).logprob }))
88
+ : undefined;
89
+ return top !== undefined ? { token, logprob, top } : { token, logprob };
90
+ });
91
+ };
92
+
93
+ // Cloudflare/CDN EDGE status codes (520-527): infrastructure failures the proxy
94
+ // returns (as HTML error pages), NOT OpenAI/API statuses. A retry re-incurs the
95
+ // same origin wait, so they fail-fast (#543).
96
+ const EDGE_LABELS: ReadonlyMap<number, string> = new Map([
97
+ [520, "web server returned an unknown error"], [521, "web server is down"],
98
+ [522, "connection timed out"], [523, "origin is unreachable"], [524, "origin timeout"],
99
+ [525, "SSL handshake failed"], [526, "invalid SSL certificate"], [527, "railgun error"],
100
+ ]);
101
+ export const isEdgeStatus = (status: number): boolean => status >= 520 && status <= 527;
102
+
103
+ export class OpenAiHttpError extends Error {
104
+ readonly status: number;
105
+ readonly body: string;
106
+ readonly retryAfter: number | null;
107
+ constructor(status: number, body: string, retryAfter: number | null) {
108
+ super(OpenAiHttpError.#describe(status, body));
109
+ this.status = status;
110
+ this.body = body;
111
+ this.retryAfter = retryAfter;
112
+ }
113
+ // A non-JSON error body (a proxy/CDN HTML page) collapses to one line and drops
114
+ // the misleading "OpenAI" prefix - an edge code is not an API status (#543).
115
+ // JSON API errors pass through verbatim.
116
+ static #describe(status: number, body: string): string {
117
+ if (body.trimStart().startsWith("<")) return `${status} ${EDGE_LABELS.get(status) ?? "edge/proxy error"}`;
118
+ return `OpenAI ${status} - ${body}`;
119
+ }
120
+ }
121
+
122
+ const parseRetryAfter = (header: string | null): number | null => {
123
+ if (header === null) return null;
124
+ const asInt = Number.parseInt(header, 10);
125
+ if (Number.isFinite(asInt)) return asInt * 1000;
126
+ const asDate = Date.parse(header);
127
+ if (Number.isFinite(asDate)) return Math.max(0, asDate - Date.now());
128
+ return null;
129
+ };
130
+
131
+ // Non-streaming sibling. Same request/error handling, but POSTs without
132
+ // `stream` and parses the single JSON body into the SAME StreamResponse shape.
133
+ // For backends whose STREAMING response misbehaves (e.g. Fireworks labels
134
+ // grammar-constrained output as `reasoning_content` instead of `content`) —
135
+ // the Provider contract is atomic either way, so the transport is free to
136
+ // choose. The fetch timeout (AbortSignal) bounds the wait; there is no proxy
137
+ // between us and the backend that would idle out a non-streamed request.
138
+ export const chatCompletion = async ({ url, headers, body, signal, captureRawBody }: StreamRequest): Promise<StreamResponse> => {
139
+ const response = await fetch(url, {
140
+ method: "POST",
141
+ headers: { "Content-Type": "application/json", ...headers },
142
+ body: JSON.stringify(body),
143
+ signal,
144
+ });
145
+ if (!response.ok) {
146
+ const errorBody = await response.text();
147
+ throw new OpenAiHttpError(response.status, errorBody, parseRetryAfter(response.headers.get("retry-after")));
148
+ }
149
+ const j = (await response.json()) as Record<string, unknown>;
150
+ const choices = j.choices as Array<Record<string, unknown>> | undefined;
151
+ const choice = (choices?.[0] ?? {}) as Record<string, unknown>;
152
+ const msg = (choice.message ?? {}) as Record<string, unknown>;
153
+ const reasoning = msg.reasoning_content ?? msg.reasoning ?? msg.thinking ?? "";
154
+ const chunkMetadata: Record<string, unknown> = {};
155
+ for (const [k, v] of Object.entries(j)) if (k !== "choices" && k !== "usage") chunkMetadata[k] = v;
156
+ return {
157
+ model: typeof j.model === "string" ? j.model : null,
158
+ content: typeof msg.content === "string" ? msg.content : "",
159
+ reasoning_content: typeof reasoning === "string" ? reasoning : "",
160
+ reasoning_encrypted: encryptedFromDetails(msg.reasoning_details),
161
+ finish_reason: typeof choice.finish_reason === "string" ? choice.finish_reason : null,
162
+ usage: (j.usage ?? null) as StreamResponse["usage"],
163
+ chunkMetadata,
164
+ logprobs: parseLogprobs(choice.logprobs),
165
+ // Non-streamed: the parsed JSON IS the verbatim wire body, exact.
166
+ rawBody: captureRawBody === true ? j : undefined,
167
+ };
168
+ };
169
+
170
+ export const chatCompletionStream = async ({ url, headers, body, signal, captureRawBody }: StreamRequest): Promise<StreamResponse> => {
171
+ const requestBody = { ...body, stream: true, stream_options: { include_usage: true } };
172
+
173
+ const response = await fetch(url, {
174
+ method: "POST",
175
+ headers: { "Content-Type": "application/json", ...headers },
176
+ body: JSON.stringify(requestBody),
177
+ signal,
178
+ });
179
+
180
+ if (!response.ok) {
181
+ const errorBody = await response.text();
182
+ throw new OpenAiHttpError(response.status, errorBody, parseRetryAfter(response.headers.get("retry-after")));
183
+ }
184
+
185
+ if (response.body === null) throw new Error("OpenAI response body is null");
186
+ const reader = response.body.getReader();
187
+ const decoder = new TextDecoder();
188
+
189
+ let buffer = "";
190
+ let content = "";
191
+ let reasoning_content = "";
192
+ let usage: StreamResponse["usage"] = null;
193
+ let model: string | null = null;
194
+ let finish_reason: string | null = null;
195
+ const chunkMetadata: Record<string, unknown> = {};
196
+ // #36: logprobs stream as per-chunk choices[0].logprobs.content[] deltas —
197
+ // accumulate the raw entries across chunks, map once at the end.
198
+ const logprobEntries: unknown[] = [];
199
+ // #482: encrypted reasoning_details stream chunked — concatenate `data` per
200
+ // reassembly key (index when present, else id, else a counter); the id/format
201
+ // ride along and items group by id at the end.
202
+ const encryptedByKey = new Map<string, RawEncrypted>();
203
+ let encryptedNoKey = 0;
204
+
205
+ while (true) {
206
+ const { done, value } = await reader.read();
207
+ if (done) break;
208
+ buffer += decoder.decode(value, { stream: true });
209
+ const lines = buffer.split("\n");
210
+ buffer = lines.pop() ?? "";
211
+
212
+ for (const rawLine of lines) {
213
+ const line = rawLine.trim();
214
+ if (!line.startsWith("data:")) continue;
215
+ const payload = line.slice(5).trimStart();
216
+ if (payload === "[DONE]" || payload === "") continue;
217
+
218
+ let chunk: Record<string, unknown>;
219
+ try { chunk = JSON.parse(payload) as Record<string, unknown>; } catch { continue; }
220
+
221
+ // A streaming server may flush HTTP 200 headers before inference
222
+ // completes, then report a terminal failure as an SSE error frame.
223
+ // That frame is a failed exchange, never an empty completion.
224
+ if (chunk.error !== null && typeof chunk.error === "object") {
225
+ const status = typeof chunk.status === "number"
226
+ && Number.isInteger(chunk.status)
227
+ && chunk.status >= 400
228
+ && chunk.status <= 599
229
+ ? chunk.status
230
+ : 500;
231
+ throw new OpenAiHttpError(status, JSON.stringify({ error: chunk.error }), null);
232
+ }
233
+
234
+ if (typeof chunk.model === "string") model = chunk.model;
235
+ if (chunk.usage !== undefined && chunk.usage !== null) usage = chunk.usage as StreamResponse["usage"];
236
+
237
+ for (const [k, v] of Object.entries(chunk)) {
238
+ if (k === "choices" || k === "usage") continue;
239
+ chunkMetadata[k] = v;
240
+ }
241
+
242
+ const choices = chunk.choices as Array<Record<string, unknown>> | undefined;
243
+ const choice = choices?.[0];
244
+ if (choice === undefined) continue;
245
+ if (typeof choice.finish_reason === "string") finish_reason = choice.finish_reason;
246
+
247
+ const chunkLogprobs = (choice.logprobs as { content?: unknown } | undefined)?.content;
248
+ if (Array.isArray(chunkLogprobs)) logprobEntries.push(...chunkLogprobs);
249
+
250
+ const delta = choice.delta as Record<string, unknown> | undefined;
251
+ if (delta === undefined) continue;
252
+ if (typeof delta.content === "string") content += delta.content;
253
+ // Reasoning surfaces under different field names per provider.
254
+ if (typeof delta.reasoning_content === "string") reasoning_content += delta.reasoning_content;
255
+ if (typeof delta.reasoning === "string") reasoning_content += delta.reasoning;
256
+ if (typeof delta.thinking === "string") reasoning_content += delta.thinking;
257
+ if (Array.isArray(delta.reasoning_details)) {
258
+ for (const e of delta.reasoning_details) {
259
+ const entry = e as { type?: unknown; id?: unknown; data?: unknown; format?: unknown; index?: unknown };
260
+ if (entry?.type !== "reasoning.encrypted" || typeof entry.data !== "string") continue;
261
+ const id = typeof entry.id === "string" ? entry.id : null;
262
+ const key = typeof entry.index === "number" ? `i${entry.index}` : id ?? `n${encryptedNoKey++}`;
263
+ const prev = encryptedByKey.get(key);
264
+ if (prev !== undefined) prev.data += entry.data;
265
+ else encryptedByKey.set(key, { id, data: entry.data, format: typeof entry.format === "string" ? entry.format : null });
266
+ }
267
+ }
268
+ }
269
+ }
270
+
271
+ const logprobs = logprobEntries.length > 0 ? parseLogprobs({ content: logprobEntries }) : null;
272
+ // Streamed turns have no single verbatim wire body; reassemble the equivalent
273
+ // (#36) — chunk-level fields (chunkMetadata) + the collected choice — only when
274
+ // asked, so serving turns pay nothing.
275
+ const rawBody = captureRawBody === true
276
+ ? { ...chunkMetadata, model, usage, choices: [{ index: 0, message: { content, reasoning_content }, finish_reason, logprobs: logprobs !== null ? { content: logprobEntries } : null }] }
277
+ : undefined;
278
+ return { model, content, reasoning_content, reasoning_encrypted: groupEncrypted(encryptedByKey.values()), finish_reason, usage, chunkMetadata, logprobs, rawBody };
279
+ };