@plurnk/plurnk-providers 1.3.4 → 1.3.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. package/.env.defaults +41 -52
  2. package/README.md +44 -53
  3. package/SPEC.md +215 -354
  4. package/dist/AiSdkProvider.d.ts +78 -0
  5. package/dist/AiSdkProvider.d.ts.map +1 -0
  6. package/dist/AiSdkProvider.js +591 -0
  7. package/dist/AiSdkProvider.js.map +1 -0
  8. package/dist/Mock.d.ts +1 -1
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +1 -1
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/OpenAICompat.d.ts +3 -5
  13. package/dist/OpenAICompat.d.ts.map +1 -1
  14. package/dist/OpenAICompat.js +44 -133
  15. package/dist/OpenAICompat.js.map +1 -1
  16. package/dist/Pool.d.ts +1 -1
  17. package/dist/Pool.d.ts.map +1 -1
  18. package/dist/Pool.js +1 -1
  19. package/dist/Pool.js.map +1 -1
  20. package/dist/ProviderRegistry.d.ts.map +1 -1
  21. package/dist/ProviderRegistry.js +37 -24
  22. package/dist/ProviderRegistry.js.map +1 -1
  23. package/dist/aiSdkTransport.d.ts +52 -0
  24. package/dist/aiSdkTransport.d.ts.map +1 -0
  25. package/dist/aiSdkTransport.js +294 -0
  26. package/dist/aiSdkTransport.js.map +1 -0
  27. package/dist/catalogProvider.d.ts +15 -0
  28. package/dist/catalogProvider.d.ts.map +1 -0
  29. package/dist/catalogProvider.js +103 -0
  30. package/dist/catalogProvider.js.map +1 -0
  31. package/dist/compatibleProvider.d.ts +3 -0
  32. package/dist/compatibleProvider.d.ts.map +1 -0
  33. package/dist/compatibleProvider.js +146 -0
  34. package/dist/compatibleProvider.js.map +1 -0
  35. package/dist/discover.d.ts.map +1 -1
  36. package/dist/discover.js.map +1 -1
  37. package/dist/env.d.ts +1 -0
  38. package/dist/env.d.ts.map +1 -1
  39. package/dist/env.js +13 -6
  40. package/dist/env.js.map +1 -1
  41. package/dist/index.d.ts +5 -7
  42. package/dist/index.d.ts.map +1 -1
  43. package/dist/index.js +4 -8
  44. package/dist/index.js.map +1 -1
  45. package/dist/ollama.d.ts +3 -0
  46. package/dist/ollama.d.ts.map +1 -0
  47. package/dist/ollama.js +39 -0
  48. package/dist/ollama.js.map +1 -0
  49. package/dist/openai.d.ts +2 -4
  50. package/dist/openai.d.ts.map +1 -1
  51. package/dist/openai.js +1 -2
  52. package/dist/openai.js.map +1 -1
  53. package/dist/sdkModels.d.ts +13 -0
  54. package/dist/sdkModels.d.ts.map +1 -0
  55. package/dist/sdkModels.js +153 -0
  56. package/dist/sdkModels.js.map +1 -0
  57. package/dist/standardProviders.d.ts +0 -1
  58. package/dist/standardProviders.d.ts.map +1 -1
  59. package/dist/standardProviders.js +9 -11
  60. package/dist/standardProviders.js.map +1 -1
  61. package/dist/telemetry.d.ts.map +1 -1
  62. package/dist/telemetry.js +20 -9
  63. package/dist/telemetry.js.map +1 -1
  64. package/dist/types.d.ts +4 -3
  65. package/dist/types.d.ts.map +1 -1
  66. package/dist/usage.d.ts +1 -1
  67. package/dist/usage.d.ts.map +1 -1
  68. package/dist/usage.js +4 -2
  69. package/dist/usage.js.map +1 -1
  70. package/package.json +19 -8
  71. package/src/{OpenAICompat.test.ts → AiSdkProvider.test.ts} +202 -177
  72. package/src/{OpenAICompat.ts → AiSdkProvider.ts} +89 -144
  73. package/src/Mock.test.ts +3 -3
  74. package/src/Mock.ts +1 -1
  75. package/src/Pool.test.ts +3 -3
  76. package/src/Pool.ts +1 -1
  77. package/src/ProviderRegistry.test.ts +40 -27
  78. package/src/ProviderRegistry.ts +35 -24
  79. package/src/aiSdkTransport.test.ts +253 -0
  80. package/src/aiSdkTransport.ts +369 -0
  81. package/src/boundaries.test.ts +2 -2
  82. package/src/catalogProvider.test.ts +100 -0
  83. package/src/catalogProvider.ts +151 -0
  84. package/src/compatibleProvider.test.ts +44 -0
  85. package/src/compatibleProvider.ts +205 -0
  86. package/src/discover.test.ts +12 -12
  87. package/src/discover.ts +3 -6
  88. package/src/env.ts +14 -6
  89. package/src/index.ts +7 -11
  90. package/src/ollama.ts +63 -0
  91. package/src/openai.ts +2 -8
  92. package/src/sdkModels.test.ts +47 -0
  93. package/src/sdkModels.ts +194 -0
  94. package/src/telemetry.test.ts +17 -10
  95. package/src/telemetry.ts +22 -14
  96. package/src/types.ts +10 -13
  97. package/src/usage.test.ts +8 -10
  98. package/src/usage.ts +7 -3
  99. package/src/openaiStream.ts +0 -310
  100. package/src/standardProviders.test.ts +0 -949
  101. package/src/standardProviders.ts +0 -635
@@ -1,21 +1,21 @@
1
- // Shared OpenAI-compatible provider. Implements the universal generate()
1
+ // PLURNK adapter over an AI SDK language model. Implements the universal generate()
2
2
  // spine — signal merging, the SSE call, usage mapping, finishReason
3
3
  // normalization, response assembly — that every sibling had duplicated.
4
4
  //
5
- // Composition, not inheritance: the per-provider deltas (resolved URL, auth
6
- // headers, reasoning translation style, tokenizer, cost) arrive as config.
7
- // A sibling's fromEnv probes whatever it needs (catalog, pricing, context
8
- // window), builds the config, and returns `new OpenAICompatProvider(config)`.
9
- // Pure-config providers come from ./standardProviders.ts with no sibling at all.
5
+ // Composition, not inheritance: an official AI SDK language model supplies the
6
+ // ordinary vendor protocol. The compatible URL path remains only for PLURNK
7
+ // extensions and local endpoint probes the SDK cannot represent.
10
8
 
11
9
  import type { ChatMessage, FinishReason, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
12
10
  import type { Reasoning, ReserveSpec } from "./env.ts";
13
- import { chatCompletionStream, chatCompletion, OpenAiHttpError, isEdgeStatus, type ProviderFetch, type StreamResponse } from "./openaiStream.ts";
14
- import { normalizeUsage } from "./usage.ts";
15
- import { toProviderError, classifyProviderError, ProviderError, type TelemetryEvent } from "./telemetry.ts";
11
+ import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.ts";
12
+ import type { LanguageModel } from "ai";
13
+ import { toProviderError, ProviderError, type TelemetryEvent } from "./telemetry.ts";
16
14
  import { validateGbnf, type Verdict } from "@plurnk/gbnf";
17
15
  import { emitWarningOnce } from "./warnings.ts";
18
16
 
17
+ export type ProviderFetch = typeof globalThis.fetch;
18
+
19
19
  // How the reasoning intent (PLURNK_PROVIDERS_REASONING: off | adaptive | on, plus
20
20
  // REASONING_BUDGET iff on — #32/#33) translates to each backend's wire mechanism
21
21
  // (SPEC §4); the per-style mapping lives in #reasoningBody. Non-obvious ones:
@@ -33,9 +33,10 @@ export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" |
33
33
  // service-managed constrained sampling; endpoint-owned settings are not inferred.
34
34
  export type GrammarStyle = "none" | "llamacpp";
35
35
 
36
- export type OpenAICompatConfig = {
36
+ export type AiSdkProviderConfig = {
37
37
  model: string;
38
- url: string; // fully-resolved chat-completions URL
38
+ url?: string; // OpenAI-compatible chat-completions URL
39
+ languageModel?: LanguageModel; // native AI SDK provider model
39
40
  fetchTimeoutMs: number;
40
41
  streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
41
42
  headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
@@ -43,7 +44,7 @@ export type OpenAICompatConfig = {
43
44
  contextWindow?: number | null; // default null; caller resolves-or-fails (#419), narrows to required with the interface
44
45
  reasoningStyle?: ReasoningStyle; // default "none"
45
46
  countTokens?: (text: string) => number; // default chars/2 upper-bound heuristic
46
- costFor?: (usage: ProviderUsage) => number; // default () => 0
47
+ calculateCost?: (usage: ProviderUsage) => number; // default () => 0
47
48
  source?: string; // telemetry source, e.g. "provider:openai"; default "provider"
48
49
  grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
49
50
  // #518: send the OpenAI-standard `prompt_cache_key` set to workerId, so a
@@ -60,7 +61,6 @@ export type OpenAICompatConfig = {
60
61
  firstPartyMetadata?: boolean; // forward per-turn attributions + client as Plurnk-* headers (plurnk only); default false
61
62
  apiKeyRejectedMessage?: string; // #537: friendly hint when a PRESENT key is 401/403-rejected (distinct from unset); default undefined
62
63
  eosText?: string; // #539: server-reported eos_token, stripped from the content tail (--special renders it as text); default undefined
63
- balanceMetaKey?: string; // top-level response field carrying account balance (pico-USD) → validated meta.balancePico (plurnk only, #23); default unset
64
64
  // Slot affinity wiring (provider-INTERNAL — never consumer-facing, #11).
65
65
  supportsSlotPinning?: boolean; // backend accepts an `id_slot` body field (llama-server); default false
66
66
  slotCount?: number | null; // probed slot count for pinning backends; default null
@@ -88,8 +88,7 @@ export type OpenAICompatConfig = {
88
88
  // DEFAULT for EVERY request, spread UNDER caller sampling (#30/endpoint#7).
89
89
  // `repeatPenalty` is the FLOOR the provider manages wherever a grammar rides
90
90
  // (greedy-under-mask loops without it, #9) — the VALUE is operator config;
91
- // WHERE it applies stays mechanism. `retryDelayMs` is the transient-retry
92
- // backoff base (attempt N waits retryDelayMs * 2^(N-1); Retry-After wins).
91
+ // WHERE it applies stays mechanism.
93
92
  temperature: number;
94
93
  repeatPenalty: number;
95
94
  // #426: anti-degeneration guard on the CLOUD path (grammarStyle "none"), where the
@@ -107,7 +106,6 @@ export type OpenAICompatConfig = {
107
106
  dryBase?: number;
108
107
  dryAllowedLength?: number;
109
108
  repeatLastN?: number;
110
- retryDelayMs: number;
111
109
  // Transient-failure retry budget — REQUIRED, no in-code default
112
110
  // (PLURNK_PROVIDERS_RETRY_ATTEMPTS, a non-negative int): 0 = surface the
113
111
  // first failure; N = up to N retries on a transient error (§4, #18).
@@ -135,13 +133,6 @@ export type OpenAICompatConfig = {
135
133
  tuningFloors?: boolean;
136
134
  };
137
135
 
138
- // Transient classifications worth retrying: rate_limit (429) and network_failure
139
- // (5xx, timeout, connection reset) are transport; grammar_invalid (a 422 output
140
- // reject a fresh sample may satisfy, #548) rides the same bounded budget.
141
- // unauthorized, quota_exceeded, invalid_response, model_refused are terminal —
142
- // retrying just burns time and budget.
143
- const RETRYABLE: ReadonlySet<string> = new Set(["rate_limit", "network_failure", "grammar_invalid"]);
144
-
145
136
  // #539: drop trailing occurrences of a server-rendered EOG marker. llama-server
146
137
  // under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
147
138
  // trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
@@ -154,42 +145,6 @@ const stripTrailingSpecial = (content: string, marker: string): string => {
154
145
  return out;
155
146
  };
156
147
 
157
- // Sleep that rejects the moment `signal` aborts (caller cancellation must not
158
- // wait out a backoff). Resolves normally on timeout.
159
- const sleepWithAbort = (ms: number, signal: AbortSignal | undefined): Promise<void> =>
160
- new Promise((resolve, reject) => {
161
- if (signal?.aborted) { reject(signal.reason); return; }
162
- const timer = setTimeout(resolve, ms);
163
- signal?.addEventListener("abort", () => { clearTimeout(timer); reject(signal.reason); }, { once: true });
164
- });
165
-
166
- // SPEC §2 closed set. The four canonical values pass through; known per-backend
167
- // synonyms translate INTO them (anthropic max_tokens/end_turn, gemini MAX_TOKENS/
168
- // SAFETY/RECITATION) so a token-cap hit canonicalizes to "length" whatever the
169
- // backend names it -- core's `finishReason === "length"` truncation check (#425)
170
- // is then an invariant by construction, not a convention each backend must
171
- // independently honor. A non-empty value outside both the set and the table
172
- // collapses to null AND warns once, so a new backend's unmapped cap string
173
- // surfaces instead of silently becoming "no signal" (which would make core miss
174
- // the truncation entirely). Case-folded: gemini shouts its reasons.
175
- const FINISH_SYNONYMS = new Map<string, Exclude<FinishReason, null>>([
176
- ["stop", "stop"], ["length", "length"], ["tool_calls", "tool_calls"], ["content_filter", "content_filter"],
177
- ["max_tokens", "length"], ["model_length", "length"], ["max_completion_tokens", "length"],
178
- ["end_turn", "stop"], ["stop_sequence", "stop"], ["eos_token", "stop"],
179
- ["tool_use", "tool_calls"],
180
- ["safety", "content_filter"], ["recitation", "content_filter"],
181
- ]);
182
- const normalizeFinishReason = (raw: string | null): FinishReason => {
183
- if (raw === null || raw.length === 0) return null;
184
- const hit = FINISH_SYNONYMS.get(raw.toLowerCase());
185
- if (hit !== undefined) return hit;
186
- emitWarningOnce(
187
- `unrecognized finish_reason "${raw}"; treated as no-signal (finishReason=null). If it denotes a token-cap hit, core's length-cap detection will miss it -- add it to FINISH_SYNONYMS.`,
188
- "PLURNK_FINISH_REASON_UNKNOWN",
189
- );
190
- return null;
191
- };
192
-
193
148
  // Shared budget→effort breakpoints (xai and google had identical copies).
194
149
  export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
195
150
  if (budget <= 1000) return "low";
@@ -235,9 +190,10 @@ const describeUnenforced = (v: Exclude<Verdict, { status: "accept" }>): string =
235
190
  return `grammar not enforced: output is an incomplete match of the transported grammar — a valid prefix of ${v.pos} code points that never terminated`;
236
191
  };
237
192
 
238
- export default class OpenAICompatProvider implements Provider {
193
+ export default class AiSdkProvider implements Provider {
239
194
  #model: string;
240
- #url: string;
195
+ #url: string | undefined;
196
+ #languageModel: LanguageModel | undefined;
241
197
  #fetchTimeoutMs: number;
242
198
  #streamIdleTimeoutMs: number | undefined;
243
199
  #headers: Record<string, string>;
@@ -254,10 +210,9 @@ export default class OpenAICompatProvider implements Provider {
254
210
  #dryBase: number | undefined;
255
211
  #dryAllowedLength: number | undefined;
256
212
  #repeatLastN: number | undefined;
257
- #retryDelayMs: number;
258
213
  #reasoningStyle: ReasoningStyle;
259
214
  #countTokens: (text: string) => number;
260
- #costFor: (usage: ProviderUsage) => number;
215
+ #calculateCost: (usage: ProviderUsage) => number;
261
216
  #source: string;
262
217
  #grammarStyle: GrammarStyle;
263
218
  #promptCacheKey: boolean;
@@ -265,7 +220,6 @@ export default class OpenAICompatProvider implements Provider {
265
220
  #gbnfDebug: boolean;
266
221
  #streaming: boolean;
267
222
  #firstPartyMetadata: boolean;
268
- #balanceMetaKey: string | undefined;
269
223
  #supportsSlotPinning: boolean;
270
224
  #slotCount: number | null;
271
225
  #retryAttempts: number;
@@ -283,9 +237,13 @@ export default class OpenAICompatProvider implements Provider {
283
237
  // the honest capability signal for every other backend.
284
238
  tokenize?: (text: string) => Promise<number[]>;
285
239
 
286
- constructor(config: OpenAICompatConfig) {
240
+ constructor(config: AiSdkProviderConfig) {
287
241
  this.#model = config.model;
288
242
  this.#url = config.url;
243
+ this.#languageModel = config.languageModel;
244
+ if ((this.#url === undefined) === (this.#languageModel === undefined)) {
245
+ throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
246
+ }
289
247
  this.#fetchTimeoutMs = config.fetchTimeoutMs;
290
248
  this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
291
249
  this.#headers = config.headers ?? {};
@@ -295,8 +253,8 @@ export default class OpenAICompatProvider implements Provider {
295
253
  // Loud guard: an out-of-date consumer (stale plugin dist) omitting the
296
254
  // required tuning fields must fail at construction, not silently send
297
255
  // undefined sampling on every grammar request.
298
- if (typeof config.temperature !== "number" || typeof config.repeatPenalty !== "number" || typeof config.retryDelayMs !== "number") {
299
- throw new Error(`${config.source ?? "provider"}: OpenAICompatConfig requires temperature + repeatPenalty + retryDelayMs (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY / _RETRY_DELAY) — rebuild against providers >= 0.33.0`);
256
+ if (typeof config.temperature !== "number" || typeof config.repeatPenalty !== "number") {
257
+ throw new Error(`${config.source ?? "provider"}: AiSdkProviderConfig requires temperature + repeatPenalty (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY)`);
300
258
  }
301
259
  this.#temperature = config.temperature;
302
260
  this.#repeatPenalty = config.repeatPenalty;
@@ -305,11 +263,10 @@ export default class OpenAICompatProvider implements Provider {
305
263
  this.#dryBase = config.dryBase;
306
264
  this.#dryAllowedLength = config.dryAllowedLength;
307
265
  this.#repeatLastN = config.repeatLastN;
308
- this.#retryDelayMs = config.retryDelayMs;
309
266
  this.#retryAttempts = config.retryAttempts;
310
267
  this.#reasoningStyle = config.reasoningStyle ?? "none";
311
268
  this.#countTokens = config.countTokens ?? heuristicTokens;
312
- this.#costFor = config.costFor ?? (() => 0);
269
+ this.#calculateCost = config.calculateCost ?? (() => 0);
313
270
  this.#source = config.source ?? "provider";
314
271
  this.#grammarStyle = config.grammarStyle ?? "none";
315
272
  this.#promptCacheKey = config.promptCacheKey ?? false;
@@ -320,7 +277,6 @@ export default class OpenAICompatProvider implements Provider {
320
277
  this.#apiKeyRejectedMessage = config.apiKeyRejectedMessage;
321
278
  this.#eosText = config.eosText;
322
279
  this.#hasApiKey = "Authorization" in this.#headers;
323
- this.#balanceMetaKey = config.balanceMetaKey;
324
280
  this.#supportsSlotPinning = config.supportsSlotPinning ?? false;
325
281
  this.#slotCount = config.slotCount ?? null;
326
282
  this.#topLogprobs = config.topLogprobs ?? null;
@@ -370,7 +326,7 @@ export default class OpenAICompatProvider implements Provider {
370
326
  get constrainsOutput(): boolean { return this.#grammarStyle !== "none"; }
371
327
 
372
328
  countTokens(text: string): number { return this.#countTokens(text); }
373
- costFor(usage: ProviderUsage): number { return this.#costFor(usage); }
329
+ calculateCost(usage: ProviderUsage): number { return this.#calculateCost(usage); }
374
330
 
375
331
  // Maps the reasoning INTENT (off | adaptive | on+budget, #33) to the
376
332
  // backend's wire mechanism — including under a transported grammar. The #32
@@ -566,18 +522,10 @@ export default class OpenAICompatProvider implements Provider {
566
522
  }
567
523
 
568
524
  // Per-turn metadata bag (#23): pass the backend's non-standard top-level fields
569
- // (the transport's `chunkMetadata`) through VERBATIM, then normalize the known
570
- // keys we hold a contract for — the spec's balance field → a validated
571
- // `balancePico` (finite pico-USD; dropped if non-numeric), renamed off its raw
572
- // key so the consumer reads one canonical name. Undefined when nothing's there;
573
- // the service merges this into its Turn metadata and filters what reaches clients.
525
+ // through verbatim. Providers do not reinterpret vendor currency or account
526
+ // metadata; a monetary value carries its own amount and currency.
574
527
  #buildMeta(chunkMetadata: Record<string, unknown>): Record<string, unknown> | undefined {
575
528
  const meta: Record<string, unknown> = { ...chunkMetadata };
576
- if (this.#balanceMetaKey !== undefined) {
577
- const raw = meta[this.#balanceMetaKey];
578
- delete meta[this.#balanceMetaKey];
579
- if (typeof raw === "number" && Number.isFinite(raw)) meta.balancePico = raw;
580
- }
581
529
  return Object.keys(meta).length > 0 ? meta : undefined;
582
530
  }
583
531
 
@@ -634,61 +582,65 @@ export default class OpenAICompatProvider implements Provider {
634
582
  ...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
635
583
  };
636
584
 
637
- // Transient-failure retry (#18). Each attempt gets a FRESH fetch timeout
638
- // (the budget is per-request, not shared across retries); the caller's
639
- // signal spans them all. Retry only the transient classifications, prefer
640
- // a server Retry-After over the backoff, and let the caller's abort cut
641
- // through both the in-flight request and the backoff sleep.
642
- const transport = this.#streaming ? chatCompletionStream : chatCompletion;
643
-
644
585
  // Per-request headers = static auth/routing + any first-party telemetry.
645
586
  const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
646
587
  const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
647
- const transportRetries: Array<{ attempt: number; kind: string; elapsedMs: number; message: string }> = [];
648
588
  let raw;
649
- for (let attempt = 0; ; attempt++) {
650
- const attemptStarted = performance.now();
651
- const timeoutSignal = AbortSignal.timeout(this.#fetchTimeoutMs);
652
- const effectiveSignal = signal !== undefined ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal;
653
- try {
654
- raw = await transport({
655
- url: this.#url,
589
+ try {
590
+ raw = this.#languageModel === undefined
591
+ ? await executeOpenAICompatible({
592
+ url: this.#url!,
593
+ model: this.#model,
656
594
  headers,
657
595
  body,
658
- signal: effectiveSignal,
596
+ messages,
597
+ signal,
659
598
  fetch: this.#fetch,
599
+ fetchTimeoutMs: this.#fetchTimeoutMs,
600
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
601
+ retryAttempts: this.#retryAttempts,
602
+ streaming: this.#streaming,
660
603
  captureRawBody: this.#rawBody,
604
+ })
605
+ : await executeAiSdkModel({
606
+ languageModel: this.#languageModel,
607
+ headers,
608
+ messages,
609
+ signal,
610
+ fetchTimeoutMs: this.#fetchTimeoutMs,
661
611
  streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
612
+ retryAttempts: this.#retryAttempts,
613
+ streaming: this.#streaming,
614
+ captureRawBody: this.#rawBody,
615
+ temperature: this.#tuningFloors
616
+ ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
617
+ : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
618
+ topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
619
+ topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
620
+ presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
621
+ frequencyPenalty: typeof sampling?.frequency_penalty === "number"
622
+ ? sampling.frequency_penalty
623
+ : this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
624
+ stopSequences: typeof sampling?.stop === "string"
625
+ ? [sampling.stop]
626
+ : Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
627
+ ? sampling.stop
628
+ : undefined,
629
+ seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
630
+ maxOutputTokens: maxTokens,
631
+ reasoning: this.#reasoning.mode === "off"
632
+ ? "none"
633
+ : this.#reasoning.mode === "adaptive"
634
+ ? "provider-default"
635
+ : effortFromBudget(this.#reasoning.budget!),
662
636
  });
663
- break;
664
- } catch (err) {
665
- // Caller-initiated abort is cancellation — never retried or wrapped.
666
- if (signal?.aborted) throw err;
667
- const { kind } = classifyProviderError(err);
668
- // #543: Cloudflare/CDN edge codes (520-527) classify as network_failure
669
- // but fail-fast - a retry just re-incurs the same origin/edge timeout.
670
- const edgeTimeout = err instanceof OpenAiHttpError && isEdgeStatus(err.status);
671
- // Terminal kind, edge failure, or budget spent -> surface the failure.
672
- if (!RETRYABLE.has(kind) || edgeTimeout || attempt >= this.#retryAttempts) {
673
- const pe = toProviderError(err, this.#source);
674
- // #537 case 2: a 401/403 with a key PRESENT is a rejected key, not a
675
- // transport failure — surface the distinct, actionable hint (kind stays
676
- // "unauthorized", so core's routing is unchanged) rather than the raw
677
- // upstream JSON. Distinct from the unset-key throw at construction.
678
- if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
679
- throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
680
- }
681
- throw pe;
682
- }
683
- transportRetries.push({
684
- attempt: attempt + 1,
685
- kind,
686
- elapsedMs: Math.round(performance.now() - attemptStarted),
687
- message: err instanceof Error ? err.message : String(err),
688
- });
689
- const retryAfter = err instanceof OpenAiHttpError ? err.retryAfter : null;
690
- await sleepWithAbort(retryAfter ?? this.#retryDelayMs * 2 ** attempt, signal);
637
+ } catch (err) {
638
+ if (signal?.aborted) throw err;
639
+ const pe = toProviderError(err, this.#source);
640
+ if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
641
+ throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
691
642
  }
643
+ throw pe;
692
644
  }
693
645
 
694
646
  // #539: llama-server --special renders EOG tokens as text, so a turn ending
@@ -707,7 +659,7 @@ export default class OpenAICompatProvider implements Provider {
707
659
  // Discard/retry/escalate/self-correct is the consumer's policy.
708
660
  let telemetry: TelemetryEvent[] | undefined;
709
661
  let railsMeta: Record<string, unknown> | undefined;
710
- const usage = normalizeUsage(raw.usage, raw.reasoning_content, raw.content);
662
+ const usage = raw.usage;
711
663
  const observedGrammar = sendGrammar ?? (wantGrammar && this.#gbnfDebug ? grammar : undefined);
712
664
  if (observedGrammar !== undefined) {
713
665
  const verdict = this.#grammarVerdict(observedGrammar, raw.content);
@@ -725,7 +677,7 @@ export default class OpenAICompatProvider implements Provider {
725
677
  // invisible, billed (12,288 billed vs 1,033 chars visible, live).
726
678
  // countTokens OVERCOUNTS text (chars/2 upper bound), so billed
727
679
  // exceeding visible-plus-slack is real vanishing, not estimator noise.
728
- const visible = this.#countTokens(raw.content) + this.#countTokens(raw.reasoning_content);
680
+ const visible = this.#countTokens(raw.content) + this.#countTokens(raw.reasoning);
729
681
  if (sendGrammar !== undefined && usage.completion > visible + 64) {
730
682
  (telemetry ??= []).push({
731
683
  source: this.#source,
@@ -736,30 +688,23 @@ export default class OpenAICompatProvider implements Provider {
736
688
  }
737
689
  }
738
690
 
739
- const builtMeta = this.#buildMeta(raw.chunkMetadata);
740
- const retryMeta = transportRetries.length > 0 ? { transportRetries } : undefined;
741
- const meta = railsMeta !== undefined || retryMeta !== undefined
742
- ? { ...builtMeta, ...railsMeta, ...retryMeta }
743
- : builtMeta;
744
-
745
- // #36: surface per-token logprobs + their mean when the backend returned
746
- // them (only possible when the flag requested them). Absent otherwise —
747
- // never synthesized.
748
- const logprobs = raw.logprobs !== null && raw.logprobs.length > 0 ? raw.logprobs : undefined;
691
+ const builtMeta = this.#buildMeta(raw.metadata);
692
+ const meta = railsMeta !== undefined ? { ...builtMeta, ...railsMeta } : builtMeta;
693
+ const logprobs = raw.logprobs.length > 0 ? raw.logprobs : undefined;
749
694
  const meanLogprob = logprobs !== undefined
750
- ? logprobs.reduce((sum, t) => sum + t.logprob, 0) / logprobs.length
695
+ ? logprobs.reduce((sum, token) => sum + token.logprob, 0) / logprobs.length
751
696
  : undefined;
752
697
 
753
698
  return {
754
699
  assistant: {
755
700
  content: raw.content,
756
- reasoning: raw.reasoning_content.length > 0 ? raw.reasoning_content : null,
757
- // #482: sealed relay blobs ride only when present — same absence
758
- // discipline as logprobs (never synthesized, never empty-array).
759
- ...(raw.reasoning_encrypted.length > 0 ? { reasoningEncrypted: raw.reasoning_encrypted } : {}),
701
+ reasoning: raw.reasoning.length > 0 ? raw.reasoning : null,
702
+ ...(raw.reasoningEncrypted.length > 0
703
+ ? { reasoningEncrypted: raw.reasoningEncrypted }
704
+ : {}),
760
705
  usage,
761
- finishReason: normalizeFinishReason(raw.finish_reason),
762
- model: raw.model ?? this.#model,
706
+ finishReason: raw.finishReason,
707
+ model: raw.model,
763
708
  ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
764
709
  },
765
710
  assistantRaw: raw,
package/src/Mock.test.ts CHANGED
@@ -31,9 +31,9 @@ test("Mock: countTokens('') is 0; non-empty is a positive integer", () => {
31
31
  assert.ok(Number.isInteger(n) && n > 0);
32
32
  });
33
33
 
34
- test("Mock: costFor zero usage is 0 (free)", () => {
34
+ test("Mock: calculateCost zero usage is 0 (free)", () => {
35
35
  const m = build();
36
- assert.equal(m.costFor({ prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 }), 0);
36
+ assert.equal(m.calculateCost({ prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 }), 0);
37
37
  });
38
38
 
39
39
  // — Transport (SPEC §10.7, §10.10) —
@@ -109,7 +109,7 @@ test("Mock: exhausted queue throws a specific error", async () => {
109
109
  // -- #507: the reserve surface lives on the Provider CONTRACT, and Mock drives core's partition suite --
110
110
 
111
111
  test("#507 the reserve getters are on the Provider interface (not just the concrete class)", () => {
112
- // Typing against the CONTRACT is the check a getter-only-on-OpenAICompat surface fails.
112
+ // Typing against the contract catches a getter-only concrete surface.
113
113
  const prevR = process.env.PLURNK_PROVIDERS_REASONING_RESERVE;
114
114
  try {
115
115
  process.env.PLURNK_PROVIDERS_REASONING_RESERVE = "10%";
package/src/Mock.ts CHANGED
@@ -68,7 +68,7 @@ export default class Mock implements Provider {
68
68
  }
69
69
 
70
70
  // Mock is free.
71
- costFor(_usage: ProviderUsage): number { return 0; }
71
+ calculateCost(_usage: ProviderUsage): number { return 0; }
72
72
 
73
73
  async generate({ signal }: { messages: ChatMessage[]; workerId?: string; signal?: AbortSignal }): Promise<{ assistant: MockReturnedAssistant; assistantRaw: unknown }> {
74
74
  // Honor abort before consuming the queue — an aborted call makes no
package/src/Pool.test.ts CHANGED
@@ -28,7 +28,7 @@ const backend = (opts: FakeOpts = {}) => {
28
28
  ...(opts.completionReserve !== undefined ? { completionReserve: opts.completionReserve } : {}),
29
29
  ...(opts.tokenize ? { tokenize: async (t: string) => [t.length] } : {}),
30
30
  countTokens: (t: string) => t.length,
31
- costFor: () => opts.cost ?? 0,
31
+ calculateCost: () => opts.cost ?? 0,
32
32
  generate: async (args: Parameters<Provider["generate"]>[0]): Promise<ProviderResponse> => {
33
33
  served.push(args.workerId);
34
34
  if (opts.throws !== undefined) throw opts.throws;
@@ -88,10 +88,10 @@ test("Pool: tokenize is exposed iff every backend has it", () => {
88
88
  assert.equal(new Pool([backend({ tokenize: true }).b, backend({ tokenize: false }).b]).tokenize, undefined);
89
89
  });
90
90
 
91
- test("Pool: countTokens + costFor delegate to a backend", () => {
91
+ test("Pool: countTokens + calculateCost delegate to a backend", () => {
92
92
  const p = new Pool([backend({ cost: 42 }).b]);
93
93
  assert.equal(p.countTokens("abcd"), 4);
94
- assert.equal(p.costFor({ prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 }), 42);
94
+ assert.equal(p.calculateCost({ prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 }), 42);
95
95
  });
96
96
 
97
97
  // --- dispatch: round-robin + affinity ---
package/src/Pool.ts CHANGED
@@ -83,7 +83,7 @@ export default class Pool implements Provider {
83
83
  }
84
84
 
85
85
  countTokens(text: string): number { return this.#backends[0].countTokens(text); }
86
- costFor(usage: ProviderUsage): number { return this.#backends[0].costFor(usage); }
86
+ calculateCost(usage: ProviderUsage): number { return this.#backends[0].calculateCost(usage); }
87
87
 
88
88
  // --- dispatch ---
89
89
 
@@ -2,7 +2,6 @@ import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
3
  import { instantiateProvider, loadActiveProvider, resetDiscoveryCache } from "./ProviderRegistry.ts";
4
4
 
5
- const fakeProvider = { contextWindow: 1, model: "m", countTokens: () => 0, costFor: () => 0, generate: async () => { throw new Error("unused"); } };
6
5
  const mapOf = (entries: Record<string, string>, skipped: Record<string, string> = {}) =>
7
6
  async () => ({ registry: new Map(Object.entries(entries)), skipped: new Map(Object.entries(skipped)), attributions: new Map<string, string | string[]>() });
8
7
 
@@ -10,16 +9,16 @@ const mapOf = (entries: Record<string, string>, skipped: Record<string, string>
10
9
  // exercise the resolution + two-tier instantiation this module owns; the active
11
10
  // alias is driven end-to-end by loadActiveProvider below.
12
11
 
13
- // — two-tier instantiation (SPEC §5) —
12
+ // — provider resolution (SPEC §5) —
14
13
 
15
14
  const fullEnv = Object.freeze({
16
15
  PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000",
17
16
  PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
18
- PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
17
+ PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0", PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "1",
19
18
  OPENAI_BASE_URL: "http://x",
20
19
  });
21
20
 
22
- test("instantiateProvider: standard name resolves in-framework, no scan, no import", async () => {
21
+ test("instantiateProvider: cataloged name resolves in-framework, no scan, no import", async () => {
23
22
  resetDiscoveryCache();
24
23
  mock.method(globalThis, "fetch", async (url: string) => {
25
24
  if (String(url).endsWith("/models")) return new Response(JSON.stringify({ data: [] }), { status: 200 });
@@ -36,24 +35,33 @@ test("instantiateProvider: standard name resolves in-framework, no scan, no impo
36
35
  mock.restoreAll();
37
36
  });
38
37
 
39
- test("instantiateProvider: bespoke name resolves via the scan and imports the discovered package", async () => {
38
+ test("instantiateProvider: an installed AI SDK provider resolves through discovery", async () => {
40
39
  resetDiscoveryCache();
41
40
  const calls: unknown[] = [];
42
- const p = await instantiateProvider("openrouter", { ...fullEnv }, "anthropic/claude-opus-latest",
43
- async (specifier) => { calls.push(specifier); return { default: { fromEnv: async (_e: NodeJS.ProcessEnv, model: string) => { calls.push(model); return fakeProvider; } } }; },
44
- mapOf({ openrouter: "@plurnk/plurnk-providers-openrouter" }));
45
- assert.equal(p, fakeProvider);
46
- assert.deepEqual(calls, ["@plurnk/plurnk-providers-openrouter", "anthropic/claude-opus-latest"]);
41
+ const p = await instantiateProvider("acme", { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "model-a",
42
+ async (specifier) => {
43
+ calls.push(specifier);
44
+ return { default: { languageModel: (model: string) => { calls.push(model); return {} as never; } } };
45
+ },
46
+ mapOf({ acme: "@acme/ai-provider" }));
47
+ assert.equal(p.model, "model-a");
48
+ assert.equal(p.contextWindow, 8192);
49
+ assert.deepEqual(calls, ["@acme/ai-provider", "model-a"]);
47
50
  });
48
51
 
49
- test("instantiateProvider: a per-alias baseUrl is passed to a bespoke factory as the 3rd-arg option", async () => {
52
+ test("instantiateProvider: a per-alias baseUrl drives the built-in Ollama probe", async () => {
50
53
  resetDiscoveryCache();
51
- let received: unknown;
54
+ const calls: string[] = [];
55
+ mock.method(globalThis, "fetch", async (url: string) => {
56
+ calls.push(String(url));
57
+ return new Response(JSON.stringify({ model_info: { "qwen.context_length": 32768 } }));
58
+ });
52
59
  await instantiateProvider("ollama", { ...fullEnv }, "qwen2.5-coder",
53
- async () => ({ default: { fromEnv: async (_e: NodeJS.ProcessEnv, _m: string, options?: unknown) => { received = options; return fakeProvider; } } }),
54
- mapOf({ ollama: "@plurnk/plurnk-providers-ollama" }),
60
+ async () => ({}),
61
+ mapOf({}),
55
62
  "http://nook:11434");
56
- assert.deepEqual(received, { baseUrl: "http://nook:11434" });
63
+ assert.deepEqual(calls, ["http://nook:11434/api/show"]);
64
+ mock.restoreAll();
57
65
  });
58
66
 
59
67
  test("instantiateProvider: a per-alias baseUrl drives the standard openai probe to the override host", async () => {
@@ -74,14 +82,14 @@ test("instantiateProvider: a per-alias baseUrl drives the standard openai probe
74
82
  test("instantiateProvider: a THIRD-PARTY scope is discovered — name maps to its package", async () => {
75
83
  resetDiscoveryCache();
76
84
  const imports: string[] = [];
77
- const p = await instantiateProvider("foo", { ...fullEnv }, "m",
78
- async (specifier) => { imports.push(specifier); return { default: { fromEnv: async () => fakeProvider } }; },
85
+ const p = await instantiateProvider("foo", { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW: "4096" }, "m",
86
+ async (specifier) => { imports.push(specifier); return { default: { languageModel: () => ({} as never) } }; },
79
87
  mapOf({ foo: "@acme/acme-provider-foo" }));
80
- assert.equal(p, fakeProvider);
88
+ assert.equal(p.model, "m");
81
89
  assert.deepEqual(imports, ["@acme/acme-provider-foo"]); // not an @plurnk/ specifier
82
90
  });
83
91
 
84
- test("instantiateProvider: a standard name is authoritative — a scanned package of the same name is shadowed", async () => {
92
+ test("instantiateProvider: a cataloged name is authoritative — a scanned package of the same name is shadowed", async () => {
85
93
  resetDiscoveryCache();
86
94
  mock.method(globalThis, "fetch", async (url: string) => {
87
95
  if (String(url).endsWith("/models")) return new Response(JSON.stringify({ data: [] }), { status: 200 });
@@ -100,7 +108,7 @@ test("instantiateProvider: unknown provider throws — no standard, no discovere
100
108
  resetDiscoveryCache();
101
109
  await assert.rejects(
102
110
  () => instantiateProvider("nope", { ...fullEnv }, "m", async () => ({}), mapOf({})),
103
- /unknown provider "nope": not a standard provider, and no installed package declares plurnk\.kind:"provider" with name "nope"/,
111
+ /unknown provider "nope"/,
104
112
  );
105
113
  });
106
114
 
@@ -116,13 +124,13 @@ test("instantiateProvider: an untrusted (skipped) provider gives a precise error
116
124
  assert.deepEqual(imports, []); // never imported an untrusted package
117
125
  });
118
126
 
119
- test("instantiateProvider: discovered package without a fromEnv factory throws (factory shape, SPEC )", async () => {
127
+ test("instantiateProvider: discovered package must export an AI SDK provider", async () => {
120
128
  resetDiscoveryCache();
121
129
  await assert.rejects(
122
130
  () => instantiateProvider("broken", { ...fullEnv }, "m",
123
131
  async () => ({ default: {} }),
124
132
  mapOf({ broken: "@acme/acme-provider-broken" })),
125
- /@acme\/acme-provider-broken default export is not a Provider factory/,
133
+ /@acme\/acme-provider-broken default export is not an AI SDK provider/,
126
134
  );
127
135
  });
128
136
 
@@ -174,6 +182,8 @@ test("#622: two Fireworks aliases independently select default and priority serv
174
182
  FIREWORKS_BASE_URL: "https://api.fireworks.ai/inference/v1",
175
183
  FIREWORKS_API_KEY: "fw",
176
184
  PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
185
+ PLURNK_PROVIDERS_PROVIDER_FIREWORKS_REASONING_STYLE: "effort_explicit",
186
+ PLURNK_PROVIDERS_TOP_LOGPROBS: "2",
177
187
  PLURNK_PROVIDERS_SERVICE_TIER_fast: "priority",
178
188
  PLURNK_PROVIDERS_SERVICE_TIER_standard: "default",
179
189
  };
@@ -184,6 +194,9 @@ test("#622: two Fireworks aliases independently select default and priority serv
184
194
  await fast.generate({ workerId: "fast-worker", messages: [] });
185
195
  await standard.generate({ workerId: "standard-worker", messages: [] });
186
196
  assert.deepEqual(bodies.map((body) => body.service_tier), ["priority", "default"]);
197
+ assert.deepEqual(bodies.map((body) => body.prompt_cache_key), ["fast-worker", "standard-worker"]);
198
+ assert.deepEqual(bodies.map((body) => body.reasoning_effort), ["none", "none"]);
199
+ assert.deepEqual(bodies.map((body) => body.top_logprobs), [2, 2]);
187
200
  assert.deepEqual(bodies.map((body) => body.model), [
188
201
  "accounts/fireworks/routers/glm-5p2-fast",
189
202
  "accounts/fireworks/models/deepseek-v4-pro",
@@ -191,13 +204,13 @@ test("#622: two Fireworks aliases independently select default and priority serv
191
204
  mock.restoreAll();
192
205
  });
193
206
 
194
- test("loadActiveProvider: resolves the alias cascade end-to-end via the scan", async () => {
207
+ test("loadActiveProvider: resolves the alias cascade to an installed AI SDK provider", async () => {
195
208
  resetDiscoveryCache();
196
- const env = { ...fullEnv, PLURNK_MODEL: "opus", PLURNK_MODEL_opus: "openrouter/anthropic/claude-opus-latest" } as NodeJS.ProcessEnv;
209
+ const env = { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW: "4096", PLURNK_MODEL: "custom", PLURNK_MODEL_custom: "acme/model-a" } as NodeJS.ProcessEnv;
197
210
  const p = await loadActiveProvider(env,
198
- async () => ({ default: { fromEnv: async () => fakeProvider } }),
199
- mapOf({ openrouter: "@plurnk/plurnk-providers-openrouter" }));
200
- assert.equal(p, fakeProvider);
211
+ async () => ({ default: { languageModel: () => ({} as never) } }),
212
+ mapOf({ acme: "@acme/ai-provider" }));
213
+ assert.equal(p.model, "model-a");
201
214
  });
202
215
 
203
216
  test("loadActiveProvider: throws a named error when no alias is active", async () => {