@plurnk/plurnk-providers 1.3.5 → 1.3.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/.env.defaults +35 -45
  2. package/README.md +44 -53
  3. package/SPEC.md +215 -371
  4. package/dist/AiSdkProvider.d.ts +78 -0
  5. package/dist/AiSdkProvider.d.ts.map +1 -0
  6. package/dist/AiSdkProvider.js +591 -0
  7. package/dist/AiSdkProvider.js.map +1 -0
  8. package/dist/OpenAICompat.d.ts +1 -2
  9. package/dist/OpenAICompat.d.ts.map +1 -1
  10. package/dist/OpenAICompat.js +39 -117
  11. package/dist/OpenAICompat.js.map +1 -1
  12. package/dist/ProviderRegistry.d.ts.map +1 -1
  13. package/dist/ProviderRegistry.js +37 -24
  14. package/dist/ProviderRegistry.js.map +1 -1
  15. package/dist/aiSdkTransport.d.ts +52 -0
  16. package/dist/aiSdkTransport.d.ts.map +1 -0
  17. package/dist/aiSdkTransport.js +294 -0
  18. package/dist/aiSdkTransport.js.map +1 -0
  19. package/dist/catalogProvider.d.ts +15 -0
  20. package/dist/catalogProvider.d.ts.map +1 -0
  21. package/dist/catalogProvider.js +103 -0
  22. package/dist/catalogProvider.js.map +1 -0
  23. package/dist/compatibleProvider.d.ts +3 -0
  24. package/dist/compatibleProvider.d.ts.map +1 -0
  25. package/dist/compatibleProvider.js +146 -0
  26. package/dist/compatibleProvider.js.map +1 -0
  27. package/dist/discover.d.ts.map +1 -1
  28. package/dist/discover.js.map +1 -1
  29. package/dist/env.d.ts +1 -0
  30. package/dist/env.d.ts.map +1 -1
  31. package/dist/env.js +13 -6
  32. package/dist/env.js.map +1 -1
  33. package/dist/index.d.ts +4 -6
  34. package/dist/index.d.ts.map +1 -1
  35. package/dist/index.js +3 -7
  36. package/dist/index.js.map +1 -1
  37. package/dist/ollama.d.ts +3 -0
  38. package/dist/ollama.d.ts.map +1 -0
  39. package/dist/ollama.js +39 -0
  40. package/dist/ollama.js.map +1 -0
  41. package/dist/openai.d.ts +2 -4
  42. package/dist/openai.d.ts.map +1 -1
  43. package/dist/openai.js +1 -2
  44. package/dist/openai.js.map +1 -1
  45. package/dist/sdkModels.d.ts +13 -0
  46. package/dist/sdkModels.d.ts.map +1 -0
  47. package/dist/sdkModels.js +153 -0
  48. package/dist/sdkModels.js.map +1 -0
  49. package/dist/standardProviders.d.ts.map +1 -1
  50. package/dist/standardProviders.js +0 -1
  51. package/dist/standardProviders.js.map +1 -1
  52. package/dist/telemetry.d.ts.map +1 -1
  53. package/dist/telemetry.js +20 -9
  54. package/dist/telemetry.js.map +1 -1
  55. package/dist/types.d.ts +3 -2
  56. package/dist/types.d.ts.map +1 -1
  57. package/package.json +18 -10
  58. package/src/{OpenAICompat.test.ts → AiSdkProvider.test.ts} +193 -152
  59. package/src/{OpenAICompat.ts → AiSdkProvider.ts} +83 -127
  60. package/src/Mock.test.ts +1 -1
  61. package/src/ProviderRegistry.test.ts +40 -27
  62. package/src/ProviderRegistry.ts +35 -24
  63. package/src/aiSdkTransport.test.ts +253 -0
  64. package/src/aiSdkTransport.ts +369 -0
  65. package/src/boundaries.test.ts +2 -2
  66. package/src/catalogProvider.test.ts +100 -0
  67. package/src/catalogProvider.ts +151 -0
  68. package/src/compatibleProvider.test.ts +44 -0
  69. package/src/compatibleProvider.ts +205 -0
  70. package/src/discover.test.ts +12 -12
  71. package/src/discover.ts +3 -6
  72. package/src/env.ts +14 -6
  73. package/src/index.ts +6 -10
  74. package/src/ollama.ts +63 -0
  75. package/src/openai.ts +2 -8
  76. package/src/sdkModels.test.ts +47 -0
  77. package/src/sdkModels.ts +194 -0
  78. package/src/telemetry.test.ts +17 -10
  79. package/src/telemetry.ts +22 -14
  80. package/src/types.ts +5 -8
  81. package/src/aiSdkAdapter.spike.test.ts +0 -242
  82. package/src/openaiStream.ts +0 -310
  83. package/src/standardProviders.test.ts +0 -939
  84. package/src/standardProviders.ts +0 -631
@@ -1,21 +1,21 @@
1
- // Shared OpenAI-compatible provider. Implements the universal generate()
1
+ // PLURNK adapter over an AI SDK language model. Implements the universal generate()
2
2
  // spine — signal merging, the SSE call, usage mapping, finishReason
3
3
  // normalization, response assembly — that every sibling had duplicated.
4
4
  //
5
- // Composition, not inheritance: the per-provider deltas (resolved URL, auth
6
- // headers, reasoning translation style, tokenizer, cost) arrive as config.
7
- // A sibling's fromEnv probes whatever it needs (catalog, pricing, context
8
- // window), builds the config, and returns `new OpenAICompatProvider(config)`.
9
- // Pure-config providers come from ./standardProviders.ts with no sibling at all.
5
+ // Composition, not inheritance: an official AI SDK language model supplies the
6
+ // ordinary vendor protocol. The compatible URL path remains only for PLURNK
7
+ // extensions and local endpoint probes the SDK cannot represent.
10
8
 
11
9
  import type { ChatMessage, FinishReason, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
12
10
  import type { Reasoning, ReserveSpec } from "./env.ts";
13
- import { chatCompletionStream, chatCompletion, OpenAiHttpError, isEdgeStatus, type ProviderFetch, type StreamResponse } from "./openaiStream.ts";
14
- import { normalizeUsage } from "./usage.ts";
15
- import { toProviderError, classifyProviderError, ProviderError, type TelemetryEvent } from "./telemetry.ts";
11
+ import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.ts";
12
+ import type { LanguageModel } from "ai";
13
+ import { toProviderError, ProviderError, type TelemetryEvent } from "./telemetry.ts";
16
14
  import { validateGbnf, type Verdict } from "@plurnk/gbnf";
17
15
  import { emitWarningOnce } from "./warnings.ts";
18
16
 
17
+ export type ProviderFetch = typeof globalThis.fetch;
18
+
19
19
  // How the reasoning intent (PLURNK_PROVIDERS_REASONING: off | adaptive | on, plus
20
20
  // REASONING_BUDGET iff on — #32/#33) translates to each backend's wire mechanism
21
21
  // (SPEC §4); the per-style mapping lives in #reasoningBody. Non-obvious ones:
@@ -33,9 +33,10 @@ export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" |
33
33
  // service-managed constrained sampling; endpoint-owned settings are not inferred.
34
34
  export type GrammarStyle = "none" | "llamacpp";
35
35
 
36
- export type OpenAICompatConfig = {
36
+ export type AiSdkProviderConfig = {
37
37
  model: string;
38
- url: string; // fully-resolved chat-completions URL
38
+ url?: string; // OpenAI-compatible chat-completions URL
39
+ languageModel?: LanguageModel; // native AI SDK provider model
39
40
  fetchTimeoutMs: number;
40
41
  streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
41
42
  headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
@@ -87,8 +88,7 @@ export type OpenAICompatConfig = {
87
88
  // DEFAULT for EVERY request, spread UNDER caller sampling (#30/endpoint#7).
88
89
  // `repeatPenalty` is the FLOOR the provider manages wherever a grammar rides
89
90
  // (greedy-under-mask loops without it, #9) — the VALUE is operator config;
90
- // WHERE it applies stays mechanism. `retryDelayMs` is the transient-retry
91
- // backoff base (attempt N waits retryDelayMs * 2^(N-1); Retry-After wins).
91
+ // WHERE it applies stays mechanism.
92
92
  temperature: number;
93
93
  repeatPenalty: number;
94
94
  // #426: anti-degeneration guard on the CLOUD path (grammarStyle "none"), where the
@@ -106,7 +106,6 @@ export type OpenAICompatConfig = {
106
106
  dryBase?: number;
107
107
  dryAllowedLength?: number;
108
108
  repeatLastN?: number;
109
- retryDelayMs: number;
110
109
  // Transient-failure retry budget — REQUIRED, no in-code default
111
110
  // (PLURNK_PROVIDERS_RETRY_ATTEMPTS, a non-negative int): 0 = surface the
112
111
  // first failure; N = up to N retries on a transient error (§4, #18).
@@ -134,13 +133,6 @@ export type OpenAICompatConfig = {
134
133
  tuningFloors?: boolean;
135
134
  };
136
135
 
137
- // Transient classifications worth retrying: rate_limit (429) and network_failure
138
- // (5xx, timeout, connection reset) are transport; grammar_invalid (a 422 output
139
- // reject a fresh sample may satisfy, #548) rides the same bounded budget.
140
- // unauthorized, quota_exceeded, invalid_response, model_refused are terminal —
141
- // retrying just burns time and budget.
142
- const RETRYABLE: ReadonlySet<string> = new Set(["rate_limit", "network_failure", "grammar_invalid"]);
143
-
144
136
  // #539: drop trailing occurrences of a server-rendered EOG marker. llama-server
145
137
  // under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
146
138
  // trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
@@ -153,42 +145,6 @@ const stripTrailingSpecial = (content: string, marker: string): string => {
153
145
  return out;
154
146
  };
155
147
 
156
- // Sleep that rejects the moment `signal` aborts (caller cancellation must not
157
- // wait out a backoff). Resolves normally on timeout.
158
- const sleepWithAbort = (ms: number, signal: AbortSignal | undefined): Promise<void> =>
159
- new Promise((resolve, reject) => {
160
- if (signal?.aborted) { reject(signal.reason); return; }
161
- const timer = setTimeout(resolve, ms);
162
- signal?.addEventListener("abort", () => { clearTimeout(timer); reject(signal.reason); }, { once: true });
163
- });
164
-
165
- // SPEC §2 closed set. The four canonical values pass through; known per-backend
166
- // synonyms translate INTO them (anthropic max_tokens/end_turn, gemini MAX_TOKENS/
167
- // SAFETY/RECITATION) so a token-cap hit canonicalizes to "length" whatever the
168
- // backend names it -- core's `finishReason === "length"` truncation check (#425)
169
- // is then an invariant by construction, not a convention each backend must
170
- // independently honor. A non-empty value outside both the set and the table
171
- // collapses to null AND warns once, so a new backend's unmapped cap string
172
- // surfaces instead of silently becoming "no signal" (which would make core miss
173
- // the truncation entirely). Case-folded: gemini shouts its reasons.
174
- const FINISH_SYNONYMS = new Map<string, Exclude<FinishReason, null>>([
175
- ["stop", "stop"], ["length", "length"], ["tool_calls", "tool_calls"], ["content_filter", "content_filter"],
176
- ["max_tokens", "length"], ["model_length", "length"], ["max_completion_tokens", "length"],
177
- ["end_turn", "stop"], ["stop_sequence", "stop"], ["eos_token", "stop"],
178
- ["tool_use", "tool_calls"],
179
- ["safety", "content_filter"], ["recitation", "content_filter"],
180
- ]);
181
- const normalizeFinishReason = (raw: string | null): FinishReason => {
182
- if (raw === null || raw.length === 0) return null;
183
- const hit = FINISH_SYNONYMS.get(raw.toLowerCase());
184
- if (hit !== undefined) return hit;
185
- emitWarningOnce(
186
- `unrecognized finish_reason "${raw}"; treated as no-signal (finishReason=null). If it denotes a token-cap hit, core's length-cap detection will miss it -- add it to FINISH_SYNONYMS.`,
187
- "PLURNK_FINISH_REASON_UNKNOWN",
188
- );
189
- return null;
190
- };
191
-
192
148
  // Shared budget→effort breakpoints (xai and google had identical copies).
193
149
  export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
194
150
  if (budget <= 1000) return "low";
@@ -234,9 +190,10 @@ const describeUnenforced = (v: Exclude<Verdict, { status: "accept" }>): string =
234
190
  return `grammar not enforced: output is an incomplete match of the transported grammar — a valid prefix of ${v.pos} code points that never terminated`;
235
191
  };
236
192
 
237
- export default class OpenAICompatProvider implements Provider {
193
+ export default class AiSdkProvider implements Provider {
238
194
  #model: string;
239
- #url: string;
195
+ #url: string | undefined;
196
+ #languageModel: LanguageModel | undefined;
240
197
  #fetchTimeoutMs: number;
241
198
  #streamIdleTimeoutMs: number | undefined;
242
199
  #headers: Record<string, string>;
@@ -253,7 +210,6 @@ export default class OpenAICompatProvider implements Provider {
253
210
  #dryBase: number | undefined;
254
211
  #dryAllowedLength: number | undefined;
255
212
  #repeatLastN: number | undefined;
256
- #retryDelayMs: number;
257
213
  #reasoningStyle: ReasoningStyle;
258
214
  #countTokens: (text: string) => number;
259
215
  #calculateCost: (usage: ProviderUsage) => number;
@@ -281,9 +237,13 @@ export default class OpenAICompatProvider implements Provider {
281
237
  // the honest capability signal for every other backend.
282
238
  tokenize?: (text: string) => Promise<number[]>;
283
239
 
284
- constructor(config: OpenAICompatConfig) {
240
+ constructor(config: AiSdkProviderConfig) {
285
241
  this.#model = config.model;
286
242
  this.#url = config.url;
243
+ this.#languageModel = config.languageModel;
244
+ if ((this.#url === undefined) === (this.#languageModel === undefined)) {
245
+ throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
246
+ }
287
247
  this.#fetchTimeoutMs = config.fetchTimeoutMs;
288
248
  this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
289
249
  this.#headers = config.headers ?? {};
@@ -293,8 +253,8 @@ export default class OpenAICompatProvider implements Provider {
293
253
  // Loud guard: an out-of-date consumer (stale plugin dist) omitting the
294
254
  // required tuning fields must fail at construction, not silently send
295
255
  // undefined sampling on every grammar request.
296
- if (typeof config.temperature !== "number" || typeof config.repeatPenalty !== "number" || typeof config.retryDelayMs !== "number") {
297
- throw new Error(`${config.source ?? "provider"}: OpenAICompatConfig requires temperature + repeatPenalty + retryDelayMs (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY / _RETRY_DELAY) — rebuild against providers >= 0.33.0`);
256
+ if (typeof config.temperature !== "number" || typeof config.repeatPenalty !== "number") {
257
+ throw new Error(`${config.source ?? "provider"}: AiSdkProviderConfig requires temperature + repeatPenalty (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY)`);
298
258
  }
299
259
  this.#temperature = config.temperature;
300
260
  this.#repeatPenalty = config.repeatPenalty;
@@ -303,7 +263,6 @@ export default class OpenAICompatProvider implements Provider {
303
263
  this.#dryBase = config.dryBase;
304
264
  this.#dryAllowedLength = config.dryAllowedLength;
305
265
  this.#repeatLastN = config.repeatLastN;
306
- this.#retryDelayMs = config.retryDelayMs;
307
266
  this.#retryAttempts = config.retryAttempts;
308
267
  this.#reasoningStyle = config.reasoningStyle ?? "none";
309
268
  this.#countTokens = config.countTokens ?? heuristicTokens;
@@ -623,61 +582,65 @@ export default class OpenAICompatProvider implements Provider {
623
582
  ...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
624
583
  };
625
584
 
626
- // Transient-failure retry (#18). Each attempt gets a FRESH fetch timeout
627
- // (the budget is per-request, not shared across retries); the caller's
628
- // signal spans them all. Retry only the transient classifications, prefer
629
- // a server Retry-After over the backoff, and let the caller's abort cut
630
- // through both the in-flight request and the backoff sleep.
631
- const transport = this.#streaming ? chatCompletionStream : chatCompletion;
632
-
633
585
  // Per-request headers = static auth/routing + any first-party telemetry.
634
586
  const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
635
587
  const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
636
- const transportRetries: Array<{ attempt: number; kind: string; elapsedMs: number; message: string }> = [];
637
588
  let raw;
638
- for (let attempt = 0; ; attempt++) {
639
- const attemptStarted = performance.now();
640
- const timeoutSignal = AbortSignal.timeout(this.#fetchTimeoutMs);
641
- const effectiveSignal = signal !== undefined ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal;
642
- try {
643
- raw = await transport({
644
- url: this.#url,
589
+ try {
590
+ raw = this.#languageModel === undefined
591
+ ? await executeOpenAICompatible({
592
+ url: this.#url!,
593
+ model: this.#model,
645
594
  headers,
646
595
  body,
647
- signal: effectiveSignal,
596
+ messages,
597
+ signal,
648
598
  fetch: this.#fetch,
599
+ fetchTimeoutMs: this.#fetchTimeoutMs,
600
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
601
+ retryAttempts: this.#retryAttempts,
602
+ streaming: this.#streaming,
649
603
  captureRawBody: this.#rawBody,
604
+ })
605
+ : await executeAiSdkModel({
606
+ languageModel: this.#languageModel,
607
+ headers,
608
+ messages,
609
+ signal,
610
+ fetchTimeoutMs: this.#fetchTimeoutMs,
650
611
  streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
612
+ retryAttempts: this.#retryAttempts,
613
+ streaming: this.#streaming,
614
+ captureRawBody: this.#rawBody,
615
+ temperature: this.#tuningFloors
616
+ ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
617
+ : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
618
+ topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
619
+ topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
620
+ presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
621
+ frequencyPenalty: typeof sampling?.frequency_penalty === "number"
622
+ ? sampling.frequency_penalty
623
+ : this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
624
+ stopSequences: typeof sampling?.stop === "string"
625
+ ? [sampling.stop]
626
+ : Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
627
+ ? sampling.stop
628
+ : undefined,
629
+ seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
630
+ maxOutputTokens: maxTokens,
631
+ reasoning: this.#reasoning.mode === "off"
632
+ ? "none"
633
+ : this.#reasoning.mode === "adaptive"
634
+ ? "provider-default"
635
+ : effortFromBudget(this.#reasoning.budget!),
651
636
  });
652
- break;
653
- } catch (err) {
654
- // Caller-initiated abort is cancellation — never retried or wrapped.
655
- if (signal?.aborted) throw err;
656
- const { kind } = classifyProviderError(err);
657
- // #543: Cloudflare/CDN edge codes (520-527) classify as network_failure
658
- // but fail-fast - a retry just re-incurs the same origin/edge timeout.
659
- const edgeTimeout = err instanceof OpenAiHttpError && isEdgeStatus(err.status);
660
- // Terminal kind, edge failure, or budget spent -> surface the failure.
661
- if (!RETRYABLE.has(kind) || edgeTimeout || attempt >= this.#retryAttempts) {
662
- const pe = toProviderError(err, this.#source);
663
- // #537 case 2: a 401/403 with a key PRESENT is a rejected key, not a
664
- // transport failure — surface the distinct, actionable hint (kind stays
665
- // "unauthorized", so core's routing is unchanged) rather than the raw
666
- // upstream JSON. Distinct from the unset-key throw at construction.
667
- if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
668
- throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
669
- }
670
- throw pe;
671
- }
672
- transportRetries.push({
673
- attempt: attempt + 1,
674
- kind,
675
- elapsedMs: Math.round(performance.now() - attemptStarted),
676
- message: err instanceof Error ? err.message : String(err),
677
- });
678
- const retryAfter = err instanceof OpenAiHttpError ? err.retryAfter : null;
679
- await sleepWithAbort(retryAfter ?? this.#retryDelayMs * 2 ** attempt, signal);
637
+ } catch (err) {
638
+ if (signal?.aborted) throw err;
639
+ const pe = toProviderError(err, this.#source);
640
+ if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
641
+ throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
680
642
  }
643
+ throw pe;
681
644
  }
682
645
 
683
646
  // #539: llama-server --special renders EOG tokens as text, so a turn ending
@@ -696,7 +659,7 @@ export default class OpenAICompatProvider implements Provider {
696
659
  // Discard/retry/escalate/self-correct is the consumer's policy.
697
660
  let telemetry: TelemetryEvent[] | undefined;
698
661
  let railsMeta: Record<string, unknown> | undefined;
699
- const usage = normalizeUsage(raw.usage, raw.reasoning_content, raw.content);
662
+ const usage = raw.usage;
700
663
  const observedGrammar = sendGrammar ?? (wantGrammar && this.#gbnfDebug ? grammar : undefined);
701
664
  if (observedGrammar !== undefined) {
702
665
  const verdict = this.#grammarVerdict(observedGrammar, raw.content);
@@ -714,7 +677,7 @@ export default class OpenAICompatProvider implements Provider {
714
677
  // invisible, billed (12,288 billed vs 1,033 chars visible, live).
715
678
  // countTokens OVERCOUNTS text (chars/2 upper bound), so billed
716
679
  // exceeding visible-plus-slack is real vanishing, not estimator noise.
717
- const visible = this.#countTokens(raw.content) + this.#countTokens(raw.reasoning_content);
680
+ const visible = this.#countTokens(raw.content) + this.#countTokens(raw.reasoning);
718
681
  if (sendGrammar !== undefined && usage.completion > visible + 64) {
719
682
  (telemetry ??= []).push({
720
683
  source: this.#source,
@@ -725,30 +688,23 @@ export default class OpenAICompatProvider implements Provider {
725
688
  }
726
689
  }
727
690
 
728
- const builtMeta = this.#buildMeta(raw.chunkMetadata);
729
- const retryMeta = transportRetries.length > 0 ? { transportRetries } : undefined;
730
- const meta = railsMeta !== undefined || retryMeta !== undefined
731
- ? { ...builtMeta, ...railsMeta, ...retryMeta }
732
- : builtMeta;
733
-
734
- // #36: surface per-token logprobs + their mean when the backend returned
735
- // them (only possible when the flag requested them). Absent otherwise —
736
- // never synthesized.
737
- const logprobs = raw.logprobs !== null && raw.logprobs.length > 0 ? raw.logprobs : undefined;
691
+ const builtMeta = this.#buildMeta(raw.metadata);
692
+ const meta = railsMeta !== undefined ? { ...builtMeta, ...railsMeta } : builtMeta;
693
+ const logprobs = raw.logprobs.length > 0 ? raw.logprobs : undefined;
738
694
  const meanLogprob = logprobs !== undefined
739
- ? logprobs.reduce((sum, t) => sum + t.logprob, 0) / logprobs.length
695
+ ? logprobs.reduce((sum, token) => sum + token.logprob, 0) / logprobs.length
740
696
  : undefined;
741
697
 
742
698
  return {
743
699
  assistant: {
744
700
  content: raw.content,
745
- reasoning: raw.reasoning_content.length > 0 ? raw.reasoning_content : null,
746
- // #482: sealed relay blobs ride only when present — same absence
747
- // discipline as logprobs (never synthesized, never empty-array).
748
- ...(raw.reasoning_encrypted.length > 0 ? { reasoningEncrypted: raw.reasoning_encrypted } : {}),
701
+ reasoning: raw.reasoning.length > 0 ? raw.reasoning : null,
702
+ ...(raw.reasoningEncrypted.length > 0
703
+ ? { reasoningEncrypted: raw.reasoningEncrypted }
704
+ : {}),
749
705
  usage,
750
- finishReason: normalizeFinishReason(raw.finish_reason),
751
- model: raw.model ?? this.#model,
706
+ finishReason: raw.finishReason,
707
+ model: raw.model,
752
708
  ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
753
709
  },
754
710
  assistantRaw: raw,
package/src/Mock.test.ts CHANGED
@@ -109,7 +109,7 @@ test("Mock: exhausted queue throws a specific error", async () => {
109
109
  // -- #507: the reserve surface lives on the Provider CONTRACT, and Mock drives core's partition suite --
110
110
 
111
111
  test("#507 the reserve getters are on the Provider interface (not just the concrete class)", () => {
112
- // Typing against the CONTRACT is the check a getter-only-on-OpenAICompat surface fails.
112
+ // Typing against the contract catches a getter-only concrete surface.
113
113
  const prevR = process.env.PLURNK_PROVIDERS_REASONING_RESERVE;
114
114
  try {
115
115
  process.env.PLURNK_PROVIDERS_REASONING_RESERVE = "10%";
@@ -2,7 +2,6 @@ import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
3
  import { instantiateProvider, loadActiveProvider, resetDiscoveryCache } from "./ProviderRegistry.ts";
4
4
 
5
- const fakeProvider = { contextWindow: 1, model: "m", countTokens: () => 0, calculateCost: () => 0, generate: async () => { throw new Error("unused"); } };
6
5
  const mapOf = (entries: Record<string, string>, skipped: Record<string, string> = {}) =>
7
6
  async () => ({ registry: new Map(Object.entries(entries)), skipped: new Map(Object.entries(skipped)), attributions: new Map<string, string | string[]>() });
8
7
 
@@ -10,16 +9,16 @@ const mapOf = (entries: Record<string, string>, skipped: Record<string, string>
10
9
  // exercise the resolution + two-tier instantiation this module owns; the active
11
10
  // alias is driven end-to-end by loadActiveProvider below.
12
11
 
13
- // — two-tier instantiation (SPEC §5) —
12
+ // — provider resolution (SPEC §5) —
14
13
 
15
14
  const fullEnv = Object.freeze({
16
15
  PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000",
17
16
  PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
18
- PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
17
+ PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0", PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "1",
19
18
  OPENAI_BASE_URL: "http://x",
20
19
  });
21
20
 
22
- test("instantiateProvider: standard name resolves in-framework, no scan, no import", async () => {
21
+ test("instantiateProvider: cataloged name resolves in-framework, no scan, no import", async () => {
23
22
  resetDiscoveryCache();
24
23
  mock.method(globalThis, "fetch", async (url: string) => {
25
24
  if (String(url).endsWith("/models")) return new Response(JSON.stringify({ data: [] }), { status: 200 });
@@ -36,24 +35,33 @@ test("instantiateProvider: standard name resolves in-framework, no scan, no impo
36
35
  mock.restoreAll();
37
36
  });
38
37
 
39
- test("instantiateProvider: bespoke name resolves via the scan and imports the discovered package", async () => {
38
+ test("instantiateProvider: an installed AI SDK provider resolves through discovery", async () => {
40
39
  resetDiscoveryCache();
41
40
  const calls: unknown[] = [];
42
- const p = await instantiateProvider("openrouter", { ...fullEnv }, "anthropic/claude-opus-latest",
43
- async (specifier) => { calls.push(specifier); return { default: { fromEnv: async (_e: NodeJS.ProcessEnv, model: string) => { calls.push(model); return fakeProvider; } } }; },
44
- mapOf({ openrouter: "@plurnk/plurnk-providers-openrouter" }));
45
- assert.equal(p, fakeProvider);
46
- assert.deepEqual(calls, ["@plurnk/plurnk-providers-openrouter", "anthropic/claude-opus-latest"]);
41
+ const p = await instantiateProvider("acme", { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "model-a",
42
+ async (specifier) => {
43
+ calls.push(specifier);
44
+ return { default: { languageModel: (model: string) => { calls.push(model); return {} as never; } } };
45
+ },
46
+ mapOf({ acme: "@acme/ai-provider" }));
47
+ assert.equal(p.model, "model-a");
48
+ assert.equal(p.contextWindow, 8192);
49
+ assert.deepEqual(calls, ["@acme/ai-provider", "model-a"]);
47
50
  });
48
51
 
49
- test("instantiateProvider: a per-alias baseUrl is passed to a bespoke factory as the 3rd-arg option", async () => {
52
+ test("instantiateProvider: a per-alias baseUrl drives the built-in Ollama probe", async () => {
50
53
  resetDiscoveryCache();
51
- let received: unknown;
54
+ const calls: string[] = [];
55
+ mock.method(globalThis, "fetch", async (url: string) => {
56
+ calls.push(String(url));
57
+ return new Response(JSON.stringify({ model_info: { "qwen.context_length": 32768 } }));
58
+ });
52
59
  await instantiateProvider("ollama", { ...fullEnv }, "qwen2.5-coder",
53
- async () => ({ default: { fromEnv: async (_e: NodeJS.ProcessEnv, _m: string, options?: unknown) => { received = options; return fakeProvider; } } }),
54
- mapOf({ ollama: "@plurnk/plurnk-providers-ollama" }),
60
+ async () => ({}),
61
+ mapOf({}),
55
62
  "http://nook:11434");
56
- assert.deepEqual(received, { baseUrl: "http://nook:11434" });
63
+ assert.deepEqual(calls, ["http://nook:11434/api/show"]);
64
+ mock.restoreAll();
57
65
  });
58
66
 
59
67
  test("instantiateProvider: a per-alias baseUrl drives the standard openai probe to the override host", async () => {
@@ -74,14 +82,14 @@ test("instantiateProvider: a per-alias baseUrl drives the standard openai probe
74
82
  test("instantiateProvider: a THIRD-PARTY scope is discovered — name maps to its package", async () => {
75
83
  resetDiscoveryCache();
76
84
  const imports: string[] = [];
77
- const p = await instantiateProvider("foo", { ...fullEnv }, "m",
78
- async (specifier) => { imports.push(specifier); return { default: { fromEnv: async () => fakeProvider } }; },
85
+ const p = await instantiateProvider("foo", { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW: "4096" }, "m",
86
+ async (specifier) => { imports.push(specifier); return { default: { languageModel: () => ({} as never) } }; },
79
87
  mapOf({ foo: "@acme/acme-provider-foo" }));
80
- assert.equal(p, fakeProvider);
88
+ assert.equal(p.model, "m");
81
89
  assert.deepEqual(imports, ["@acme/acme-provider-foo"]); // not an @plurnk/ specifier
82
90
  });
83
91
 
84
- test("instantiateProvider: a standard name is authoritative — a scanned package of the same name is shadowed", async () => {
92
+ test("instantiateProvider: a cataloged name is authoritative — a scanned package of the same name is shadowed", async () => {
85
93
  resetDiscoveryCache();
86
94
  mock.method(globalThis, "fetch", async (url: string) => {
87
95
  if (String(url).endsWith("/models")) return new Response(JSON.stringify({ data: [] }), { status: 200 });
@@ -100,7 +108,7 @@ test("instantiateProvider: unknown provider throws — no standard, no discovere
100
108
  resetDiscoveryCache();
101
109
  await assert.rejects(
102
110
  () => instantiateProvider("nope", { ...fullEnv }, "m", async () => ({}), mapOf({})),
103
- /unknown provider "nope": not a standard provider, and no installed package declares plurnk\.kind:"provider" with name "nope"/,
111
+ /unknown provider "nope"/,
104
112
  );
105
113
  });
106
114
 
@@ -116,13 +124,13 @@ test("instantiateProvider: an untrusted (skipped) provider gives a precise error
116
124
  assert.deepEqual(imports, []); // never imported an untrusted package
117
125
  });
118
126
 
119
- test("instantiateProvider: discovered package without a fromEnv factory throws (factory shape, SPEC )", async () => {
127
+ test("instantiateProvider: discovered package must export an AI SDK provider", async () => {
120
128
  resetDiscoveryCache();
121
129
  await assert.rejects(
122
130
  () => instantiateProvider("broken", { ...fullEnv }, "m",
123
131
  async () => ({ default: {} }),
124
132
  mapOf({ broken: "@acme/acme-provider-broken" })),
125
- /@acme\/acme-provider-broken default export is not a Provider factory/,
133
+ /@acme\/acme-provider-broken default export is not an AI SDK provider/,
126
134
  );
127
135
  });
128
136
 
@@ -174,6 +182,8 @@ test("#622: two Fireworks aliases independently select default and priority serv
174
182
  FIREWORKS_BASE_URL: "https://api.fireworks.ai/inference/v1",
175
183
  FIREWORKS_API_KEY: "fw",
176
184
  PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
185
+ PLURNK_PROVIDERS_PROVIDER_FIREWORKS_REASONING_STYLE: "effort_explicit",
186
+ PLURNK_PROVIDERS_TOP_LOGPROBS: "2",
177
187
  PLURNK_PROVIDERS_SERVICE_TIER_fast: "priority",
178
188
  PLURNK_PROVIDERS_SERVICE_TIER_standard: "default",
179
189
  };
@@ -184,6 +194,9 @@ test("#622: two Fireworks aliases independently select default and priority serv
184
194
  await fast.generate({ workerId: "fast-worker", messages: [] });
185
195
  await standard.generate({ workerId: "standard-worker", messages: [] });
186
196
  assert.deepEqual(bodies.map((body) => body.service_tier), ["priority", "default"]);
197
+ assert.deepEqual(bodies.map((body) => body.prompt_cache_key), ["fast-worker", "standard-worker"]);
198
+ assert.deepEqual(bodies.map((body) => body.reasoning_effort), ["none", "none"]);
199
+ assert.deepEqual(bodies.map((body) => body.top_logprobs), [2, 2]);
187
200
  assert.deepEqual(bodies.map((body) => body.model), [
188
201
  "accounts/fireworks/routers/glm-5p2-fast",
189
202
  "accounts/fireworks/models/deepseek-v4-pro",
@@ -191,13 +204,13 @@ test("#622: two Fireworks aliases independently select default and priority serv
191
204
  mock.restoreAll();
192
205
  });
193
206
 
194
- test("loadActiveProvider: resolves the alias cascade end-to-end via the scan", async () => {
207
+ test("loadActiveProvider: resolves the alias cascade to an installed AI SDK provider", async () => {
195
208
  resetDiscoveryCache();
196
- const env = { ...fullEnv, PLURNK_MODEL: "opus", PLURNK_MODEL_opus: "openrouter/anthropic/claude-opus-latest" } as NodeJS.ProcessEnv;
209
+ const env = { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW: "4096", PLURNK_MODEL: "custom", PLURNK_MODEL_custom: "acme/model-a" } as NodeJS.ProcessEnv;
197
210
  const p = await loadActiveProvider(env,
198
- async () => ({ default: { fromEnv: async () => fakeProvider } }),
199
- mapOf({ openrouter: "@plurnk/plurnk-providers-openrouter" }));
200
- assert.equal(p, fakeProvider);
211
+ async () => ({ default: { languageModel: () => ({} as never) } }),
212
+ mapOf({ acme: "@acme/ai-provider" }));
213
+ assert.equal(p.model, "model-a");
201
214
  });
202
215
 
203
216
  test("loadActiveProvider: throws a named error when no alias is active", async () => {
@@ -3,18 +3,19 @@
3
3
  // overrides) lives in @plurnk/plurnk-aliases — the zero-dep parser shared with
4
4
  // thin clients (#27); this module resolves the active alias to a Provider.
5
5
  //
6
- // Two-tier resolution (SPEC §5): tier 1 is the closed standard-provider table;
7
- // tier 2 is a SCOPE-AGNOSTIC node_modules scan (discover()) for packages
8
- // declaring `plurnk.kind:"provider"` — first-party plugins (installed flat
9
- // via @plurnk/plurnk-providers-all) AND third-party providers under any scope.
10
- // The framework is contract-only — it does NOT depend on its plugins; the
11
- // scan is what surfaces them (#12/#14).
6
+ // Resolution order (SPEC §5): models.dev catalog → PLURNK provider declaration
7
+ // → local protocol adapter → scope-agnostic AI SDK plugin discovery. Generic
8
+ // provider facts belong to models.dev or operator config; PLURNK owns only the
9
+ // stable Provider contract and product-specific local behavior.
12
10
 
13
- import type { Provider, ProviderFactory } from "./types.ts";
14
- import { isStandardProvider, standardProviderFromEnv } from "./standardProviders.ts";
11
+ import type { AiSdkProviderPlugin, Provider } from "./types.ts";
12
+ import { catalogProviderFromEnv, providerFromSdkModel } from "./catalogProvider.ts";
15
13
  import { discover, type DiscoverOptions, type Discovery } from "./discover.ts";
16
14
  import { resolveActiveAlias } from "@plurnk/plurnk-aliases";
17
15
  import { scopeEnvToAlias } from "./env.ts";
16
+ import { ollamaProviderFromEnv } from "./ollama.ts";
17
+ import { compatibleProviderFromEnv } from "./compatibleProvider.ts";
18
+ import { contextWindowFromEnv } from "./env.ts";
18
19
 
19
20
  // Two injectable seams, both defaulting to production behavior and never passed
20
21
  // by real callers: the module importer (tests exercise the bespoke path without
@@ -33,10 +34,8 @@ const providerPackages = async (discoverFn: DiscoverFn, env: NodeJS.ProcessEnv):
33
34
  return discoveredCache;
34
35
  };
35
36
 
36
- // Two-tier resolution (SPEC §5): tier 1 standard table → tier 2 discovered
37
- // package (scope-agnostic scan, trust-gated) → fail-hard. The standard table is
38
- // authoritative — a scanned package whose name duplicates a standard one is
39
- // shadowed here (never reached), since tier 1 returns first.
37
+ // Catalog and explicit declarations are authoritative. Discovery is the
38
+ // extensibility seam for an AI SDK provider that neither source describes.
40
39
  export const instantiateProvider = async (
41
40
  name: string,
42
41
  env: NodeJS.ProcessEnv,
@@ -46,14 +45,13 @@ export const instantiateProvider = async (
46
45
  baseUrl?: string, // per-alias endpoint override (PLURNK_BASEURL_<alias>); threaded to both tiers
47
46
  alias?: string, // the alias this instantiation serves — scopes PLURNK_PROVIDERS_<KNOB>_<alias> overrides
48
47
  ): Promise<Provider> => {
49
- // Per-alias knob scoping: overlay any _<alias>-suffixed knob onto its bare
50
- // name so both tiers (and every fromEnv) read plain vars, per-alias-resolved.
48
+ // Per-alias knob scoping overlays _<alias>-suffixed knobs onto their bare
49
+ // names before any resolver reads them.
51
50
  if (alias !== undefined) env = scopeEnvToAlias(env, alias);
52
- if (isStandardProvider(name)) {
53
- const standard = await standardProviderFromEnv(name, env, model, baseUrl);
54
- if (standard === null) throw new Error(`provider "${name}": standard registry resolution failed`);
55
- return standard;
56
- }
51
+ const catalog = catalogProviderFromEnv(name, env, model, baseUrl);
52
+ if (catalog !== null) return catalog;
53
+ if (name === "ollama") return ollamaProviderFromEnv(env, model, baseUrl === undefined ? undefined : { baseUrl });
54
+ if (name === "openai" || name === "plurnk") return compatibleProviderFromEnv(name, env, model, baseUrl);
57
55
  const { registry, skipped } = await providerPackages(discoverFn, env);
58
56
  const specifier = registry.get(name);
59
57
  if (specifier === undefined) {
@@ -61,7 +59,7 @@ export const instantiateProvider = async (
61
59
  if (declined !== undefined) {
62
60
  throw new Error(`provider "${name}" resolves to ${declined}, but it is untrusted under PLURNK_PLUGINS_TRUSTED_ONLY — add it to the allowlist (or publish under @plurnk/)`);
63
61
  }
64
- throw new Error(`unknown provider "${name}": not a standard provider, and no installed package declares plurnk.kind:"provider" with name "${name}"`);
62
+ throw new Error(`unknown provider "${name}": absent from models.dev, operator declarations, local adapters, and installed AI SDK provider plugins`);
65
63
  }
66
64
  let mod: unknown;
67
65
  try {
@@ -69,11 +67,24 @@ export const instantiateProvider = async (
69
67
  } catch (cause) {
70
68
  throw new Error(`provider "${name}" resolves to ${specifier}, but importing it failed`, { cause });
71
69
  }
72
- const factory = (mod as { default?: ProviderFactory }).default;
73
- if (factory === undefined || typeof factory.fromEnv !== "function") {
74
- throw new Error(`${specifier} default export is not a Provider factory (missing static fromEnv)`);
70
+ const sdkProvider = (mod as { default?: AiSdkProviderPlugin }).default;
71
+ if (sdkProvider === undefined || typeof sdkProvider.languageModel !== "function") {
72
+ throw new Error(`${specifier} default export is not an AI SDK provider (missing languageModel)`);
73
+ }
74
+ if (baseUrl !== undefined) {
75
+ throw new Error(`${specifier}: PLURNK_BASEURL_${alias ?? "<alias>"} cannot reconfigure an installed AI SDK provider; declare the provider through PLURNK_PROVIDERS_PROVIDER_* instead`);
76
+ }
77
+ const contextWindow = contextWindowFromEnv(env, name);
78
+ if (contextWindow === null) {
79
+ throw new Error(`${specifier}: PLURNK_PROVIDERS_CONTEXT_WINDOW must be set because Models.dev has no metadata for provider "${name}"`);
75
80
  }
76
- return await factory.fromEnv(env, model, baseUrl !== undefined ? { baseUrl } : undefined);
81
+ return providerFromSdkModel({
82
+ name,
83
+ env,
84
+ model,
85
+ languageModel: sdkProvider.languageModel(model),
86
+ contextWindow,
87
+ });
77
88
  };
78
89
 
79
90
  // Test-only: drop the memoized discovery so a fresh scan/injection runs next.