@plurnk/plurnk-providers 1.3.12 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/.env.defaults +39 -22
  2. package/README.md +65 -4
  3. package/SPEC.md +222 -56
  4. package/dist/AiSdkProvider.d.ts +18 -5
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +240 -153
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +14 -15
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +26 -10
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -2
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +41 -8
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/ProviderRegistry.d.ts +4 -1
  17. package/dist/ProviderRegistry.d.ts.map +1 -1
  18. package/dist/ProviderRegistry.js +7 -3
  19. package/dist/ProviderRegistry.js.map +1 -1
  20. package/dist/aiSdkTransport.d.ts +4 -2
  21. package/dist/aiSdkTransport.d.ts.map +1 -1
  22. package/dist/aiSdkTransport.js +18 -3
  23. package/dist/aiSdkTransport.js.map +1 -1
  24. package/dist/catalogProvider.d.ts +3 -1
  25. package/dist/catalogProvider.d.ts.map +1 -1
  26. package/dist/catalogProvider.js +20 -7
  27. package/dist/catalogProvider.js.map +1 -1
  28. package/dist/compatibleProvider.d.ts.map +1 -1
  29. package/dist/compatibleProvider.js +11 -4
  30. package/dist/compatibleProvider.js.map +1 -1
  31. package/dist/cost.d.ts +11 -0
  32. package/dist/cost.d.ts.map +1 -0
  33. package/dist/cost.js +64 -0
  34. package/dist/cost.js.map +1 -0
  35. package/dist/discover.d.ts +2 -0
  36. package/dist/discover.d.ts.map +1 -1
  37. package/dist/discover.js +15 -9
  38. package/dist/discover.js.map +1 -1
  39. package/dist/env.d.ts +4 -0
  40. package/dist/env.d.ts.map +1 -1
  41. package/dist/env.js +29 -9
  42. package/dist/env.js.map +1 -1
  43. package/dist/errors.d.ts +27 -0
  44. package/dist/errors.d.ts.map +1 -0
  45. package/dist/errors.js +150 -0
  46. package/dist/errors.js.map +1 -0
  47. package/dist/index.d.ts +11 -5
  48. package/dist/index.d.ts.map +1 -1
  49. package/dist/index.js +7 -4
  50. package/dist/index.js.map +1 -1
  51. package/dist/notices.d.ts +10 -0
  52. package/dist/notices.d.ts.map +1 -0
  53. package/dist/notices.js +11 -0
  54. package/dist/notices.js.map +1 -0
  55. package/dist/ollama.d.ts.map +1 -1
  56. package/dist/ollama.js +3 -3
  57. package/dist/ollama.js.map +1 -1
  58. package/dist/openai.d.ts +1 -1
  59. package/dist/openai.d.ts.map +1 -1
  60. package/dist/promptTokens.d.ts +4 -0
  61. package/dist/promptTokens.d.ts.map +1 -0
  62. package/dist/promptTokens.js +32 -0
  63. package/dist/promptTokens.js.map +1 -0
  64. package/dist/sdkModels.d.ts.map +1 -1
  65. package/dist/sdkModels.js +4 -3
  66. package/dist/sdkModels.js.map +1 -1
  67. package/dist/types.d.ts +43 -16
  68. package/dist/types.d.ts.map +1 -1
  69. package/dist/types.js +1 -1
  70. package/dist/types.js.map +1 -1
  71. package/dist/usage.d.ts +3 -0
  72. package/dist/usage.d.ts.map +1 -1
  73. package/dist/usage.js +26 -14
  74. package/dist/usage.js.map +1 -1
  75. package/dist/warnings.js +0 -0
  76. package/dist/warnings.js.map +1 -1
  77. package/package.json +13 -9
  78. package/src/AiSdkProvider.test.ts +480 -159
  79. package/src/AiSdkProvider.ts +320 -196
  80. package/src/Mock.test.ts +29 -14
  81. package/src/Mock.ts +33 -15
  82. package/src/Pool.test.ts +43 -6
  83. package/src/Pool.ts +56 -10
  84. package/src/ProviderRegistry.test.ts +158 -9
  85. package/src/ProviderRegistry.ts +19 -6
  86. package/src/aiSdkTransport.ts +25 -6
  87. package/src/boundaries.test.ts +8 -3
  88. package/src/catalogProvider.test.ts +17 -0
  89. package/src/catalogProvider.ts +25 -10
  90. package/src/compatibleProvider.test.ts +96 -0
  91. package/src/compatibleProvider.ts +15 -6
  92. package/src/cost.test.ts +63 -0
  93. package/src/cost.ts +83 -0
  94. package/src/defaults.test.ts +1 -0
  95. package/src/discover.test.ts +48 -7
  96. package/src/discover.ts +31 -21
  97. package/src/env.test.ts +38 -23
  98. package/src/env.ts +45 -18
  99. package/src/errors.test.ts +148 -0
  100. package/src/errors.ts +207 -0
  101. package/src/index.ts +29 -7
  102. package/src/lexicon-guard.test.ts +6 -6
  103. package/src/notices.ts +22 -0
  104. package/src/ollama.test.ts +64 -0
  105. package/src/ollama.ts +6 -3
  106. package/src/openai.ts +3 -0
  107. package/src/promptTokens.ts +41 -0
  108. package/src/sdkModels.test.ts +7 -0
  109. package/src/sdkModels.ts +4 -8
  110. package/src/types.ts +106 -64
  111. package/src/usage.test.ts +15 -4
  112. package/src/usage.ts +32 -14
  113. package/src/warnings.test.ts +10 -10
  114. package/src/warnings.ts +0 -0
  115. package/dist/OpenAICompat.d.ts +0 -76
  116. package/dist/OpenAICompat.d.ts.map +0 -1
  117. package/dist/OpenAICompat.js +0 -555
  118. package/dist/OpenAICompat.js.map +0 -1
  119. package/dist/openaiStream.d.ts +0 -47
  120. package/dist/openaiStream.d.ts.map +0 -1
  121. package/dist/openaiStream.js +0 -280
  122. package/dist/openaiStream.js.map +0 -1
  123. package/dist/standardProviders.d.ts +0 -31
  124. package/dist/standardProviders.d.ts.map +0 -1
  125. package/dist/standardProviders.js +0 -518
  126. package/dist/standardProviders.js.map +0 -1
  127. package/dist/telemetry.d.ts +0 -24
  128. package/dist/telemetry.d.ts.map +0 -1
  129. package/dist/telemetry.js +0 -85
  130. package/dist/telemetry.js.map +0 -1
  131. package/src/telemetry.test.ts +0 -69
  132. package/src/telemetry.ts +0 -116
package/src/index.ts CHANGED
@@ -1,17 +1,25 @@
1
1
  export type {
2
2
  ChatMessage,
3
3
  FinishReason,
4
+ GrammarEvidence,
4
5
  Provider,
5
6
  ProviderAssistant,
7
+ ProviderAttempt,
8
+ ProviderAttemptFinishReason,
6
9
  AiSdkProviderPlugin,
7
10
  ProviderOptions,
8
11
  ProviderResponse,
12
+ ProviderEncryptedReasoningItem,
9
13
  ProviderUsage,
14
+ PromptTokenMeasurement,
10
15
  TokenLogprob,
11
16
  TokenAlternative,
17
+ AuthoritativeCharge,
12
18
  } from "./types.ts";
19
+ export type { ProviderCost } from "@plurnk/plurnk-contracts";
20
+ export { assertPromptTokenMeasurement } from "./promptTokens.ts";
13
21
 
14
- // Alias cascade — re-exported from the zero-dep @plurnk/plurnk-aliases (#27), so
22
+ // Alias cascade — re-exported from the zero-dep @plurnk/plurnk-aliases, so
15
23
  // the "." surface is unchanged for existing importers and there's one source of
16
24
  // truth for the parser (thin clients depend on that package directly).
17
25
  export type { ProviderAlias } from "@plurnk/plurnk-aliases";
@@ -23,24 +31,38 @@ export {
23
31
  resetDiscoveryCache,
24
32
  } from "./ProviderRegistry.ts";
25
33
 
26
- // Scope-agnostic plugin discovery (SPEC §5).
34
+ // Scope-agnostic plugin discovery ({§plugin-family-kind}).
27
35
  export { discover } from "./discover.ts";
28
36
  export type { DiscoverOptions, Discovery } from "./discover.ts";
29
37
 
30
38
  // Stable PLURNK adapter over AI SDK language models and compatible local URLs.
31
39
  export { default as AiSdkProvider, effortFromBudget } from "./AiSdkProvider.ts";
32
40
  export type { AiSdkProviderConfig, ReasoningStyle, GrammarStyle } from "./AiSdkProvider.ts";
33
- // Capacity pool (SPEC §15): front N interchangeable backends as one Provider -
41
+ // {§provider-capacity-pool} Front N interchangeable backends as one Provider -
34
42
  // worker-sticky for KV-cache reuse, overflow to a healthy sibling; the blend
35
43
  // DECISION stays the consumer's, by choosing which pool to call.
36
44
  export { default as Pool } from "./Pool.ts";
37
45
  export type { ProviderFetch } from "./AiSdkProvider.ts";
38
- export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, envelopeFromEnv, resolveReserve } from "./env.ts";
39
- export type { Reasoning, ReasoningMode, ReserveSpec } from "./env.ts";
46
+ export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, reasoningResponseStyleFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, effectiveContextWindow, envelopeFromEnv, resolveReserve, PROVIDERS_KNOBS } from "./env.ts";
47
+ export type { Reasoning, ReasoningMode, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
40
48
  export { normalizeUsage, calculateCostUsd } from "./usage.ts";
49
+ export {
50
+ providerCostFor,
51
+ providerCostUsd,
52
+ resolveProviderCost,
53
+ validateAuthoritativeCharge,
54
+ validateProviderCost,
55
+ } from "./cost.ts";
41
56
  export type { RawUsage, TokenRates } from "./usage.ts";
42
- export { ProviderError, classifyProviderError, toProviderError, providerSource } from "./telemetry.ts";
43
- export type { TelemetryEvent, ProviderTelemetryKind } from "./telemetry.ts";
57
+ export { ProviderError, classifyProviderError, toProviderError } from "./errors.ts";
58
+ export { providerSource } from "./notices.ts";
59
+ export type { ProviderErrorKind } from "./errors.ts";
60
+ export type { ProviderNotice, ProviderNoticeKind } from "./notices.ts";
61
+ export type {
62
+ PluginAttributionContext,
63
+ PluginAttributionDeclaration,
64
+ PluginAttributionSource,
65
+ } from "@plurnk/plurnk-meta";
44
66
 
45
67
  export { default as Mock } from "./Mock.ts";
46
68
  export type { MockAssistant, MockResponse, MockReturnedAssistant } from "./Mock.ts";
@@ -1,4 +1,4 @@
1
- // [§lexicon] the providers-lane standing guard (#472/#477, owner-ruled OpenAI
1
+ // {§lexicon} The providers-lane standing guard (OpenAI
2
2
  // lexicon). Retired terms fail CI here, not at the next audit — the mirror of
3
3
  // core's plurnk-core guard, tuned to what PROVIDERS retired. Scope: src/ non-test
4
4
  // + SPEC.md. A `lexicon-allow` line marker exempts the shed's own call sites
@@ -33,13 +33,13 @@ const WIRE_THINKING = /enable_thinking|`thinking`|thinking\s*:|\.thinking\b|thin
33
33
 
34
34
  // Each entry: the banned pattern and the canonical term the violation must become.
35
35
  const BANNED: Array<{ label: string; re: RegExp; canon: string; exempt?: RegExp }> = [
36
- { label: "thinking (our-voice)", re: /\bthinking\b/i, canon: "reasoning (the #472 lexicon ruling)", exempt: WIRE_THINKING },
37
- { label: "contextSize", re: /\bcontextSize\b/, canon: "contextWindow — the provider window (#472)" },
38
- { label: "retired providers knob", re: /PLURNK_PROVIDERS_(THINKING|LOGPROB\b|CONTEXT_SIZE\b)/, canon: "PLURNK_PROVIDERS_{REASONING,TOP_LOGPROBS,CONTEXT_WINDOW} (#399/#472) — only the shed may name these" },
39
- // #511: catch the retired run/session noun in the WIRE-HEADER form too (a
36
+ { label: "thinking (our-voice)", re: /\bthinking\b/i, canon: "reasoning ({§lexicon})", exempt: WIRE_THINKING },
37
+ { label: "contextSize", re: /\bcontextSize\b/, canon: "contextWindow — the provider window ({§model-fact-resolution})" },
38
+ { label: "retired providers knob", re: /PLURNK_PROVIDERS_(THINKING|LOGPROB\b|CONTEXT_SIZE\b)/, canon: "PLURNK_PROVIDERS_{REASONING,TOP_LOGPROBS,CONTEXT_WINDOW} ({§provider-configuration}) — only the shed may name these" },
39
+ // Catch the retired run/session noun in the wire-header form too (a
40
40
  // quoted string, not an identifier — the hole the old `Plurnk-Run-Id` hid in),
41
41
  // alongside the coordinate identifiers.
42
- { label: "run/session (retired noun — coordinate or wire header)", re: /\b(sessionId|runId)\b|Plurnk-(Run|Session)-Id/, canon: "workerId/workspaceId, Plurnk-Worker-Id/Plurnk-Workspace-Id (#486/#511)" },
42
+ { label: "run/session (retired noun — coordinate or wire header)", re: /\b(sessionId|runId)\b|Plurnk-(Run|Session)-Id/, canon: "workerId/workspaceId, Plurnk-Worker-Id/Plurnk-Workspace-Id ({§lifecycle-terms})" },
43
43
  ];
44
44
 
45
45
  test("retired provider terms never reappear in src or SPEC — drift fails CI, not the next audit", () => {
package/src/notices.ts ADDED
@@ -0,0 +1,22 @@
1
+ export type ProviderNoticeKind = "grammar_unenforced";
2
+
3
+ // Observations about a completed model exchange. These never represent a
4
+ // failed provider operation; transport failures throw ProviderError with an
5
+ // RFC 9457 Problem Details object.
6
+ export interface ProviderNotice {
7
+ readonly source: string;
8
+ readonly kind: ProviderNoticeKind;
9
+ readonly level: "warn";
10
+ readonly message: string;
11
+ readonly position: number | null;
12
+ }
13
+
14
+ export const providerSource = (vendor: string): string => {
15
+ const raw = vendor.startsWith("provider:") ? vendor.slice("provider:".length) : vendor;
16
+ const normalized = raw
17
+ .toLowerCase()
18
+ .replace(/[^a-z0-9-]+/g, "-")
19
+ .replace(/^-+|-+$/g, "");
20
+ if (normalized.length === 0) throw new TypeError("provider source must name a provider");
21
+ return `provider:${/^[a-z]/.test(normalized) ? normalized : `p-${normalized}`}`;
22
+ };
@@ -0,0 +1,64 @@
1
+ import test, { mock } from "node:test";
2
+ import { strict as assert } from "node:assert";
3
+ import { ollamaProviderFromEnv } from "./ollama.ts";
4
+
5
+ const env = Object.freeze({
6
+ PLURNK_PROVIDERS_FETCH_TIMEOUT: "1000",
7
+ PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
8
+ PLURNK_PROVIDERS_REASONING: "off",
9
+ PLURNK_PROVIDERS_TEMPERATURE: "0.2",
10
+ PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15",
11
+ PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0",
12
+ PLURNK_PROVIDERS_REASONING_RESERVE: "10%",
13
+ PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
14
+ PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
15
+ PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
16
+ PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "1",
17
+ });
18
+
19
+ test.afterEach(() => mock.restoreAll());
20
+
21
+ // {§model-fact-resolution}
22
+ test("#126: Ollama always probes model physics and applies an operator context-window ceiling", async () => {
23
+ const calls: string[] = [];
24
+ mock.method(globalThis, "fetch", async (input: string | URL | Request) => {
25
+ calls.push(String(input));
26
+ return new Response(JSON.stringify({
27
+ model_info: { "qwen.context_length": 32_768 },
28
+ }));
29
+ });
30
+
31
+ const windows: Array<number | null> = [];
32
+ for (const operatorCap of [undefined, "8192", "65536"] as const) {
33
+ const provider = await ollamaProviderFromEnv({
34
+ ...env,
35
+ ...(operatorCap === undefined ? {} : { PLURNK_PROVIDERS_CONTEXT_WINDOW: operatorCap }),
36
+ }, "qwen2.5-coder", { baseUrl: "http://ollama.test:11434/v1" });
37
+ windows.push(provider.contextWindow);
38
+ }
39
+
40
+ assert.deepEqual(windows, [32_768, 8_192, 32_768]);
41
+ assert.deepEqual(calls, Array.from({ length: 3 }, () => "http://ollama.test:11434/api/show"));
42
+ });
43
+
44
+ test("#126: an operator ceiling does not hide an Ollama probe HTTP failure", async () => {
45
+ mock.method(globalThis, "fetch", async () => new Response(null, { status: 503 }));
46
+ await assert.rejects(
47
+ () => ollamaProviderFromEnv({
48
+ ...env,
49
+ PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
50
+ }, "qwen2.5-coder", { baseUrl: "http://ollama.test:11434" }),
51
+ /ollama provider: \/api\/show returned 503/,
52
+ );
53
+ });
54
+
55
+ test("#126: an operator ceiling does not hide a missing Ollama model fact", async () => {
56
+ mock.method(globalThis, "fetch", async () => new Response(JSON.stringify({ model_info: {} })));
57
+ await assert.rejects(
58
+ () => ollamaProviderFromEnv({
59
+ ...env,
60
+ PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
61
+ }, "qwen2.5-coder", { baseUrl: "http://ollama.test:11434" }),
62
+ /ollama provider: \/api\/show has no \*\.context_length key for "qwen2\.5-coder"/,
63
+ );
64
+ });
package/src/ollama.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  import { createOpenAICompatible } from "@ai-sdk/openai-compatible";
2
- import { contextWindowFromEnv, parseRequiredInt, requireEnv } from "./env.ts";
2
+ import { contextWindowFromEnv, effectiveContextWindow, parseRequiredInt, requireEnv } from "./env.ts";
3
3
  import { providerFromSdkModel } from "./catalogProvider.ts";
4
4
  import type { Provider, ProviderOptions } from "./types.ts";
5
5
 
@@ -46,8 +46,11 @@ export const ollamaProviderFromEnv = async (
46
46
  "PLURNK_PROVIDERS_FETCH_TIMEOUT",
47
47
  "ollama",
48
48
  );
49
- const contextWindow = contextWindowFromEnv(env, "ollama")
50
- ?? await fetchContextWindow({ baseUrl, model, timeout });
49
+ // {§model-fact-resolution}
50
+ const contextWindow = effectiveContextWindow(
51
+ contextWindowFromEnv(env, "ollama"),
52
+ await fetchContextWindow({ baseUrl, model, timeout }),
53
+ );
51
54
  const languageModel = createOpenAICompatible({
52
55
  name: "ollama",
53
56
  baseURL: `${baseUrl}/v1`,
package/src/openai.ts CHANGED
@@ -5,8 +5,11 @@ export type {
5
5
  FinishReason,
6
6
  Provider,
7
7
  ProviderAssistant,
8
+ ProviderAttempt,
9
+ ProviderAttemptFinishReason,
8
10
  ProviderResponse,
9
11
  ProviderUsage,
12
+ PromptTokenMeasurement,
10
13
  TokenAlternative,
11
14
  TokenLogprob,
12
15
  } from "./types.ts";
@@ -0,0 +1,41 @@
1
+ import type { ChatMessage, PromptTokenMeasurement } from "./types.ts";
2
+
3
+ const KINDS = new Set<PromptTokenMeasurement["kind"]>([
4
+ "exact",
5
+ "upper_bound",
6
+ "estimate",
7
+ ]);
8
+
9
+ export const assertPromptTokenMeasurement = (
10
+ value: unknown,
11
+ owner = "provider",
12
+ ): PromptTokenMeasurement => {
13
+ if (typeof value !== "object" || value === null) {
14
+ throw new TypeError(`${owner}: prompt token measurement must be an object`);
15
+ }
16
+ const candidate = value as Partial<PromptTokenMeasurement>;
17
+ if (!KINDS.has(candidate.kind as PromptTokenMeasurement["kind"])) {
18
+ throw new TypeError(`${owner}: prompt token measurement has invalid kind ${JSON.stringify(candidate.kind)}`);
19
+ }
20
+ if (!Number.isInteger(candidate.tokens) || candidate.tokens! < 0) {
21
+ throw new TypeError(`${owner}: prompt token measurement tokens must be a non-negative integer`);
22
+ }
23
+ if (typeof candidate.source !== "string" || candidate.source.length === 0) {
24
+ throw new TypeError(`${owner}: prompt token measurement source must be a non-empty string`);
25
+ }
26
+ if (candidate.kind === "estimate"
27
+ && (typeof candidate.detail !== "string" || candidate.detail.length === 0)) {
28
+ throw new TypeError(`${owner}: estimated prompt token measurement requires detail`);
29
+ }
30
+ return value as PromptTokenMeasurement;
31
+ };
32
+
33
+ export const estimatePromptTokens = (
34
+ messages: readonly ChatMessage[],
35
+ detail = "chars/2 over message content; provider request framing is unknown",
36
+ ): PromptTokenMeasurement => ({
37
+ kind: "estimate",
38
+ tokens: Math.ceil(messages.reduce((sum, { content }) => sum + content.length, 0) / 2),
39
+ source: "heuristic:chars2",
40
+ detail,
41
+ });
@@ -37,6 +37,13 @@ test("createSdkModel expands catalog endpoint variables without treating them as
37
37
  });
38
38
  });
39
39
 
40
+ test("#157: a cataloged compatible provider fails before transport when its declared credential is absent", () => {
41
+ assert.throws(
42
+ () => createSdkModel("deepseek", "deepseek-v4-flash", {}),
43
+ /deepseek provider: DEEPSEEK_API_KEY must be set/,
44
+ );
45
+ });
46
+
40
47
  test("createSdkModel fails clearly for a declared but unsupported SDK package", () => {
41
48
  assert.throws(
42
49
  () => createSdkModel("acme", "model", {
package/src/sdkModels.ts CHANGED
@@ -85,12 +85,6 @@ const baseUrl = (
85
85
  return value === undefined ? undefined : expandEnv(value, env, provider).replace(/\/+$/, "");
86
86
  };
87
87
 
88
- const apiKey = (
89
- provider: string,
90
- env: NodeJS.ProcessEnv,
91
- catalog: ProviderInfo,
92
- ): string | undefined => firstSet(env, configuredKeyNames(provider, env, catalog));
93
-
94
88
  const requireApiKey = (
95
89
  provider: string,
96
90
  env: NodeJS.ProcessEnv,
@@ -179,12 +173,14 @@ export const createSdkModel = (
179
173
  };
180
174
  case "@ai-sdk/openai-compatible":
181
175
  if (url === undefined) throw new Error(`${provider} provider: Models.dev supplies no API URL and no base URL was configured`);
176
+ const keyNames = configuredKeyNames(provider, env, catalog);
177
+ const key = keyNames.length === 0 ? undefined : requireApiKey(provider, env, catalog);
182
178
  return {
183
179
  compatible: {
184
180
  url: `${url}/chat/completions`,
185
- headers: apiKey(provider, env, catalog) === undefined
181
+ headers: key === undefined
186
182
  ? {}
187
- : { Authorization: `Bearer ${apiKey(provider, env, catalog)}` },
183
+ : { Authorization: `Bearer ${key}` },
188
184
  },
189
185
  catalog,
190
186
  };
package/src/types.ts CHANGED
@@ -1,15 +1,36 @@
1
1
  // Provider transport contract. Providers return raw wire-level output —
2
- // content unparsed (consumer parses via @plurnk/plurnk-grammar), reasoning
2
+ // content unparsed (consumer parses via @plurnk/plurnk-contracts), reasoning
3
3
  // is the wire-reported CoT only.
4
4
 
5
- import type { TelemetryEvent } from "./telemetry.ts";
5
+ import type { ProviderNotice } from "./notices.ts";
6
6
  import type { LanguageModel } from "ai";
7
+ import type {
8
+ PluginAttribution,
9
+ PluginAttributionContext,
10
+ PluginAttributionSource,
11
+ } from "@plurnk/plurnk-meta";
12
+ import type { ProviderCost } from "@plurnk/plurnk-contracts";
7
13
 
8
14
  export interface ChatMessage {
9
15
  role: "system" | "user" | "assistant";
10
16
  content: string;
11
17
  }
12
18
 
19
+ // Preflight evidence for the complete provider request. An empirical estimate
20
+ // is useful telemetry but cannot authorize a hard physical-capacity decision.
21
+ export type PromptTokenMeasurement =
22
+ | {
23
+ readonly kind: "exact" | "upper_bound";
24
+ readonly tokens: number;
25
+ readonly source: string;
26
+ }
27
+ | {
28
+ readonly kind: "estimate";
29
+ readonly tokens: number;
30
+ readonly source: string;
31
+ readonly detail: string;
32
+ };
33
+
13
34
  // Normalized token accounting. Invariant (enforced by normalizeUsage at the
14
35
  // provider boundary): total = prompt + completion + reasoning; cached is a
15
36
  // subset of prompt. `completion` is visible output EXCLUDING reasoning; the
@@ -23,11 +44,14 @@ export interface ProviderUsage {
23
44
  readonly total: number; // prompt + completion + reasoning
24
45
  }
25
46
 
26
- // Closed set per SPEC §2. Relay/aggregator providers MUST normalize wire
27
- // values back to one of these at the provider boundary.
47
+ export type AuthoritativeCharge = Extract<ProviderCost, { kind: "authoritative" }>;
48
+
49
+ // A successful exchange's closed finish set. ProviderAttemptFinishReason adds
50
+ // the failed disposition that may occur only on ProviderError attempt evidence.
28
51
  export type FinishReason = "stop" | "length" | "tool_calls" | "content_filter" | null;
52
+ export type ProviderAttemptFinishReason = FinishReason | "resource_interrupted";
29
53
 
30
- // A per-token logprob (#36, SPEC §14). `logprob` is the backend's RAW model
54
+ // {§provider-evidence} A per-token logprob. `logprob` is the backend's raw model
31
55
  // log-probability of the emitted token — the sampling-transform-invariant
32
56
  // confidence, chosen over Fireworks' post-mask `sampling_logprob` (measured
33
57
  // IDENTICAL under grammar, incl. an adversarial mask; the raw value is the honest
@@ -45,20 +69,24 @@ export interface TokenLogprob {
45
69
  readonly top?: readonly TokenAlternative[];
46
70
  }
47
71
 
48
- export interface ProviderAssistant {
72
+ // {§provider-encrypted-reasoning} `id` is provider detail identity; `subtype`
73
+ // is the provider's evidence-backed classification. Neither is a client entity
74
+ // correlation, so consumers must not substitute `id` for a message/tool-call ID.
75
+ export interface ProviderEncryptedReasoningItem {
76
+ readonly id: string | null;
77
+ readonly subtype: string;
78
+ readonly encrypted: ReadonlyArray<{ data: string; format: string | null }>;
79
+ }
80
+
81
+ export interface ProviderAssistant<TFinish extends ProviderAttemptFinishReason = FinishReason> {
49
82
  readonly content: string;
50
83
  readonly reasoning: string | null;
51
- // Sealed reasoning (#482): a relay backend (OpenRouter fronting OpenAI
52
- // o-series) returns the chain-of-thought ENCRYPTED — surfaced verbatim as
53
- // items { id, subtype, encrypted: [{data, format}] } (id from the wire,
54
- // subtype from wire position), never decoded, never synthesized. Readable text
55
- // stays on `reasoning`. Absent when the turn produced none; consumers (agui)
56
- // project it as REASONING_ENCRYPTED_VALUE.
57
- readonly reasoningEncrypted?: ReadonlyArray<{ id: string | null; subtype: string; encrypted: ReadonlyArray<{ data: string; format: string | null }> }>;
84
+ // Encrypted reasoning remains distinct from readable `reasoning`.
85
+ readonly reasoningEncrypted?: ReadonlyArray<ProviderEncryptedReasoningItem>;
58
86
  readonly usage: ProviderUsage;
59
- readonly finishReason: FinishReason;
87
+ readonly finishReason: TFinish;
60
88
  readonly model: string;
61
- // Per-token logprobs (#36), present ONLY when PLURNK_PROVIDERS_TOP_LOGPROBS is set
89
+ // Per-token logprobs, present only when PLURNK_PROVIDERS_TOP_LOGPROBS is set
62
90
  // AND the backend returned them. Absent otherwise — NEVER synthesized. Opt-in,
63
91
  // per-alias: a scraping alias enables it; serving turns carry nothing.
64
92
  readonly logprobs?: readonly TokenLogprob[];
@@ -66,42 +94,59 @@ export interface ProviderAssistant {
66
94
  readonly meanLogprob?: number;
67
95
  }
68
96
 
69
- export interface ProviderResponse {
70
- readonly assistant: ProviderAssistant;
97
+ export interface GrammarEvidence {
98
+ // Exact sentence observed at the grammar boundary before any reasoning/content
99
+ // projection. Offsets are Unicode code points, matching @plurnk/gbnf verdicts.
100
+ readonly input: string;
101
+ readonly contentStart: number;
102
+ readonly transported: boolean;
103
+ }
104
+
105
+ export interface ProviderResponse<TFinish extends ProviderAttemptFinishReason = FinishReason> {
106
+ readonly assistant: ProviderAssistant<TFinish>;
71
107
  readonly assistantRaw: unknown;
108
+ // A settled upstream charge is a validated public fact, not opaque metadata.
109
+ // Non-USD settlement carries an explicit provider-owned USD equivalent for
110
+ // the platform's existing USD aggregate. Core never supplies an FX rate.
111
+ readonly charge?: AuthoritativeCharge;
112
+ // {§gbnf-response-observation} — evidence only; the consumer owns the verdict.
113
+ readonly grammarEvidence?: GrammarEvidence;
72
114
  // Per-turn provider→client metadata bag: the backend's non-standard top-level
73
115
  // response fields passed through verbatim. Monetary values carry their own
74
116
  // amount and currency; the provider does not reinterpret them. The consumer
75
117
  // (service) merges this into its Turn metadata and
76
118
  // filters what reaches the client; it reads `meta`, never mines `assistantRaw`.
77
- // Absent when the backend reported no extra fields (#23, generalized).
119
+ // Absent when the backend reported no extra fields.
78
120
  readonly meta?: Record<string, unknown>;
79
- // The VERBATIM backend response body (#36, SPEC §14) — the full wire JSON for
121
+ // The verbatim backend response body ({§provider-evidence}) — the full wire JSON for
80
122
  // a non-streamed turn, or the reassembled equivalent for a streamed one.
81
123
  // `assistantRaw` is a normalized DIGEST (it drops choices[]); this is the
82
124
  // capture-everything record for the endpoint's fine-tune corpus. Present ONLY
83
125
  // when PLURNK_PROVIDERS_RAWBODY is on — off by default so serving turns never
84
126
  // carry it. Absent otherwise.
85
127
  readonly rawBody?: unknown;
86
- // Observations attached to a COMPLETED exchange (#24, SPEC §13). The model's
87
- // bytes always flow through `assistant`; these events annotate them. Today: a
88
- // `grammar_unenforced` event whenever the output diverges from the grammar —
89
- // transported OR withheld (filter mode) — carrying the divergence `position`.
90
- // The provider never adjudicates conformance; discard/retry/escalate/
91
- // self-correct is consumer policy. Absent when the turn produced no telemetry.
92
- readonly telemetry?: readonly TelemetryEvent[];
128
+ // Notices attached to the represented attempt. Successful
129
+ // returns may relay them; interrupted attempt notices remain forensic.
130
+ // Grammar conformance itself is consumer-owned.
131
+ readonly notices?: readonly ProviderNotice[];
93
132
  }
94
133
 
134
+ export type ProviderAttempt = ProviderResponse<ProviderAttemptFinishReason>;
135
+
95
136
  export interface Provider {
96
- // `grammar` is an optional GBNF string (canonically @plurnk/plurnk-grammar's
137
+ // Optional package-authored folksonomy evaluated by the consumer immediately
138
+ // before a provider emission attempt ({§plugin-attribution}).
139
+ attributions?(context: PluginAttributionContext): PluginAttribution;
140
+ // `grammar` is an optional GBNF string (canonically @plurnk/plurnk-contracts'
97
141
  // plurnk.gbnf, possibly root-substituted by the consumer). Backends that
98
142
  // support grammar-constrained sampling attach it verbatim; all others
99
143
  // ignore it. The provider never chooses or modifies the grammar — whether
100
- // to constrain and which root variant to send is consumer policy (SPEC §13).
144
+ // to constrain and which root variant to send is consumer policy
145
+ // ({§gbnf-response-observation}).
101
146
  //
102
147
  // `maxTokens` is the consumer's per-call output ceiling (wire `max_tokens`).
103
148
  // Without it, most servers generate UNBOUNDED (llama-server n_predict -1) —
104
- // under a multi-op grammar that degenerates to the context wall (SPEC §13),
149
+ // under a multi-op grammar that degenerates to the context wall,
105
150
  // so a constrained consumer is expected to pass it. Policy stays the
106
151
  // consumer's; the provider only transports.
107
152
  //
@@ -109,70 +154,64 @@ export interface Provider {
109
154
  // stream (loop/run). Providers MAY key backend affinity on it — e.g.
110
155
  // llama-server slot pinning for KV-cache reuse — and MUST NOT interpret
111
156
  // its content. The consumer never sees or chooses backend resources
112
- // (slot integers, connections); the *mechanism* is the provider's (#11).
157
+ // (slot integers, connections); the mechanism is the provider's.
113
158
  //
114
- // `attributions` (per-turn, runtime-observed) and `client` (workspace-stable,
115
- // self-identified) are first-party telemetry the consumer hands down: which
116
- // installed plugin packages dispatched this turn, and which frontend
117
- // originated the worker. They are forwarded ONLY by a provider whose spec opts
118
- // in (the first-party `plurnk` endpoint, via `Plurnk-Attribution` /
119
- // `Plurnk-Client` headers); every other provider DROPS them — the gate is
120
- // structural so first-party metadata can never leak to a third-party backend.
159
+ // `attributions` is opaque consumer-supplied creator telemetry; the consumer
160
+ // owns what contribution that set claims ({§attribution}). `client` is the
161
+ // consumer's workspace-stable, self-identified frontend. They are forwarded ONLY by a
162
+ // provider whose spec opts in (the first-party `plurnk` endpoint, via
163
+ // `Plurnk-Attribution` / `Plurnk-Client` headers); every other provider DROPS
164
+ // them — the gate is structural so first-party metadata can never leak to a
165
+ // third-party backend.
121
166
  //
122
167
  // `sampling` is an optional bag of standard OpenAI-compat sampling params
123
168
  // (temperature, top_p, top_k, min_p, penalties, stop, seed, …) forwarded into
124
169
  // the request body UNDER the provider's managed fields — model/messages/grammar/
125
170
  // reasoning/max_tokens/slot always win, and transport/protocol keys (stream,
126
171
  // response_format, grammar, id_slot) are stripped, so it carries sampling intent
127
- // only and can't bypass grammar transport (SPEC §8 holds). A PROXY consumer (the
172
+ // only and can't bypass grammar transport ({§provider-request-authority}). A
173
+ // proxy consumer (the
128
174
  // plurnk endpoint fronting its own backends) uses it to pass its caller's sampling
129
175
  // knobs through; a direct consumer typically leaves it unset.
130
176
  //
131
177
  // `strikes` is the worker's CURRENT rail-strike streak at time-of-generate
132
- // (0 = clean; a clean turn zeroes it; every loop starts at 0 — contract:
133
- // plurnk-service#313). Forwarded as a `Plurnk-Strikes` header ONLY under the
178
+ // (0 = clean; a clean turn zeroes it; every loop starts at 0 — contract
179
+ // {§strikes-first-party-metadata}). Forwarded as a `Plurnk-Strikes` header ONLY under the
134
180
  // same firstPartyMetadata gate as attributions/client; dropped everywhere
135
181
  // else. Headers only — the packet NEVER carries strike state (the model must
136
182
  // not see engine accounting; it would become a metric to game).
137
183
  //
138
- // `workspaceId`/`loop`/`turn` (#404, per #391) are the turn COORDINATE — the
184
+ // `workspaceId`/`loop`/`turn` are the turn coordinate ({§lifecycle-terms}) — the
139
185
  // daemon-side sequence of the turn being generated, which the endpoint can
140
186
  // never scrape from the wire. Forwarded as `Plurnk-Workspace-Id`/`Plurnk-Loop`/
141
187
  // `Plurnk-Turn` ONLY under the same firstPartyMetadata gate; dropped
142
188
  // everywhere else. Coordinates are 1-based: absent/0 emits no header (no
143
189
  // strikes-style zero exception). Headers only, never the packet.
144
190
  generate(args: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse>;
145
- // The model's context window in tokens. The provider RESOLVES it (operator pin
146
- // -> live probe -> @plurnk/plurnk-models catalog). A CLOUD provider (no probe)
147
- // FAILS AT CONSTRUCTION when it can't (#419/#417: never budget against a wrong
148
- // number). A PROBING provider (openai/llama-server) instead DEGRADES to null on a
149
- // probe miss - a blip must not crash it (#34) - and surfaces it once
150
- // (PLURNK_CONTEXT_UNKNOWN). So null still means "window unknown -> no cap"; the
151
- // consumer must NOT improvise a stand-in from it (#421). NOTE: under llama-server
152
- // --parallel N, the window is PER SLOT (the server splits --ctx-size across slots
153
- // and reports the divided value).
191
+ // {§model-fact-resolution} — effective physical context in tokens. `null`
192
+ // means unknown; under llama-server parallelism the probed value is per slot.
154
193
  readonly contextWindow: number | null;
155
194
  readonly model: string;
156
- // OPTIONAL (#37): the backend's SELF-REPORTED served model id, from a
195
+ // Optional: the backend's self-reported served model id, from a
157
196
  // /v1/models-shaped probe (llama-server today; any such backend). For a local
158
197
  // alias, `model` is the alias but this is the real served name (the .gguf) the
159
198
  // tokenizer seam maps exactly. Read-only, best-effort, no extra probing —
160
199
  // absent when no probe ran. Consumers resolve `servedModel ?? model`.
161
200
  readonly servedModel?: string;
162
- // OPTIONAL resolved capability (#34): true when a transported grammar will
201
+ // Optional resolved capability: true when a transported grammar will
163
202
  // actually constrain the decode (rails LIVE), false/undefined otherwise —
164
203
  // introspectable so the consumer can fail hard on a dark-rails boot instead
165
204
  // of discovering it from unconstrained emissions.
166
205
  readonly constrainsOutput?: boolean;
167
- // OPTIONAL resolved capability (#43): true when this backend decodes
206
+ // Optional resolved capability: true when this backend decodes
168
207
  // UNBOUNDED absent a caller cap — llama-server honors n_predict to the
169
- // context wall (the 30,736-junk-token wall-run, providers#10), so a consumer
170
- // MUST bring an output envelope (SPEC §13). Cloud backends that silently
208
+ // context wall (observed in a 30,736-junk-token wall run), so a consumer
209
+ // MUST bring an output envelope ({§provider-generation-envelope}). Cloud backends that silently
171
210
  // clamp an over-ask (fireworks/xai, verified live) never set this; undefined
172
211
  // = no claim. Introspectable so a consumer can refuse AT BOOT a local alias
173
212
  // with no declared envelope, instead of dying mid-turn in partition math.
174
213
  readonly requiresMaxTokens?: boolean;
175
- // OPTIONAL generation-envelope reserves (#507, owner-ruled) — the amounts OF
214
+ // Optional generation-envelope reserves ({§provider-generation-envelope}) — the amounts of
176
215
  // the DETECTED window reserved for reasoning and completion: floor
177
216
  // percentages of `contextWindow`, or absolute per-alias pins that win
178
217
  // outright. The consumer's prompt budget is `contextWindow - reasoningReserve
@@ -182,22 +221,25 @@ export interface Provider {
182
221
  // as null). All first-party providers claim, so null means genuinely-unknown.
183
222
  readonly reasoningReserve?: number | null;
184
223
  readonly completionReserve?: number | null;
185
- // Provider-owned tokenizer. Synchronous, non-negative integer. Without an
186
- // exact family configured this is the chars/2 UPPER BOUND (surfaced at
187
- // construction, never silent) — safe for refusal math, not an exact count.
188
- countTokens(text: string): number;
224
+ // Provider-owned preflight measurement of the complete chat request,
225
+ // including provider/template framing when the adapter can know it.
226
+ // Estimates are explicit and MUST NOT authorize hard physical admission.
227
+ countPromptTokens(messages: readonly ChatMessage[], signal?: AbortSignal): Promise<PromptTokenMeasurement>;
189
228
  // OPTIONAL capability: exact tokenization served by the backend's own vocab
190
229
  // (llama-server /tokenize) — token ids in the model's real vocabulary.
191
230
  // Present ONLY when the backend exposes such an endpoint (probe-gated);
192
231
  // `tokenize === undefined` means the backend can't. Exact-counting
193
232
  // consumers (the tokenizer seam) prefer this over any client-side data.
194
233
  tokenize?(text: string): Promise<number[]>;
195
- // Provider-owned estimated cost calculation. Returns USD.
196
- // Returns 0 for siblings/models with no known rates.
234
+ // {§model-fact-resolution} — frozen 1.x local USD estimate compatibility.
235
+ // Consumers adapt its zero to unknown; only calculateCharge can prove free.
197
236
  calculateCost(usage: ProviderUsage): number;
237
+ // Current monetary result. The numeric calculateCost surface remains frozen
238
+ // for 1.x compatibility; zero on that legacy surface cannot prove `free`.
239
+ calculateCharge?(usage: ProviderUsage): Exclude<ProviderCost, AuthoritativeCharge>;
198
240
  }
199
241
 
200
- // ProviderAlias moved to @plurnk/plurnk-aliases (the zero-dep parser, #27);
242
+ // ProviderAlias lives in @plurnk/plurnk-aliases (the zero-dependency parser);
201
243
  // index.ts re-exports it so the "." surface is unchanged.
202
244
 
203
245
  // Per-alias instantiation overrides, threaded from the alias cascade into the
@@ -211,6 +253,6 @@ export interface ProviderOptions {
211
253
 
212
254
  // A discovered provider plugin default-exports an AI SDK provider. PLURNK owns
213
255
  // the adapter into Provider; the plugin owns only its protocol binding.
214
- export interface AiSdkProviderPlugin {
256
+ export interface AiSdkProviderPlugin extends PluginAttributionSource {
215
257
  languageModel(model: string): LanguageModel;
216
258
  }