@plurnk/plurnk-providers 1.3.12 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. package/.env.defaults +47 -38
  2. package/README.md +68 -4
  3. package/SPEC.md +245 -62
  4. package/dist/AiSdkProvider.d.ts +19 -5
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +264 -156
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +15 -17
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +27 -10
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -2
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +41 -8
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/ProviderRegistry.d.ts +4 -1
  17. package/dist/ProviderRegistry.d.ts.map +1 -1
  18. package/dist/ProviderRegistry.js +7 -3
  19. package/dist/ProviderRegistry.js.map +1 -1
  20. package/dist/accounting.d.ts +3 -0
  21. package/dist/accounting.d.ts.map +1 -0
  22. package/dist/accounting.js +84 -0
  23. package/dist/accounting.js.map +1 -0
  24. package/dist/aiSdkTransport.d.ts +5 -2
  25. package/dist/aiSdkTransport.d.ts.map +1 -1
  26. package/dist/aiSdkTransport.js +91 -5
  27. package/dist/aiSdkTransport.js.map +1 -1
  28. package/dist/catalogProvider.d.ts +5 -2
  29. package/dist/catalogProvider.d.ts.map +1 -1
  30. package/dist/catalogProvider.js +24 -11
  31. package/dist/catalogProvider.js.map +1 -1
  32. package/dist/compatibleProvider.d.ts.map +1 -1
  33. package/dist/compatibleProvider.js +11 -4
  34. package/dist/compatibleProvider.js.map +1 -1
  35. package/dist/cost.d.ts +11 -0
  36. package/dist/cost.d.ts.map +1 -0
  37. package/dist/cost.js +61 -0
  38. package/dist/cost.js.map +1 -0
  39. package/dist/discover.d.ts +2 -0
  40. package/dist/discover.d.ts.map +1 -1
  41. package/dist/discover.js +15 -9
  42. package/dist/discover.js.map +1 -1
  43. package/dist/env.d.ts +4 -6
  44. package/dist/env.d.ts.map +1 -1
  45. package/dist/env.js +27 -29
  46. package/dist/env.js.map +1 -1
  47. package/dist/errors.d.ts +27 -0
  48. package/dist/errors.d.ts.map +1 -0
  49. package/dist/errors.js +152 -0
  50. package/dist/errors.js.map +1 -0
  51. package/dist/index.d.ts +12 -6
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +8 -5
  54. package/dist/index.js.map +1 -1
  55. package/dist/notices.d.ts +10 -0
  56. package/dist/notices.d.ts.map +1 -0
  57. package/dist/notices.js +11 -0
  58. package/dist/notices.js.map +1 -0
  59. package/dist/ollama.d.ts.map +1 -1
  60. package/dist/ollama.js +3 -3
  61. package/dist/ollama.js.map +1 -1
  62. package/dist/openai.d.ts +1 -1
  63. package/dist/openai.d.ts.map +1 -1
  64. package/dist/promptTokens.d.ts +4 -0
  65. package/dist/promptTokens.d.ts.map +1 -0
  66. package/dist/promptTokens.js +32 -0
  67. package/dist/promptTokens.js.map +1 -0
  68. package/dist/sdkModels.d.ts +2 -0
  69. package/dist/sdkModels.d.ts.map +1 -1
  70. package/dist/sdkModels.js +17 -6
  71. package/dist/sdkModels.js.map +1 -1
  72. package/dist/types.d.ts +52 -16
  73. package/dist/types.d.ts.map +1 -1
  74. package/dist/types.js +1 -1
  75. package/dist/types.js.map +1 -1
  76. package/dist/usage.d.ts +4 -0
  77. package/dist/usage.d.ts.map +1 -1
  78. package/dist/usage.js +64 -19
  79. package/dist/usage.js.map +1 -1
  80. package/dist/warnings.js +0 -0
  81. package/dist/warnings.js.map +1 -1
  82. package/package.json +15 -10
  83. package/src/AiSdkProvider.test.ts +750 -169
  84. package/src/AiSdkProvider.ts +354 -200
  85. package/src/Mock.test.ts +29 -14
  86. package/src/Mock.ts +36 -15
  87. package/src/Pool.test.ts +43 -6
  88. package/src/Pool.ts +56 -10
  89. package/src/ProviderRegistry.test.ts +158 -9
  90. package/src/ProviderRegistry.ts +19 -6
  91. package/src/accounting.test.ts +58 -0
  92. package/src/accounting.ts +88 -0
  93. package/src/aiSdkTransport.ts +101 -8
  94. package/src/boundaries.test.ts +9 -3
  95. package/src/catalogProvider.test.ts +43 -15
  96. package/src/catalogProvider.ts +32 -16
  97. package/src/compatibleProvider.test.ts +96 -0
  98. package/src/compatibleProvider.ts +15 -6
  99. package/src/cost.test.ts +64 -0
  100. package/src/cost.ts +78 -0
  101. package/src/defaults.test.ts +1 -0
  102. package/src/discover.test.ts +48 -7
  103. package/src/discover.ts +31 -21
  104. package/src/env.test.ts +38 -48
  105. package/src/env.ts +43 -40
  106. package/src/errors.test.ts +148 -0
  107. package/src/errors.ts +208 -0
  108. package/src/index.ts +30 -8
  109. package/src/lexicon-guard.test.ts +6 -6
  110. package/src/notices.ts +22 -0
  111. package/src/ollama.test.ts +64 -0
  112. package/src/ollama.ts +6 -3
  113. package/src/openai.ts +3 -0
  114. package/src/promptTokens.ts +41 -0
  115. package/src/sdkModels.test.ts +29 -3
  116. package/src/sdkModels.ts +19 -11
  117. package/src/types.ts +125 -64
  118. package/src/usage.test.ts +24 -5
  119. package/src/usage.ts +72 -21
  120. package/src/warnings.test.ts +10 -10
  121. package/src/warnings.ts +0 -0
  122. package/dist/OpenAICompat.d.ts +0 -76
  123. package/dist/OpenAICompat.d.ts.map +0 -1
  124. package/dist/OpenAICompat.js +0 -555
  125. package/dist/OpenAICompat.js.map +0 -1
  126. package/dist/openaiStream.d.ts +0 -47
  127. package/dist/openaiStream.d.ts.map +0 -1
  128. package/dist/openaiStream.js +0 -280
  129. package/dist/openaiStream.js.map +0 -1
  130. package/dist/standardProviders.d.ts +0 -31
  131. package/dist/standardProviders.d.ts.map +0 -1
  132. package/dist/standardProviders.js +0 -518
  133. package/dist/standardProviders.js.map +0 -1
  134. package/dist/telemetry.d.ts +0 -24
  135. package/dist/telemetry.d.ts.map +0 -1
  136. package/dist/telemetry.js +0 -85
  137. package/dist/telemetry.js.map +0 -1
  138. package/src/telemetry.test.ts +0 -69
  139. package/src/telemetry.ts +0 -116
package/src/sdkModels.ts CHANGED
@@ -6,13 +6,15 @@ import { createGroq } from "@ai-sdk/groq";
6
6
  import { createMistral } from "@ai-sdk/mistral";
7
7
  import { createOpenAI } from "@ai-sdk/openai";
8
8
  import { createTogetherAI } from "@ai-sdk/togetherai";
9
- import { createXai } from "@ai-sdk/xai";
10
9
  import { createOpenRouter } from "@openrouter/ai-sdk-provider";
11
10
  import { lookupProvider, type ProviderInfo } from "@plurnk/plurnk-models";
12
11
  import type { LanguageModel } from "ai";
12
+ import { authoritativeChargeNormalizer } from "./accounting.ts";
13
+ import type { AuthoritativeChargeNormalizer } from "./types.ts";
13
14
 
14
15
  export type SdkModel = {
15
16
  readonly languageModel?: LanguageModel;
17
+ readonly normalizeCharge?: AuthoritativeChargeNormalizer;
16
18
  readonly compatible?: {
17
19
  readonly url: string;
18
20
  readonly headers: Readonly<Record<string, string>>;
@@ -85,12 +87,6 @@ const baseUrl = (
85
87
  return value === undefined ? undefined : expandEnv(value, env, provider).replace(/\/+$/, "");
86
88
  };
87
89
 
88
- const apiKey = (
89
- provider: string,
90
- env: NodeJS.ProcessEnv,
91
- catalog: ProviderInfo,
92
- ): string | undefined => firstSet(env, configuredKeyNames(provider, env, catalog));
93
-
94
90
  const requireApiKey = (
95
91
  provider: string,
96
92
  env: NodeJS.ProcessEnv,
@@ -111,6 +107,7 @@ export const createSdkModel = (
111
107
  const catalog = lookupProvider(provider) ?? configuredProviderInfo(provider, env);
112
108
  if (catalog === null) return null;
113
109
  const url = baseUrl(provider, env, catalog, baseUrlOverride);
110
+ const normalizeCharge = authoritativeChargeNormalizer(catalog.npm);
114
111
 
115
112
  switch (catalog.npm) {
116
113
  case "@ai-sdk/openai":
@@ -136,6 +133,7 @@ export const createSdkModel = (
136
133
  case "@ai-sdk/deepinfra":
137
134
  return {
138
135
  languageModel: createDeepInfra({ apiKey: requireApiKey(provider, env, catalog), baseURL: url }).languageModel(model),
136
+ ...(normalizeCharge === undefined ? {} : { normalizeCharge }),
139
137
  catalog,
140
138
  };
141
139
  case "@ai-sdk/google":
@@ -143,11 +141,18 @@ export const createSdkModel = (
143
141
  languageModel: createGoogle({ apiKey: requireApiKey(provider, env, catalog), baseURL: url }).languageModel(model),
144
142
  catalog,
145
143
  };
146
- case "@ai-sdk/xai":
144
+ case "@ai-sdk/xai": {
145
+ const key = requireApiKey(provider, env, catalog);
146
+ const compatibleBase = url ?? "https://api.x.ai/v1";
147
147
  return {
148
- languageModel: createXai({ apiKey: requireApiKey(provider, env, catalog), baseURL: url }).languageModel(model),
148
+ compatible: {
149
+ url: `${compatibleBase}/chat/completions`,
150
+ headers: { Authorization: `Bearer ${key}` },
151
+ },
152
+ ...(normalizeCharge === undefined ? {} : { normalizeCharge }),
149
153
  catalog,
150
154
  };
155
+ }
151
156
  case "@ai-sdk/anthropic":
152
157
  return {
153
158
  languageModel: createAnthropic({ apiKey: requireApiKey(provider, env, catalog), baseURL: url }).languageModel(model),
@@ -175,16 +180,19 @@ export const createSdkModel = (
175
180
  ...(env.OPENROUTER_X_TITLE === undefined ? {} : { "X-Title": env.OPENROUTER_X_TITLE }),
176
181
  },
177
182
  }).languageModel(model),
183
+ ...(normalizeCharge === undefined ? {} : { normalizeCharge }),
178
184
  catalog,
179
185
  };
180
186
  case "@ai-sdk/openai-compatible":
181
187
  if (url === undefined) throw new Error(`${provider} provider: Models.dev supplies no API URL and no base URL was configured`);
188
+ const keyNames = configuredKeyNames(provider, env, catalog);
189
+ const key = keyNames.length === 0 ? undefined : requireApiKey(provider, env, catalog);
182
190
  return {
183
191
  compatible: {
184
192
  url: `${url}/chat/completions`,
185
- headers: apiKey(provider, env, catalog) === undefined
193
+ headers: key === undefined
186
194
  ? {}
187
- : { Authorization: `Bearer ${apiKey(provider, env, catalog)}` },
195
+ : { Authorization: `Bearer ${key}` },
188
196
  },
189
197
  catalog,
190
198
  };
package/src/types.ts CHANGED
@@ -1,15 +1,36 @@
1
1
  // Provider transport contract. Providers return raw wire-level output —
2
- // content unparsed (consumer parses via @plurnk/plurnk-grammar), reasoning
2
+ // content unparsed (consumer parses via @plurnk/plurnk-contracts), reasoning
3
3
  // is the wire-reported CoT only.
4
4
 
5
- import type { TelemetryEvent } from "./telemetry.ts";
5
+ import type { ProviderNotice } from "./notices.ts";
6
6
  import type { LanguageModel } from "ai";
7
+ import type {
8
+ PluginAttribution,
9
+ PluginAttributionContext,
10
+ PluginAttributionSource,
11
+ } from "@plurnk/plurnk-meta";
12
+ import type { ProviderCost } from "@plurnk/plurnk-contracts";
7
13
 
8
14
  export interface ChatMessage {
9
15
  role: "system" | "user" | "assistant";
10
16
  content: string;
11
17
  }
12
18
 
19
+ // Preflight evidence for the complete provider request. An empirical estimate
20
+ // is useful telemetry but cannot authorize a hard physical-capacity decision.
21
+ export type PromptTokenMeasurement =
22
+ | {
23
+ readonly kind: "exact" | "upper_bound";
24
+ readonly tokens: number;
25
+ readonly source: string;
26
+ }
27
+ | {
28
+ readonly kind: "estimate";
29
+ readonly tokens: number;
30
+ readonly source: string;
31
+ readonly detail: string;
32
+ };
33
+
13
34
  // Normalized token accounting. Invariant (enforced by normalizeUsage at the
14
35
  // provider boundary): total = prompt + completion + reasoning; cached is a
15
36
  // subset of prompt. `completion` is visible output EXCLUDING reasoning; the
@@ -23,11 +44,33 @@ export interface ProviderUsage {
23
44
  readonly total: number; // prompt + completion + reasoning
24
45
  }
25
46
 
26
- // Closed set per SPEC §2. Relay/aggregator providers MUST normalize wire
27
- // values back to one of these at the provider boundary.
47
+ export type AuthoritativeCharge = Extract<ProviderCost, { kind: "authoritative" }>;
48
+
49
+ // Evidence exposed by the transport to the provider adapter that owns its
50
+ // vendor protocol. Core and downstream consumers receive only the normalized
51
+ // charge, never a requirement to understand provider metadata fields.
52
+ export interface ProviderChargeEvidence {
53
+ readonly providerMetadata?: unknown;
54
+ // Provider-owned raw usage projection retained independently of optional
55
+ // full-body capture. Accounting fields cannot disappear merely because
56
+ // forensic raw-body capture is disabled.
57
+ readonly usage?: unknown;
58
+ readonly response: {
59
+ readonly id: string;
60
+ readonly headers?: Readonly<Record<string, string>>;
61
+ };
62
+ }
63
+
64
+ export type AuthoritativeChargeNormalizer = (
65
+ evidence: ProviderChargeEvidence,
66
+ ) => AuthoritativeCharge | undefined;
67
+
68
+ // A successful exchange's closed finish set. ProviderAttemptFinishReason adds
69
+ // the failed disposition that may occur only on ProviderError attempt evidence.
28
70
  export type FinishReason = "stop" | "length" | "tool_calls" | "content_filter" | null;
71
+ export type ProviderAttemptFinishReason = FinishReason | "resource_interrupted";
29
72
 
30
- // A per-token logprob (#36, SPEC §14). `logprob` is the backend's RAW model
73
+ // {§provider-evidence} A per-token logprob. `logprob` is the backend's raw model
31
74
  // log-probability of the emitted token — the sampling-transform-invariant
32
75
  // confidence, chosen over Fireworks' post-mask `sampling_logprob` (measured
33
76
  // IDENTICAL under grammar, incl. an adversarial mask; the raw value is the honest
@@ -45,20 +88,24 @@ export interface TokenLogprob {
45
88
  readonly top?: readonly TokenAlternative[];
46
89
  }
47
90
 
48
- export interface ProviderAssistant {
91
+ // {§provider-encrypted-reasoning} `id` is provider detail identity; `subtype`
92
+ // is the provider's evidence-backed classification. Neither is a client entity
93
+ // correlation, so consumers must not substitute `id` for a message/tool-call ID.
94
+ export interface ProviderEncryptedReasoningItem {
95
+ readonly id: string | null;
96
+ readonly subtype: string;
97
+ readonly encrypted: ReadonlyArray<{ data: string; format: string | null }>;
98
+ }
99
+
100
+ export interface ProviderAssistant<TFinish extends ProviderAttemptFinishReason = FinishReason> {
49
101
  readonly content: string;
50
102
  readonly reasoning: string | null;
51
- // Sealed reasoning (#482): a relay backend (OpenRouter fronting OpenAI
52
- // o-series) returns the chain-of-thought ENCRYPTED — surfaced verbatim as
53
- // items { id, subtype, encrypted: [{data, format}] } (id from the wire,
54
- // subtype from wire position), never decoded, never synthesized. Readable text
55
- // stays on `reasoning`. Absent when the turn produced none; consumers (agui)
56
- // project it as REASONING_ENCRYPTED_VALUE.
57
- readonly reasoningEncrypted?: ReadonlyArray<{ id: string | null; subtype: string; encrypted: ReadonlyArray<{ data: string; format: string | null }> }>;
103
+ // Encrypted reasoning remains distinct from readable `reasoning`.
104
+ readonly reasoningEncrypted?: ReadonlyArray<ProviderEncryptedReasoningItem>;
58
105
  readonly usage: ProviderUsage;
59
- readonly finishReason: FinishReason;
106
+ readonly finishReason: TFinish;
60
107
  readonly model: string;
61
- // Per-token logprobs (#36), present ONLY when PLURNK_PROVIDERS_TOP_LOGPROBS is set
108
+ // Per-token logprobs, present only when PLURNK_PROVIDERS_TOP_LOGPROBS is set
62
109
  // AND the backend returned them. Absent otherwise — NEVER synthesized. Opt-in,
63
110
  // per-alias: a scraping alias enables it; serving turns carry nothing.
64
111
  readonly logprobs?: readonly TokenLogprob[];
@@ -66,42 +113,59 @@ export interface ProviderAssistant {
66
113
  readonly meanLogprob?: number;
67
114
  }
68
115
 
69
- export interface ProviderResponse {
70
- readonly assistant: ProviderAssistant;
116
+ export interface GrammarEvidence {
117
+ // Exact sentence observed at the grammar boundary before any reasoning/content
118
+ // projection. Offsets are Unicode code points, matching @plurnk/gbnf verdicts.
119
+ readonly input: string;
120
+ readonly contentStart: number;
121
+ readonly transported: boolean;
122
+ }
123
+
124
+ export interface ProviderResponse<TFinish extends ProviderAttemptFinishReason = FinishReason> {
125
+ readonly assistant: ProviderAssistant<TFinish>;
71
126
  readonly assistantRaw: unknown;
127
+ // A settled upstream charge is a validated public fact, not opaque metadata.
128
+ // Non-USD settlement carries an explicit provider-owned USD equivalent for
129
+ // the platform's existing USD aggregate. Core never supplies an FX rate.
130
+ readonly charge?: AuthoritativeCharge;
131
+ // {§gbnf-response-observation} — evidence only; the consumer owns the verdict.
132
+ readonly grammarEvidence?: GrammarEvidence;
72
133
  // Per-turn provider→client metadata bag: the backend's non-standard top-level
73
134
  // response fields passed through verbatim. Monetary values carry their own
74
135
  // amount and currency; the provider does not reinterpret them. The consumer
75
136
  // (service) merges this into its Turn metadata and
76
137
  // filters what reaches the client; it reads `meta`, never mines `assistantRaw`.
77
- // Absent when the backend reported no extra fields (#23, generalized).
138
+ // Absent when the backend reported no extra fields.
78
139
  readonly meta?: Record<string, unknown>;
79
- // The VERBATIM backend response body (#36, SPEC §14) — the full wire JSON for
140
+ // The verbatim backend response body ({§provider-evidence}) — the full wire JSON for
80
141
  // a non-streamed turn, or the reassembled equivalent for a streamed one.
81
142
  // `assistantRaw` is a normalized DIGEST (it drops choices[]); this is the
82
143
  // capture-everything record for the endpoint's fine-tune corpus. Present ONLY
83
144
  // when PLURNK_PROVIDERS_RAWBODY is on — off by default so serving turns never
84
145
  // carry it. Absent otherwise.
85
146
  readonly rawBody?: unknown;
86
- // Observations attached to a COMPLETED exchange (#24, SPEC §13). The model's
87
- // bytes always flow through `assistant`; these events annotate them. Today: a
88
- // `grammar_unenforced` event whenever the output diverges from the grammar —
89
- // transported OR withheld (filter mode) — carrying the divergence `position`.
90
- // The provider never adjudicates conformance; discard/retry/escalate/
91
- // self-correct is consumer policy. Absent when the turn produced no telemetry.
92
- readonly telemetry?: readonly TelemetryEvent[];
147
+ // Notices attached to the represented attempt. Successful
148
+ // returns may relay them; interrupted attempt notices remain forensic.
149
+ // Grammar conformance itself is consumer-owned.
150
+ readonly notices?: readonly ProviderNotice[];
93
151
  }
94
152
 
153
+ export type ProviderAttempt = ProviderResponse<ProviderAttemptFinishReason>;
154
+
95
155
  export interface Provider {
96
- // `grammar` is an optional GBNF string (canonically @plurnk/plurnk-grammar's
156
+ // Optional package-authored folksonomy evaluated by the consumer immediately
157
+ // before a provider emission attempt ({§plugin-attribution}).
158
+ attributions?(context: PluginAttributionContext): PluginAttribution;
159
+ // `grammar` is an optional GBNF string (canonically @plurnk/plurnk-contracts'
97
160
  // plurnk.gbnf, possibly root-substituted by the consumer). Backends that
98
161
  // support grammar-constrained sampling attach it verbatim; all others
99
162
  // ignore it. The provider never chooses or modifies the grammar — whether
100
- // to constrain and which root variant to send is consumer policy (SPEC §13).
163
+ // to constrain and which root variant to send is consumer policy
164
+ // ({§gbnf-response-observation}).
101
165
  //
102
166
  // `maxTokens` is the consumer's per-call output ceiling (wire `max_tokens`).
103
167
  // Without it, most servers generate UNBOUNDED (llama-server n_predict -1) —
104
- // under a multi-op grammar that degenerates to the context wall (SPEC §13),
168
+ // under a multi-op grammar that degenerates to the context wall,
105
169
  // so a constrained consumer is expected to pass it. Policy stays the
106
170
  // consumer's; the provider only transports.
107
171
  //
@@ -109,70 +173,64 @@ export interface Provider {
109
173
  // stream (loop/run). Providers MAY key backend affinity on it — e.g.
110
174
  // llama-server slot pinning for KV-cache reuse — and MUST NOT interpret
111
175
  // its content. The consumer never sees or chooses backend resources
112
- // (slot integers, connections); the *mechanism* is the provider's (#11).
176
+ // (slot integers, connections); the mechanism is the provider's.
113
177
  //
114
- // `attributions` (per-turn, runtime-observed) and `client` (workspace-stable,
115
- // self-identified) are first-party telemetry the consumer hands down: which
116
- // installed plugin packages dispatched this turn, and which frontend
117
- // originated the worker. They are forwarded ONLY by a provider whose spec opts
118
- // in (the first-party `plurnk` endpoint, via `Plurnk-Attribution` /
119
- // `Plurnk-Client` headers); every other provider DROPS them — the gate is
120
- // structural so first-party metadata can never leak to a third-party backend.
178
+ // `attributions` is opaque consumer-supplied creator telemetry; the consumer
179
+ // owns what contribution that set claims ({§attribution}). `client` is the
180
+ // consumer's workspace-stable, self-identified frontend. They are forwarded ONLY by a
181
+ // provider whose spec opts in (the first-party `plurnk` endpoint, via
182
+ // `Plurnk-Attribution` / `Plurnk-Client` headers); every other provider DROPS
183
+ // them — the gate is structural so first-party metadata can never leak to a
184
+ // third-party backend.
121
185
  //
122
186
  // `sampling` is an optional bag of standard OpenAI-compat sampling params
123
187
  // (temperature, top_p, top_k, min_p, penalties, stop, seed, …) forwarded into
124
188
  // the request body UNDER the provider's managed fields — model/messages/grammar/
125
189
  // reasoning/max_tokens/slot always win, and transport/protocol keys (stream,
126
190
  // response_format, grammar, id_slot) are stripped, so it carries sampling intent
127
- // only and can't bypass grammar transport (SPEC §8 holds). A PROXY consumer (the
191
+ // only and can't bypass grammar transport ({§provider-request-authority}). A
192
+ // proxy consumer (the
128
193
  // plurnk endpoint fronting its own backends) uses it to pass its caller's sampling
129
194
  // knobs through; a direct consumer typically leaves it unset.
130
195
  //
131
196
  // `strikes` is the worker's CURRENT rail-strike streak at time-of-generate
132
- // (0 = clean; a clean turn zeroes it; every loop starts at 0 — contract:
133
- // plurnk-service#313). Forwarded as a `Plurnk-Strikes` header ONLY under the
197
+ // (0 = clean; a clean turn zeroes it; every loop starts at 0 — contract
198
+ // {§strikes-first-party-metadata}). Forwarded as a `Plurnk-Strikes` header ONLY under the
134
199
  // same firstPartyMetadata gate as attributions/client; dropped everywhere
135
200
  // else. Headers only — the packet NEVER carries strike state (the model must
136
201
  // not see engine accounting; it would become a metric to game).
137
202
  //
138
- // `workspaceId`/`loop`/`turn` (#404, per #391) are the turn COORDINATE — the
203
+ // `workspaceId`/`loop`/`turn` are the turn coordinate ({§lifecycle-terms}) — the
139
204
  // daemon-side sequence of the turn being generated, which the endpoint can
140
205
  // never scrape from the wire. Forwarded as `Plurnk-Workspace-Id`/`Plurnk-Loop`/
141
206
  // `Plurnk-Turn` ONLY under the same firstPartyMetadata gate; dropped
142
207
  // everywhere else. Coordinates are 1-based: absent/0 emits no header (no
143
208
  // strikes-style zero exception). Headers only, never the packet.
144
209
  generate(args: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse>;
145
- // The model's context window in tokens. The provider RESOLVES it (operator pin
146
- // -> live probe -> @plurnk/plurnk-models catalog). A CLOUD provider (no probe)
147
- // FAILS AT CONSTRUCTION when it can't (#419/#417: never budget against a wrong
148
- // number). A PROBING provider (openai/llama-server) instead DEGRADES to null on a
149
- // probe miss - a blip must not crash it (#34) - and surfaces it once
150
- // (PLURNK_CONTEXT_UNKNOWN). So null still means "window unknown -> no cap"; the
151
- // consumer must NOT improvise a stand-in from it (#421). NOTE: under llama-server
152
- // --parallel N, the window is PER SLOT (the server splits --ctx-size across slots
153
- // and reports the divided value).
210
+ // {§model-fact-resolution} — effective physical context in tokens. `null`
211
+ // means unknown; under llama-server parallelism the probed value is per slot.
154
212
  readonly contextWindow: number | null;
155
213
  readonly model: string;
156
- // OPTIONAL (#37): the backend's SELF-REPORTED served model id, from a
214
+ // Optional: the backend's self-reported served model id, from a
157
215
  // /v1/models-shaped probe (llama-server today; any such backend). For a local
158
216
  // alias, `model` is the alias but this is the real served name (the .gguf) the
159
217
  // tokenizer seam maps exactly. Read-only, best-effort, no extra probing —
160
218
  // absent when no probe ran. Consumers resolve `servedModel ?? model`.
161
219
  readonly servedModel?: string;
162
- // OPTIONAL resolved capability (#34): true when a transported grammar will
220
+ // Optional resolved capability: true when a transported grammar will
163
221
  // actually constrain the decode (rails LIVE), false/undefined otherwise —
164
222
  // introspectable so the consumer can fail hard on a dark-rails boot instead
165
223
  // of discovering it from unconstrained emissions.
166
224
  readonly constrainsOutput?: boolean;
167
- // OPTIONAL resolved capability (#43): true when this backend decodes
225
+ // Optional resolved capability: true when this backend decodes
168
226
  // UNBOUNDED absent a caller cap — llama-server honors n_predict to the
169
- // context wall (the 30,736-junk-token wall-run, providers#10), so a consumer
170
- // MUST bring an output envelope (SPEC §13). Cloud backends that silently
227
+ // context wall (observed in a 30,736-junk-token wall run), so a consumer
228
+ // MUST bring an output envelope ({§provider-generation-envelope}). Cloud backends that silently
171
229
  // clamp an over-ask (fireworks/xai, verified live) never set this; undefined
172
230
  // = no claim. Introspectable so a consumer can refuse AT BOOT a local alias
173
231
  // with no declared envelope, instead of dying mid-turn in partition math.
174
232
  readonly requiresMaxTokens?: boolean;
175
- // OPTIONAL generation-envelope reserves (#507, owner-ruled) — the amounts OF
233
+ // Optional generation-envelope reserves ({§provider-generation-envelope}) — the amounts of
176
234
  // the DETECTED window reserved for reasoning and completion: floor
177
235
  // percentages of `contextWindow`, or absolute per-alias pins that win
178
236
  // outright. The consumer's prompt budget is `contextWindow - reasoningReserve
@@ -182,22 +240,25 @@ export interface Provider {
182
240
  // as null). All first-party providers claim, so null means genuinely-unknown.
183
241
  readonly reasoningReserve?: number | null;
184
242
  readonly completionReserve?: number | null;
185
- // Provider-owned tokenizer. Synchronous, non-negative integer. Without an
186
- // exact family configured this is the chars/2 UPPER BOUND (surfaced at
187
- // construction, never silent) — safe for refusal math, not an exact count.
188
- countTokens(text: string): number;
243
+ // Provider-owned preflight measurement of the complete chat request,
244
+ // including provider/template framing when the adapter can know it.
245
+ // Estimates are explicit and MUST NOT authorize hard physical admission.
246
+ countPromptTokens(messages: readonly ChatMessage[], signal?: AbortSignal): Promise<PromptTokenMeasurement>;
189
247
  // OPTIONAL capability: exact tokenization served by the backend's own vocab
190
248
  // (llama-server /tokenize) — token ids in the model's real vocabulary.
191
249
  // Present ONLY when the backend exposes such an endpoint (probe-gated);
192
250
  // `tokenize === undefined` means the backend can't. Exact-counting
193
251
  // consumers (the tokenizer seam) prefer this over any client-side data.
194
252
  tokenize?(text: string): Promise<number[]>;
195
- // Provider-owned estimated cost calculation. Returns USD.
196
- // Returns 0 for siblings/models with no known rates.
253
+ // {§model-fact-resolution} — frozen 1.x local USD estimate compatibility.
254
+ // This is not a monetary-reporting authority.
197
255
  calculateCost(usage: ProviderUsage): number;
256
+ // Models.dev-derived monetary fallback. Direct response charges travel on
257
+ // ProviderResponse and take precedence at the consuming attempt boundary.
258
+ calculateCharge?(usage: ProviderUsage): Exclude<ProviderCost, AuthoritativeCharge>;
198
259
  }
199
260
 
200
- // ProviderAlias moved to @plurnk/plurnk-aliases (the zero-dep parser, #27);
261
+ // ProviderAlias lives in @plurnk/plurnk-aliases (the zero-dependency parser);
201
262
  // index.ts re-exports it so the "." surface is unchanged.
202
263
 
203
264
  // Per-alias instantiation overrides, threaded from the alias cascade into the
@@ -211,6 +272,6 @@ export interface ProviderOptions {
211
272
 
212
273
  // A discovered provider plugin default-exports an AI SDK provider. PLURNK owns
213
274
  // the adapter into Provider; the plugin owns only its protocol binding.
214
- export interface AiSdkProviderPlugin {
275
+ export interface AiSdkProviderPlugin extends PluginAttributionSource {
215
276
  languageModel(model: string): LanguageModel;
216
277
  }
package/src/usage.test.ts CHANGED
@@ -1,6 +1,6 @@
1
1
  import test from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
- import { normalizeUsage, calculateCostUsd } from "./usage.ts";
3
+ import { normalizeUsage, calculateCostUsd, calculateCostUsdDecimal } from "./usage.ts";
4
4
 
5
5
  // — normalizeUsage —
6
6
 
@@ -23,7 +23,7 @@ test("normalizeUsage: OpenAI-style — reasoning split out of completion_tokens"
23
23
  assert.equal(u.completion + u.reasoning, 100); // billable output unchanged
24
24
  });
25
25
 
26
- test("normalizeUsage: xAI/Grok-style — reasoning is ADDITIVE, not subtracted from completion (#28)", () => {
26
+ test("normalizeUsage: xAI/Grok-style — reasoning is ADDITIVE, not subtracted from completion", () => {
27
27
  // Real grok-4.3 shape: completion_tokens is visible-only; reasoning_tokens is
28
28
  // detailed but ADDITIVE — total = prompt + completion + reasoning.
29
29
  const u = normalizeUsage({
@@ -48,6 +48,17 @@ test("normalizeUsage: top-level cached_tokens still honored", () => {
48
48
  assert.equal(u.cached, 12);
49
49
  });
50
50
 
51
+ test("#157: normalizeUsage maps DeepSeek's prompt cache hit count", () => {
52
+ const u = normalizeUsage({
53
+ prompt_tokens: 50,
54
+ prompt_cache_hit_tokens: 30,
55
+ prompt_cache_miss_tokens: 20,
56
+ completion_tokens: 10,
57
+ total_tokens: 60,
58
+ });
59
+ assert.deepEqual(u, { prompt: 50, completion: 10, reasoning: 0, cached: 30, total: 60 });
60
+ });
61
+
51
62
  test("normalizeUsage: no reasoning — plain prompt+completion", () => {
52
63
  const u = normalizeUsage({ prompt_tokens: 10, completion_tokens: 20, total_tokens: 30 });
53
64
  assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
@@ -63,9 +74,9 @@ test("normalizeUsage: absent usage → all zeros", () => {
63
74
  assert.deepEqual(normalizeUsage(undefined), { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 });
64
75
  });
65
76
 
66
- // -- Fireworks-style: reasoning shipped as TEXT, folded into completion, not itemized (#425) --
77
+ // -- Fireworks-style: reasoning shipped as TEXT, folded into completion, not itemized --
67
78
 
68
- test("normalizeUsage: fireworks folds reasoning into completion -- re-split by text proportion, sum preserved (#425)", () => {
79
+ test("normalizeUsage: fireworks folds reasoning into completion -- re-split by text proportion, sum preserved", () => {
69
80
  // total = prompt + completion (no gap), reasoning_tokens absent, but 750 vs 250
70
81
  // chars of reasoning vs content came back. Split completion 75/25; cost base held.
71
82
  const u = normalizeUsage(
@@ -78,7 +89,7 @@ test("normalizeUsage: fireworks folds reasoning into completion -- re-split by t
78
89
  assert.equal(u.prompt + u.completion + u.reasoning, u.total); // invariant
79
90
  });
80
91
 
81
- test("normalizeUsage: pure-reasoning turn (empty content) attributes all completion to reasoning (#425)", () => {
92
+ test("normalizeUsage: pure-reasoning turn (empty content) attributes all completion to reasoning", () => {
82
93
  // The run52 runaway shape: 0 visible content, the whole budget spent reasoning.
83
94
  const u = normalizeUsage(
84
95
  { prompt_tokens: 100, completion_tokens: 500, total_tokens: 600 },
@@ -132,3 +143,11 @@ test("calculateCostUsd: zero rates → 0", () => {
132
143
  const usage = { prompt: 9, completion: 9, reasoning: 9, cached: 9, total: 27 };
133
144
  assert.equal(calculateCostUsd(usage, { input: 0, output: 0, cached: 0 }), 0);
134
145
  });
146
+
147
+ test("calculateCostUsdDecimal preserves Models.dev rates without floating-point artifacts", () => {
148
+ const usage = { prompt: 1_000, completion: 100, reasoning: 50, cached: 400, total: 1_150 };
149
+ assert.equal(
150
+ calculateCostUsdDecimal(usage, { input: 0.14, output: 0.28, cached: 0.0028 }),
151
+ "0.00012712",
152
+ );
153
+ });
package/src/usage.ts CHANGED
@@ -20,16 +20,39 @@ export type RawUsage = {
20
20
  completion_tokens?: number;
21
21
  total_tokens?: number;
22
22
  cached_tokens?: number;
23
+ prompt_cache_hit_tokens?: number;
24
+ prompt_cache_miss_tokens?: number;
23
25
  prompt_tokens_details?: { cached_tokens?: number };
24
26
  completion_tokens_details?: { reasoning_tokens?: number };
25
27
  };
26
28
 
29
+ // Some providers return distinct reasoning text while reporting one combined
30
+ // output count. Attribute that count without disturbing an upstream split.
31
+ export const attributeUnitemizedReasoning = (
32
+ usage: ProviderUsage,
33
+ reasoningText: string,
34
+ contentText: string,
35
+ ): ProviderUsage => {
36
+ if (usage.reasoning !== 0 || usage.completion === 0 || reasoningText.length === 0) return usage;
37
+ const reasoning = contentText.length === 0
38
+ ? usage.completion
39
+ : Math.round(usage.completion * reasoningText.length / (reasoningText.length + contentText.length));
40
+ return {
41
+ ...usage,
42
+ completion: usage.completion - reasoning,
43
+ reasoning,
44
+ };
45
+ };
46
+
27
47
  export const normalizeUsage = (raw: RawUsage | null | undefined, reasoningText = "", contentText = ""): ProviderUsage => {
28
48
  const prompt = raw?.prompt_tokens ?? 0;
29
49
  const completionRaw = raw?.completion_tokens ?? 0;
30
50
  const reportedTotal = raw?.total_tokens ?? 0;
31
51
  // OpenAI nests cached under prompt_tokens_details; others put it top-level.
32
- const cached = raw?.prompt_tokens_details?.cached_tokens ?? raw?.cached_tokens ?? 0;
52
+ const cached = raw?.prompt_tokens_details?.cached_tokens
53
+ ?? raw?.prompt_cache_hit_tokens
54
+ ?? raw?.cached_tokens
55
+ ?? 0;
33
56
  const reasoningDetail = raw?.completion_tokens_details?.reasoning_tokens;
34
57
 
35
58
  let completion: number;
@@ -40,7 +63,7 @@ export const normalizeUsage = (raw: RawUsage | null | undefined, reasoningText =
40
63
  // o-series folds it INTO completion_tokens (subset: total = prompt + completion,
41
64
  // so completion must have reasoning subtracted out). xAI/Grok reports it
42
65
  // ADDITIVE to a visible-only completion_tokens (total = prompt + completion +
43
- // reasoning), where subtracting wrongly zeroes the visible output (#28). Tell
66
+ // reasoning), where subtracting wrongly zeroes the visible output. Tell
44
67
  // them apart by the total identity; with no total reported, fall back on the
45
68
  // impossible-subset signal — reasoning can't exceed the completion it's a
46
69
  // subset of.
@@ -53,34 +76,62 @@ export const normalizeUsage = (raw: RawUsage | null | undefined, reasoningText =
53
76
  // reasoning. Only trust the gap when a total was actually reported.
54
77
  reasoning = reportedTotal > 0 ? Math.max(0, reportedTotal - prompt - completionRaw) : 0;
55
78
  completion = completionRaw;
56
- // Fireworks folds reasoning INTO completion_tokens and itemizes no
57
- // reasoning_tokens, so a turn that shipped only reasoning reads reasoning=0
58
- // though 300k chars of it arrived (#425). When reasoning TEXT came back but
59
- // the reported totals leave no gap, re-split the reported completion by the
60
- // emitted text proportions. Sum-preserving: billable output
61
- // (completion+reasoning) and cost are byte-identical; only the
62
- // visible/reasoning gauge is corrected (pure-reasoning turn -> completion 0).
63
- if (reasoning === 0 && reasoningText.length > 0 && reportedTotal > 0 && completionRaw > 0) {
64
- reasoning = Math.round(completionRaw * reasoningText.length / (reasoningText.length + contentText.length));
65
- completion = completionRaw - reasoning;
66
- }
67
79
  }
68
80
  const total = reportedTotal > 0 ? reportedTotal : prompt + completion + reasoning;
69
- return { prompt, completion, reasoning, cached, total };
81
+ const usage = { prompt, completion, reasoning, cached, total };
82
+ // Fireworks folds reasoning INTO completion_tokens and itemizes no
83
+ // reasoning_tokens. A reported total establishes that completion is an
84
+ // upstream quantity rather than a locally synthesized fallback.
85
+ return reasoningDetail === undefined && reportedTotal > 0
86
+ ? attributeUnitemizedReasoning(usage, reasoningText, contentText)
87
+ : usage;
70
88
  };
71
89
 
72
90
  // Conventional provider pricing: USD per million tokens, matching Models.dev.
73
91
  export type TokenRates = { input: number; output: number; cached: number };
74
92
 
93
+ const decimalParts = (value: number): { coefficient: bigint; scale: number } => {
94
+ if (!Number.isFinite(value) || value < 0) {
95
+ throw new TypeError("token rates must be finite non-negative numbers");
96
+ }
97
+ const match = /^(\d+)(?:\.(\d+))?(?:e([+-]?\d+))?$/i.exec(String(value));
98
+ if (match === null) throw new TypeError(`cannot represent token rate ${value} as a decimal`);
99
+ const fraction = match[2] ?? "";
100
+ const exponent = Number(match[3] ?? "0");
101
+ const scale = fraction.length - exponent;
102
+ const coefficient = BigInt(`${match[1]}${fraction}`);
103
+ return scale < 0
104
+ ? { coefficient: coefficient * 10n ** BigInt(-scale), scale: 0 }
105
+ : { coefficient, scale };
106
+ };
107
+
108
+ const decimalString = (coefficient: bigint, scale: number): string => {
109
+ const digits = String(coefficient).padStart(scale + 1, "0");
110
+ if (scale === 0) return digits;
111
+ const integer = digits.slice(0, -scale);
112
+ const fraction = digits.slice(-scale).replace(/0+$/, "");
113
+ return fraction.length === 0 ? integer : `${integer}.${fraction}`;
114
+ };
115
+
116
+ // Models.dev rates are decimal USD-per-million values. Calculate their
117
+ // projection as decimal arithmetic so ProviderCost preserves the table and
118
+ // exact response counts without binary floating-point artifacts.
119
+ export const calculateCostUsdDecimal = (usage: ProviderUsage, rates: TokenRates): string => {
120
+ const parts = [rates.input, rates.cached, rates.output].map(decimalParts);
121
+ const rateScale = Math.max(...parts.map(({ scale }) => scale));
122
+ const [input, cached, output] = parts.map(({ coefficient, scale }) =>
123
+ coefficient * 10n ** BigInt(rateScale - scale));
124
+ const nonCachedPrompt = Math.max(0, usage.prompt - usage.cached);
125
+ const outputTokens = usage.completion + usage.reasoning;
126
+ const coefficient = BigInt(nonCachedPrompt) * input!
127
+ + BigInt(usage.cached) * cached!
128
+ + BigInt(outputTokens) * output!;
129
+ return decimalString(coefficient, rateScale + 6);
130
+ };
131
+
75
132
  // The one cost formula every provider uses: non-cached prompt at the input
76
133
  // rate, cached prompt at the cache rate, and billable output (completion +
77
134
  // reasoning) at the output rate.
78
135
  export const calculateCostUsd = (usage: ProviderUsage, rates: TokenRates): number => {
79
- const nonCachedPrompt = Math.max(0, usage.prompt - usage.cached);
80
- const output = usage.completion + usage.reasoning;
81
- return (
82
- nonCachedPrompt * rates.input
83
- + usage.cached * rates.cached
84
- + output * rates.output
85
- ) / 1_000_000;
136
+ return Number(calculateCostUsdDecimal(usage, rates));
86
137
  };
@@ -4,24 +4,24 @@ import { emitWarningOnce, resetEmittedWarnings } from "./warnings.ts";
4
4
 
5
5
  test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); });
6
6
 
7
- test("#40: same (code, message) fires once per process", () => {
7
+ test("same (code, message) fires once per process", () => {
8
8
  const seen: string[] = [];
9
9
  mock.method(process, "emitWarning", (msg: string | Error) => { seen.push(String(msg)); });
10
- emitWarningOnce("openai provider: heuristic", "PLURNK_TOKENIZER_HEURISTIC");
11
- emitWarningOnce("openai provider: heuristic", "PLURNK_TOKENIZER_HEURISTIC");
12
- emitWarningOnce("openai provider: heuristic", "PLURNK_TOKENIZER_HEURISTIC");
13
- assert.deepEqual(seen, ["openai provider: heuristic"]);
10
+ emitWarningOnce("openai provider: estimate", "PLURNK_PROMPT_COUNT_ESTIMATE");
11
+ emitWarningOnce("openai provider: estimate", "PLURNK_PROMPT_COUNT_ESTIMATE");
12
+ emitWarningOnce("openai provider: estimate", "PLURNK_PROMPT_COUNT_ESTIMATE");
13
+ assert.deepEqual(seen, ["openai provider: estimate"]);
14
14
  });
15
15
 
16
- test("#40: dedup is by (code, MESSAGE), not code — a second provider's surfacing is never suppressed", () => {
16
+ test("dedup is by (code, MESSAGE), not code — a second provider's surfacing is never suppressed", () => {
17
17
  const seen: string[] = [];
18
18
  mock.method(process, "emitWarning", (msg: string | Error) => { seen.push(String(msg)); });
19
- emitWarningOnce("openai provider: heuristic", "PLURNK_TOKENIZER_HEURISTIC");
20
- emitWarningOnce("groq provider: heuristic", "PLURNK_TOKENIZER_HEURISTIC"); // same code, different provider
21
- assert.deepEqual(seen, ["openai provider: heuristic", "groq provider: heuristic"]);
19
+ emitWarningOnce("openai provider: estimate", "PLURNK_PROMPT_COUNT_ESTIMATE");
20
+ emitWarningOnce("groq provider: estimate", "PLURNK_PROMPT_COUNT_ESTIMATE"); // same code, different provider
21
+ assert.deepEqual(seen, ["openai provider: estimate", "groq provider: estimate"]);
22
22
  });
23
23
 
24
- test("#40: resetEmittedWarnings clears the set (test-order independence)", () => {
24
+ test("resetEmittedWarnings clears the set (test-order independence)", () => {
25
25
  const seen: string[] = [];
26
26
  mock.method(process, "emitWarning", (msg: string | Error) => { seen.push(String(msg)); });
27
27
  emitWarningOnce("m", "C");