@plurnk/plurnk-providers 1.5.0 → 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/.env.defaults +41 -34
  2. package/README.md +15 -0
  3. package/SPEC.md +242 -89
  4. package/dist/AiSdkProvider.d.ts +33 -33
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +442 -133
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +10 -11
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +87 -25
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +9 -24
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +86 -25
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts +5 -2
  17. package/dist/accounting.d.ts.map +1 -1
  18. package/dist/accounting.js +100 -16
  19. package/dist/accounting.js.map +1 -1
  20. package/dist/accountingPublic.d.ts +5 -0
  21. package/dist/accountingPublic.d.ts.map +1 -0
  22. package/dist/accountingPublic.js +3 -0
  23. package/dist/accountingPublic.js.map +1 -0
  24. package/dist/aiSdkTransport.d.ts +9 -2
  25. package/dist/aiSdkTransport.d.ts.map +1 -1
  26. package/dist/aiSdkTransport.js +160 -62
  27. package/dist/aiSdkTransport.js.map +1 -1
  28. package/dist/capacity.d.ts +26 -0
  29. package/dist/capacity.d.ts.map +1 -0
  30. package/dist/capacity.js +90 -0
  31. package/dist/capacity.js.map +1 -0
  32. package/dist/catalogProvider.d.ts +8 -3
  33. package/dist/catalogProvider.d.ts.map +1 -1
  34. package/dist/catalogProvider.js +45 -41
  35. package/dist/catalogProvider.js.map +1 -1
  36. package/dist/compatibleProvider.d.ts.map +1 -1
  37. package/dist/compatibleProvider.js +26 -12
  38. package/dist/compatibleProvider.js.map +1 -1
  39. package/dist/cost.d.ts +10 -10
  40. package/dist/cost.d.ts.map +1 -1
  41. package/dist/cost.js +90 -42
  42. package/dist/cost.js.map +1 -1
  43. package/dist/env.d.ts +13 -11
  44. package/dist/env.d.ts.map +1 -1
  45. package/dist/env.js +83 -46
  46. package/dist/env.js.map +1 -1
  47. package/dist/errors.d.ts +17 -3
  48. package/dist/errors.d.ts.map +1 -1
  49. package/dist/errors.js +91 -8
  50. package/dist/errors.js.map +1 -1
  51. package/dist/index.d.ts +7 -6
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +5 -3
  54. package/dist/index.js.map +1 -1
  55. package/dist/ollama.js +3 -3
  56. package/dist/ollama.js.map +1 -1
  57. package/dist/promptTokens.d.ts.map +1 -1
  58. package/dist/promptTokens.js +7 -4
  59. package/dist/promptTokens.js.map +1 -1
  60. package/dist/sdkModels.d.ts +7 -2
  61. package/dist/sdkModels.d.ts.map +1 -1
  62. package/dist/sdkModels.js +43 -13
  63. package/dist/sdkModels.js.map +1 -1
  64. package/dist/types.d.ts +55 -33
  65. package/dist/types.d.ts.map +1 -1
  66. package/dist/usage.d.ts +22 -5
  67. package/dist/usage.d.ts.map +1 -1
  68. package/dist/usage.js +169 -83
  69. package/dist/usage.js.map +1 -1
  70. package/package.json +18 -7
  71. package/src/AiSdkProvider.test.ts +964 -206
  72. package/src/AiSdkProvider.ts +545 -155
  73. package/src/Mock.test.ts +69 -30
  74. package/src/Mock.ts +99 -29
  75. package/src/Pool.test.ts +90 -19
  76. package/src/Pool.ts +96 -27
  77. package/src/ProviderRegistry.test.ts +16 -11
  78. package/src/accounting.test.ts +58 -22
  79. package/src/accounting.ts +119 -18
  80. package/src/accountingPublic.ts +9 -0
  81. package/src/aiSdkTransport.test.ts +42 -49
  82. package/src/aiSdkTransport.ts +174 -62
  83. package/src/boundaries.test.ts +2 -0
  84. package/src/capacity.test.ts +92 -0
  85. package/src/capacity.ts +140 -0
  86. package/src/catalogProvider.test.ts +339 -30
  87. package/src/catalogProvider.ts +65 -47
  88. package/src/compatibleProvider.test.ts +7 -5
  89. package/src/compatibleProvider.ts +29 -13
  90. package/src/cost.test.ts +86 -36
  91. package/src/cost.ts +111 -50
  92. package/src/defaults.test.ts +13 -3
  93. package/src/env.test.ts +103 -25
  94. package/src/env.ts +153 -65
  95. package/src/errors.test.ts +80 -2
  96. package/src/errors.ts +107 -8
  97. package/src/index.ts +26 -7
  98. package/src/ollama.test.ts +5 -3
  99. package/src/ollama.ts +3 -3
  100. package/src/promptTokens.ts +8 -5
  101. package/src/sdkModels.test.ts +77 -8
  102. package/src/sdkModels.ts +51 -15
  103. package/src/types.ts +112 -51
  104. package/src/usage.test.ts +112 -116
  105. package/src/usage.ts +214 -93
package/SPEC.md CHANGED
@@ -9,8 +9,8 @@ ordinary provider protocols.
9
9
  The provider stack has four owners:
10
10
 
11
11
  1. Models.dev supplies a release-time snapshot of provider package, API
12
- endpoint, credential names, models, context windows, output limits,
13
- reasoning capability, and USD prices.
12
+ endpoint, credential names, models, context/input/output limits, reasoning
13
+ capability, and USD rates including distinct reasoning rates when supplied.
14
14
  2. Official AI SDK providers own vendor request and response protocols.
15
15
  3. This package owns the PLURNK contract: aliases, envelopes, normalized usage
16
16
  and errors, evidence, local capabilities, and first-party metadata.
@@ -22,6 +22,14 @@ model prefix, context window, price, or vendor request shape into a PLURNK
22
22
  table. A missing or wrong catalog fact is fixed upstream, overridden through a
23
23
  provider declaration, or left explicitly unknown.
24
24
 
25
+ §provider-runtime-neutral-accounting The public
26
+ `@plurnk/plurnk-providers/accounting` subpath exposes provider-request
27
+ aggregation and Models.dev cost estimation, plus their wire types, without
28
+ evaluating Node-only provider discovery, filesystem defaults, or runtime
29
+ construction. It re-exports the package's sole accounting implementation. The
30
+ package root remains the Node provider-runtime composition surface and is not a
31
+ Worker entrypoint.
32
+
25
33
  ## §2 Provider interface
26
34
 
27
35
  §provider-interface `Provider` exposes immutable model facts and one generation
@@ -31,88 +39,139 @@ operation:
31
39
  interface Provider {
32
40
  readonly model: string;
33
41
  readonly contextWindow: number | null;
42
+ readonly maxInputTokens: number | null;
43
+ readonly maxOutputTokens: number | null;
44
+ readonly outputBudget: number | null;
45
+ readonly reasoningBudget: number | null;
46
+ readonly inputCapacity: number | null;
34
47
  readonly servedModel?: string;
35
48
  readonly constrainsOutput?: boolean;
36
- readonly requiresMaxTokens?: boolean;
37
- readonly reasoningReserve?: number | null;
38
- readonly completionReserve?: number | null;
49
+ readonly requiresOutputBudget?: boolean;
39
50
 
40
51
  countPromptTokens(
41
52
  messages: readonly ChatMessage[],
42
53
  signal?: AbortSignal,
43
54
  ): Promise<PromptTokenMeasurement>;
55
+ assessRequestCapacity(
56
+ messages: readonly ChatMessage[],
57
+ maxOutputTokens?: number,
58
+ signal?: AbortSignal,
59
+ ): Promise<ProviderRequestCapacity>;
44
60
  tokenize?(text: string): Promise<number[]>;
45
- calculateCost(usage: ProviderUsage): number;
46
- calculateCharge?(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }>;
47
61
  generate(args: GenerateArgs): Promise<ProviderResponse>;
48
62
  }
49
63
  ```
50
64
 
51
- `contextWindow` is provider physics and resolves under
52
- {§model-fact-resolution}. `null` means genuinely unknown; a consumer MUST NOT
53
- invent a stand-in. The context-window knob never carries model-facing prompt
54
- policy or grinder pressure.
65
+ `contextWindow` is the effective total context envelope resolved under
66
+ {§model-fact-resolution}: the minimum of known model capacity and any stricter
67
+ operator cap. `null` means genuinely unknown; a consumer MUST NOT invent a
68
+ stand-in. The context-window knob is a hard cap, never model-facing grinder
69
+ pressure.
55
70
 
56
- `PromptTokenMeasurement` is a discriminated request-level result:
71
+ §provider-prompt-measurement `PromptTokenMeasurement` is a discriminated
72
+ request-level result:
57
73
 
58
- | `kind` | Meaning | May authorize physical admission |
59
- | ------------- | ------------------------------------------------------ | -------------------------------- |
60
- | `exact` | Exact count for the complete provider request. | Yes. |
61
- | `upper_bound` | Proven upper bound for the complete provider request. | Yes. |
62
- | `estimate` | Empirical prediction with required causal `detail`. | No. |
74
+ | `kind` | Meaning | Capacity authority |
75
+ | --- | --- | --- |
76
+ | `exact` | Exact count for the complete provider request. | May prove fit or overflow. |
77
+ | `upper_bound` | Proven upper bound for the complete provider request. | May prove fit; exceeding a limit does not prove overflow. |
78
+ | `estimate` | Empirical prediction with required causal `detail`. | Cannot admit or reject. |
79
+ | `unavailable` | No quantified measurement, with required causal `detail`. | Cannot admit or reject. |
63
80
 
64
- Every result carries a non-negative integer `tokens` and a non-empty `source`.
81
+ Every result carries a non-empty `source`; quantified kinds carry non-negative
82
+ integer `tokens`.
65
83
  `countPromptTokens` receives the same messages supplied to `generate` and may
66
84
  perform cancellable provider I/O. The common fallback is chars/2 over message
67
85
  content; it is announced once and reported honestly as an estimate because it
68
86
  knows neither the serving vocabulary nor provider-owned request framing.
69
- An unavailable optional counting endpoint likewise returns an estimate naming
70
- the cause. A malformed measurement is a provider contract violation and fails
71
- hard; consumers do not reinterpret it as ordinary unavailability.
87
+ An adapter may retain that estimate when an optional counting endpoint fails,
88
+ provided its detail names the cause; one unable to quantify anything returns
89
+ `unavailable`. A malformed measurement is a provider contract violation and
90
+ fails hard.
91
+
92
+ §provider-capacity-admission `assessRequestCapacity` intersects every known
93
+ physical input constraint: independent `maxInputTokens` and
94
+ `contextWindow - outputBudget`. Its result is `admit`, `reject`, or `defer` and
95
+ retains the complete limit and measurement evidence. Exact fit admits; exact
96
+ overflow rejects. A proven upper bound admits only when it fits. Unknown limits,
97
+ an upper bound above a limit, estimates, and unavailable measurements defer to
98
+ the upstream provider as capacity oracle. The same stable intersection is
99
+ exposed as `inputCapacity`; `null` means the available limits cannot establish
100
+ one. A known combined context and output budget must leave positive input
101
+ capacity. Consumers may display or use that fact as policy, but MUST NOT
102
+ substitute their own content heuristic for request-shaped admission.
72
103
 
73
104
  `tokenize` is the separate content-token capability and exists only when the
74
105
  endpoint exposes its real vocabulary. Content tokenization does not substitute
75
106
  for complete-request measurement.
76
107
 
77
- §provider-monetary-evidence One precedence path converts each provider response
78
- into {§provider-cost} evidence:
108
+ §provider-monetary-evidence One precedence path converts each physical provider
109
+ request into {§provider-cost} evidence before the request leaves the provider
110
+ boundary:
79
111
 
80
112
  | Precedence | Evidence | Result |
81
113
  | --- | --- | --- |
82
- | 1 | A documented monetary field on that response | The adapter normalizes it as `ProviderResponse.charge`; its decimal USD equivalent wins. |
83
- | 2 | Exact response usage and the exact model's Models.dev rates | `calculateCharge` returns estimated USD, or explicit free when every applicable rate is zero. |
84
- | 3 | Neither | Unknown; no cost is reported. |
114
+ | 1 | A documented monetary field on that response or error | The adapter validates and preserves its documented `charged` or `estimated` character; its exact amount wins. |
115
+ | 2 | Known response usage and the exact model's Models.dev rates | The adapter returns an exact decimal USD `estimated` amount only when every differently-priced applicable category is known. |
116
+ | 3 | Neither | `unknown` with a concrete reason. |
85
117
 
86
- Models.dev is the sole supported fallback rate table. Missing usage or rates
87
- never proves free. The numeric `calculateCost` method remains callable only as a
88
- frozen 1.x compatibility surface and is not a monetary-reporting authority.
89
- `providerCostUsd` exposes authoritative, estimated, and free evidence as decimal
90
- USD and returns `null` for unknown evidence.
118
+ Models.dev is the sole supported fallback rate table. Missing usage, a missing
119
+ applicable category, or missing rates never proves zero. An exact zero rate
120
+ produces an ordinary estimated amount of USD `0`. Rate calculation is internal
121
+ to the provider request; Core, digest, ping, and clients never call a parallel
122
+ pricing method.
91
123
 
92
124
  ### Generation
93
125
 
94
- `generate` requires a non-empty, stable, opaque `workerId`. It accepts:
126
+ §provider-cache-identity `generate` requires a non-empty, stable, opaque
127
+ `workerId`. A durable worker uses one globally unique value for its lifetime;
128
+ independent databases and processes cannot mint the same local sequence. A
129
+ `bare` call instead uses a fresh per-call value, preventing unrelated prompts
130
+ from acquiring affinity with either the parent worker or another BARE call.
131
+ Providers MUST NOT interpret either value.
132
+
133
+ `generate` accepts:
95
134
 
96
135
  - `messages`: system, user, and assistant text messages;
97
136
  - caller cancellation through `signal`;
98
- - optional `grammar` and `maxTokens`;
137
+ - optional `grammar` and call-specific `maxOutputTokens` tightening;
99
138
  - standard `sampling` intent;
139
+ - the caller-owned `callKind` output contract when one applies;
100
140
  - opaque attribution tags plus client, strike, workspace, loop, and turn metadata.
101
141
 
142
+ §provider-call-kind `callKind` is either `emission` (the response is a PLURNK
143
+ turn emission) or `bare` (the response is unconstrained answer text). The
144
+ consumer states this semantic fact explicitly; providers MUST NOT infer it from
145
+ message count, grammar presence, worker identity, or another incidental request
146
+ shape. The first-party adapter transports a supplied value as
147
+ `Plurnk-Call-Kind`; the metadata gate drops it for every third-party backend.
148
+ The signal is request metadata and never enters model-facing messages. Generic
149
+ provider callers MAY omit it; Core supplies it for every model call.
150
+
102
151
  A successful return carries the model's raw content and reasoning, normalized
103
- usage, normalized finish reason, model identity, opaque evidence, optional
104
- metadata, and optional notices. The provider transports and observes model
152
+ finish reason, model identity, its ordered {§provider-request-accounting}, the
153
+ request's `ProviderRequestCapacity`, opaque evidence, optional metadata, and
154
+ optional notices. A `ProviderError` carries the same available capacity and
155
+ accounting evidence. The provider transports and observes model
105
156
  output; it never retries, discards, or repairs an otherwise completed exchange
106
157
  because PLURNK grammar did not accept it.
107
158
 
108
- Usage obeys:
159
+ §provider-request-observer When a consumer supplies the request observer, the
160
+ provider opens one durable identity through it immediately before each physical
161
+ I/O and settles that identity with the resulting
162
+ `ProviderRequestAccounting`. This applies to every automatic retry and capacity
163
+ failover request. The observer is a durability sink, not an alternate evidence
164
+ representation; the same ordered records remain on the final response or error.
165
+
166
+ Usage obeys {§provider-usage}:
109
167
 
110
168
  ```text
111
- total = prompt + completion + reasoning
112
- cached ⊆ prompt
169
+ totalTokens = inputTokens + outputTokens
170
+ cacheReadTokens, cacheWriteTokens ⊆ inputTokens
171
+ reasoningTokens ⊆ outputTokens
113
172
  ```
114
173
 
115
- `completion` excludes reasoning. Ordinary vendor finish reasons normalize to
174
+ Unknown fields remain absent. Ordinary vendor finish reasons normalize to
116
175
  `stop`, `length`, `tool_calls`, or `content_filter`; an unknown value becomes
117
176
  `null` and emits a warning. `resource_interrupted` is the distinct failed-attempt
118
177
  disposition defined by {§provider-interrupted-attempt}.
@@ -132,9 +191,9 @@ explicit alias-scoped response style:
132
191
 
133
192
  Tag projection never runs when readable structured reasoning is already
134
193
  present. Streamed and buffered transports converge on this response boundary.
135
- When the upstream reports only combined output usage, the existing
136
- sum-preserving text-proportion estimate reclassifies completion versus reasoning
137
- without changing prompt, cached input, total output, or billed output.
194
+ When the upstream reports only combined output usage, that value remains
195
+ `outputTokens` and its unavailable text/reasoning detail stays absent. The
196
+ adapter never apportions tokens from character lengths.
138
197
 
139
198
  Grammar evidence retains the exact pre-projection sentence and its Unicode
140
199
  content offset. Response classification cannot rewrite what a transported GBNF
@@ -144,8 +203,10 @@ rail observed.
144
203
 
145
204
  §provider-sdk-boundary Cataloged providers instantiate their
146
205
  Models.dev-declared AI SDK package.
147
- Standard request shaping, streaming, retries, cancellation, timeouts, usage,
148
- and vendor error parsing belong to the SDK.
206
+ Standard request shaping, streaming, usage, and vendor error parsing belong to
207
+ the SDK. PLURNK supplies cancellation and deadline signals and owns the sole
208
+ cross-attempt scheduler so every physical request remains observable and
209
+ accountable.
149
210
 
150
211
  PLURNK maps its generic settings to AI SDK call settings:
151
212
 
@@ -153,19 +214,48 @@ PLURNK maps its generic settings to AI SDK call settings:
153
214
  - presence and frequency penalties;
154
215
  - stop sequences and seed;
155
216
  - output-token ceiling;
156
- - `off`, `adaptive`, or budget-derived reasoning intent.
217
+ - `off`, provider-default `adaptive`, or explicit `on` reasoning intent, with
218
+ an optional operator budget.
157
219
 
158
220
  Provider-specific options are permitted only where they preserve a documented
159
221
  PLURNK product contract the generic SDK surface cannot express.
160
222
 
223
+ §provider-readable-reasoning When the effective reasoning posture is not
224
+ `off`, a native adapter MUST request readable reasoning summaries if its
225
+ provider requires a separate response-visibility option. That option neither
226
+ activates reasoning nor selects its depth. The exact wire projection belongs to
227
+ the provider adapter; Models.dev's reasoning bit remains capability metadata.
228
+
229
+ The portable SDK surface has no boolean-enabled reasoning value. An unqualified
230
+ `on` therefore projects to its conventional `medium` enabled posture. This is a
231
+ wire activation value, not a reasoning budget or output-token ceiling.
232
+
233
+ §provider-cache-affinity **Cache affinity is route-owned request projection.**
234
+ When a provider documents a semantics-preserving conversation, session, or
235
+ prompt-cache routing key, its catalog adapter projects `workerId` through that
236
+ provider's documented header, body field, or native SDK option. The common
237
+ transport neither guesses from protocol resemblance nor sends a generic cache
238
+ field to an unknown provider. The operator may disable affinity globally or per
239
+ alias; automatic provider caching without an affinity control remains untouched.
240
+
241
+ §provider-cache-write-policy **Cache-write policy is separate from affinity.**
242
+ `PLURNK_PROVIDERS_CACHE_WRITE_POLICY` is `off` or `stable-system`. The latter
243
+ marks only the final leading system instruction as an explicit reusable cache
244
+ boundary, and only on routes whose native SDK documents that control. It does
245
+ not mark the changing user packet or enable an API-wide automatic cache mode.
246
+ Unsupported routes receive no invented option. The default five-minute
247
+ provider lifetime is used; a longer, differently priced lifetime is not an
248
+ implicit transport choice.
249
+
161
250
  §deepseek-reasoning-request The direct DeepSeek catalog path maps the common
162
251
  reasoning intent to its OpenAI-compatible controls:
163
252
 
164
- | PLURNK mode | `thinking` | `reasoning_effort` |
165
- | ----------- | --------------------- | -------------------- |
166
- | `off` | `{ type: disabled }` | omitted |
167
- | `adaptive` | omitted | omitted |
168
- | `on` | `{ type: enabled }` | budget-derived tier |
253
+ | PLURNK posture | `thinking` | `reasoning_effort` |
254
+ | --------------- | --------------------- | -------------------- |
255
+ | `off` | `{ type: disabled }` | omitted |
256
+ | `adaptive` | omitted | omitted |
257
+ | `on` | `{ type: enabled }` | omitted |
258
+ | `on` + budget | `{ type: enabled }` | budget-derived tier |
169
259
 
170
260
  The compatible transport is deliberately retained for:
171
261
 
@@ -195,9 +285,10 @@ The universal groups are:
195
285
  - reasoning activation and optional explicit budget;
196
286
  - explicit reasoning response-content style;
197
287
  - decode tuning;
198
- - request, stream-idle, retry, and probe budgets;
288
+ - operation, physical-attempt, first-content, stream-idle, retry, and probe budgets;
199
289
  - local GBNF and llama-server capability pins;
200
290
  - context-window and generation-envelope overrides;
291
+ - provider-documented cache affinity and explicit cache-write policy;
201
292
  - opt-in logprob and raw-body capture.
202
293
 
203
294
  Operator secrets and machine-specific values never belong in committed
@@ -214,16 +305,20 @@ alias.
214
305
 
215
306
  Provider and model facts resolve independently:
216
307
 
217
- | Fact | Natural source | Operator source | Effective value |
218
- | -------------------- | ------------------------------------------------------ | ---------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------ |
219
- | Context window | Catalog metadata or a local endpoint probe. | `PLURNK_PROVIDERS_CONTEXT_WINDOW`. | Minimum when both exist; the sole value otherwise. A cataloged cloud miss fails construction; a compatible probe miss remains `null` with one warning. |
220
- | Completion envelope | Catalog `maxOutput`; there is no live limit probe. | `PLURNK_PROVIDERS_COMPLETION_RESERVE`. | An absolute reserve wins. A percentage derives from the effective window and is capped by catalog `maxOutput` when present. |
221
- | Reasoning capability | Catalog `reasoning: true`, exposed by snapshot lookup. | Runtime activation, reserve, and adapter wire style. | The catalog bit is informational; provider construction neither activates nor blocks reasoning from it. |
222
- | Estimated USD rates | Models.dev input, output, and optional cache-read rates. | None. | Missing rates produce unknown evidence; explicit all-zero rates produce free evidence. No live price fetch exists. |
223
-
224
- Models.dev cache-read cost defaults to its input cost when omitted. Cache-write
225
- cost remains snapshot information; the current usage record has no cache-write
226
- quantity to price.
308
+ | Fact | Natural source | Operator source | Effective value |
309
+ | --- | --- | --- | --- |
310
+ | Context window | Catalog metadata or local endpoint probe. | `PLURNK_PROVIDERS_CONTEXT_WINDOW`. | Minimum when both exist; sole value otherwise. Cataloged cloud miss fails construction; compatible probe miss remains `null` with one warning. |
311
+ | Maximum input | Catalog `limit.input`; no generic live probe. | None. | Catalog value or `null`; never reconstructed from context and output. |
312
+ | Maximum output | Catalog `limit.output`; no generic live probe. | None. | Minimum of catalog value and effective context, or `null`. |
313
+ | Total output budget | None. | `PLURNK_PROVIDERS_OUTPUT_BUDGET`. | Percentage of effective context or absolute count, capped by known context/output limits; a call may only tighten it. |
314
+ | Reasoning budget | None. | Optional `PLURNK_PROVIDERS_REASONING_BUDGET`. | Percentage of effective context or absolute count; valid only as a strict subset of total output and effective only while reasoning is on or adaptive. |
315
+ | Reasoning capability | Catalog `reasoning: true`. | Runtime activation and adapter wire style. | Catalog bit remains informational; it neither activates nor blocks reasoning. |
316
+ | Estimated USD rates | Models.dev input, output, optional reasoning, and optional cache rates. | None. | Missing differently-priced usage or rates produces unknown; exact all-zero rates produces estimated USD zero. |
317
+
318
+ Models.dev cache-read and cache-write rates default to the input rate, and the
319
+ reasoning rate defaults to the output rate, when omitted. A differently priced
320
+ category requires its corresponding usage detail or the request estimate is
321
+ unknown.
227
322
 
228
323
  `instantiateProvider` resolves in this order:
229
324
 
@@ -298,7 +393,7 @@ also expose:
298
393
  - EOS marker removal;
299
394
  - exact complete-request counting through `/v1/chat/completions/input_tokens`;
300
395
  - exact content token IDs through `/tokenize`;
301
- - the requirement that the caller provide `maxTokens`.
396
+ - the requirement that the adapter apply a finite output budget.
302
397
 
303
398
  `PLURNK_PROVIDERS_LLAMA_SERVER` may force or disable detection. Probe attempts
304
399
  and delay are knobs. A failed probe does not silently assert capabilities.
@@ -311,14 +406,14 @@ OpenAI-compatible generation endpoint through the SDK adapter.
311
406
  For a detected llama-server, PLURNK sends the complete reasoning contract on
312
407
  every request:
313
408
 
314
- | Mode | Template activation | `thinking_budget_tokens` |
409
+ | Posture | Template activation | `thinking_budget_tokens` |
315
410
  |---|---:|---:|
316
411
  | `off` | false | `0` |
317
- | `adaptive` | true | resolved reasoning reserve |
318
- | `on` | true | explicit reasoning budget |
412
+ | `adaptive` | true | configured reasoning subset, otherwise omitted |
413
+ | `on` | true | configured reasoning subset, otherwise omitted |
319
414
 
320
- The explicit budget may tighten but MUST NOT exceed the resolved reasoning
321
- reserve. Template calls normally use `reasoning_format: "auto"` for a separate
415
+ The allowance is contained by the request's total output budget. Template calls
416
+ normally use `reasoning_format: "auto"` for a separate
322
417
  readable channel. A GBNF-bearing call uses `"none"` so the exact constrained
323
418
  sentence survives response projection; the adapter separates its leading
324
419
  reasoning enclosure only after preserving grammar evidence. Process-wide
@@ -342,7 +437,7 @@ intent. It cannot override:
342
437
  - data-capture settings;
343
438
  - tool, modality, or multi-choice behavior;
344
439
  - the consumer-owned output envelope;
345
- - prompt-cache identity.
440
+ - cache affinity identity or cache-write policy.
346
441
 
347
442
  Generic AI SDK calls accept only settings represented by the SDK's portable
348
443
  surface. Compatible endpoints may carry additional sampling keys after reserved
@@ -366,16 +461,40 @@ normal value. Retry exhaustion is preserved as `attempts` and
366
461
  `retryExhausted`, and the resulting Problem is not marked retryable after the
367
462
  provider has consumed its automatic retry budget.
368
463
 
369
- The AI SDK owns one attempt scheduler around the complete generation exchange.
370
- `PLURNK_PROVIDERS_RETRY_ATTEMPTS` is the maximum retry count, including a
371
- configured stream-chunk deadline that expires while consuming a response body.
372
- The total generation deadline and caller cancellation span that scheduler and
373
- every attempt; neither restarts during replay. Total and stream-chunk deadlines
374
- are separately configurable.
464
+ §provider-capacity-failure A proven exact preflight overflow and an upstream
465
+ context rejection normalize to `ProviderError(kind="capacity_exceeded")` and
466
+ an RFC 9457 status 413. `capacityStage` is `preflight` or `upstream`; a
467
+ non-413 upstream status remains `providerStatus`, while physical request
468
+ accounting retains the status actually received. Preflight rejection occurs
469
+ before provider I/O and therefore opens no request identity and creates no
470
+ request-accounting row. Capacity failures are not connectivity failures and are
471
+ never retried by the provider scheduler; bounded packet recovery belongs to the
472
+ consumer.
473
+
474
+ §provider-connectivity The provider adapter owns one attempt scheduler around
475
+ the complete generation exchange; SDK-internal retries are disabled.
476
+ `PLURNK_PROVIDERS_RETRY_ATTEMPTS=N` permits at most `N + 1` physical requests.
477
+ The layers are independent and a configured value of zero disables only that
478
+ deadline:
479
+
480
+ | Layer | Operator knob | Boundary | Expiry |
481
+ | --- | --- | --- | --- |
482
+ | Operation | `PLURNK_PROVIDERS_OPERATION_TIMEOUT` | Complete logical call, including every attempt and retry delay. | Final `deadline_exceeded` Problem at 504 with `timeoutPhase=operation`; never retried. |
483
+ | Attempt | `PLURNK_PROVIDERS_FETCH_TIMEOUT` | One physical generation request, including response consumption. | Retryable `network_failure` with `timeoutPhase=attempt`. |
484
+ | First content | `PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT` | Response-stream start through first semantic model content; metadata, empty deltas, and transport activity do not satisfy it. | Retryable `network_failure` with `timeoutPhase=first_content`. |
485
+ | Stream idle | `PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT` | Silence between semantic content chunks after content begins. | Retryable `network_failure` with `timeoutPhase=stream_idle`. |
486
+
487
+ Caller cancellation spans the operation and preserves the caller's reason.
488
+ Inner deadline failures consume the ordinary retry budget; retry exhaustion
489
+ adds `attempts` and `retryExhausted`, retains the inner `timeoutPhase` and
490
+ `timeoutMs`, and is final. Every scheduler iteration opens and settles exactly
491
+ one ordered {§provider-request-accounting} record, including response-less
492
+ network failures and timed-out attempts.
375
493
 
376
494
  HTTP 408, 409, 429, and ordinary 5xx responses are retryable unless the endpoint
377
- explicitly says otherwise. Endpoint control responses 520–527 are final so a
378
- router can prevent multiplicative retries behind its own retry policy.
495
+ explicitly says otherwise through `X-Should-Retry`. That header is authoritative;
496
+ without an explicit directive, endpoint control responses 520–527 are final so
497
+ a router can prevent multiplicative retries behind its own policy.
379
498
 
380
499
  ### §provider-interrupted-attempt Provider-declared interruption
381
500
 
@@ -386,10 +505,10 @@ completed exchange.
386
505
  | Concern | Contract |
387
506
  | -------------------------- | ---------------------------------------------------------------------------------------------------------------------------- |
388
507
  | Normalized finish reason | Exact `insufficient_system_resource` becomes `resource_interrupted`; the raw value remains in `assistantRaw`. |
389
- | `generate` outcome | Throw `ProviderError(kind="resource_interrupted")` at local status 503 with the normalized attempt on `error.attempt`. |
508
+ | `generate` outcome | Throw `ProviderError(kind="resource_interrupted")` at local status 503 with the normalized attempt on `error.attempt` and the same accounting on `error.accounting`. |
390
509
  | Partial response | Preserve content, reasoning, usage, model, metadata, optional raw body, and other evidence without admitting it as success. |
391
510
  | Automatic replay | None; the Problem has `retryable: false`, and AI SDK retry scheduling has already completed at the successful transport. |
392
- | Capacity-pool overflow | None; a sibling success would erase the known billed failed attempt from the current `ProviderResponse` surface. |
511
+ | Capacity-pool overflow | None under the existing routing policy; when other overflow-eligible failures do reach a sibling, the pool concatenates their request accounting. |
393
512
  | Consumer admission | Persist the evidence as an unaccepted attempt; never parse it into executable work, even when its frame looks complete. |
394
513
 
395
514
  ## §10 Grammar
@@ -398,8 +517,9 @@ GBNF is a local llama-server capability, not a generic provider expectation.
398
517
  The consumer chooses whether to supply a grammar. The provider never creates or
399
518
  rewrites one.
400
519
 
401
- GBNF defines the accepted raw sampled text; it runs before response reasoning
402
- is separated from regular content. The shipped PLURNK sentence is owned by
520
+ GBNF defines the accepted sampled text; it runs before response reasoning is
521
+ separated from regular content. A generated rail may declare a response root
522
+ that composes a template-provided prefix for independent evidence grading. The shipped PLURNK sentence is owned by
403
523
  `plurnk-contracts` {§gbnf-turn-shape} and {§gbnf-reasoning-boundary}; this package does
404
524
  not restate or rewrite it.
405
525
 
@@ -418,7 +538,6 @@ The consumer validates `grammarEvidence.input` outside the enforcer's failure
418
538
  domain. `PLURNK_PROVIDERS_GBNF_DEBUG` still validates grammar syntax before the
419
539
  call and sets `transported: false` for the unconstrained comparison.
420
540
 
421
-
422
541
  ## §11 Evidence and metadata
423
542
 
424
543
  §provider-evidence `assistantRaw` is an opaque normalized transport record.
@@ -449,11 +568,35 @@ correlate them to an entity it actually created rather than reusing `id`.
449
568
 
450
569
  ## §12 Generation envelopes
451
570
 
452
- §provider-generation-envelope Reasoning and completion reserves are percentages
453
- of the resolved context
454
- window or absolute token counts. Absolute pins win. The provider reports the
455
- resolved reserves; the consumer owns prompt packing and the per-call output
456
- cap. Model-limit precedence is defined once in {§model-fact-resolution}.
571
+ §provider-generation-envelope Every request has at most one total output
572
+ budget. It includes visible output and hidden reasoning. An optional reasoning
573
+ budget is a strict subset of that total, never an additive reserve. The
574
+ configured total is a percentage of effective context or an absolute count;
575
+ percentages resolve to the nearest whole token with a one-token minimum. It is
576
+ capped by known context and model-output limits; `generate.maxOutputTokens` may
577
+ only tighten it for one call. The effective reasoning subset tightens with that
578
+ total and remains strictly smaller.
579
+
580
+ The adapter owns native projection. A backend whose generic SDK maximum already
581
+ includes reasoning receives the total directly. When a native SDK instead adds
582
+ an explicit reasoning allowance to its generic visible-output maximum, the
583
+ adapter sends `total - reasoning` through the generic field and the reasoning
584
+ subset through the documented provider option. Core and other callers never
585
+ reconstruct this arithmetic.
586
+
587
+ §provider-output-budget-conformance When a completed response reports
588
+ normalized output-token usage greater than its effective total output budget,
589
+ the exchange is an `invalid_response` at 502 rather than an admitted result or
590
+ a prompt-capacity 413. Its complete failed-attempt evidence and settled charged
591
+ request remain available. The violation is final and is never automatically
592
+ replayed. Missing output usage cannot prove a violation.
593
+
594
+ `PLURNK_PROVIDERS_OUTPUT_BUDGET` is required for standard providers and ships
595
+ as `35%`. `PLURNK_PROVIDERS_REASONING_BUDGET` is optional; leaving it unset
596
+ preserves provider-adaptive depth. A backend known to decode without a finite
597
+ limit advertises `requiresOutputBudget` and fails construction when no total can
598
+ be resolved. The retired additive reserve knobs fail hard rather than creating
599
+ a second envelope contract.
457
600
 
458
601
  ## §13 Capacity pool
459
602
 
@@ -466,8 +609,13 @@ availability and rate-limit failures that carry no normalized response attempt;
466
609
  {§provider-interrupted-attempt} propagates without overflow.
467
610
 
468
611
  Prompt measurement covers every backend that could receive the request. The
469
- pool takes the largest result; differing exact counts or any proven bound yield
470
- an `upper_bound`, while any estimate makes the aggregate an estimate.
612
+ pool takes the largest quantified result; differing exact counts or any proven
613
+ bound yield an `upper_bound`, any estimate makes the aggregate an estimate, and
614
+ any unavailable backend makes it unavailable. Physical limits and budgets are
615
+ independent safe minima across the pool. `inputCapacity` is the minimum of each
616
+ backend's complete derived input capacity, never a synthetic subtraction across
617
+ minima from different backends; request-specific output tightening repeats the
618
+ complete-envelope derivation per backend before taking the minimum.
471
619
 
472
620
  ## §14 Conformance
473
621
 
@@ -479,7 +627,12 @@ Coverage MUST prove:
479
627
  - compatible extension preservation;
480
628
  - timeout, retry, cancellation, interrupted-attempt, and final-error behavior;
481
629
  - local capability probes and pins;
482
- - exact, bounded, and estimated complete-request token measurements;
630
+ - exact, bounded, estimated, and unavailable complete-request measurements;
631
+ - independent input/context/output limits, asymmetric admission, and normalized
632
+ local/upstream capacity failures;
633
+ - one total output budget and native additive-reasoning projection;
634
+ - provider-reported output beyond that budget failing once with complete
635
+ attempt and accounting evidence;
483
636
  - local reasoning activation, response-wide allowance, and GBNF coexistence;
484
637
  - explicit tagged-reasoning projection across streamed, buffered, capped, and
485
638
  literal-tag responses;
@@ -1,32 +1,48 @@
1
- import type { AuthoritativeChargeNormalizer, ChatMessage, PromptTokenMeasurement, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
1
+ import type { ChatMessage, PromptTokenMeasurement, Provider, ProviderCostNormalizer, ProviderGenerateArgs, ProviderRequestCapacity, ProviderResponse, ProviderUsage } from "./types.ts";
2
2
  import type { ProviderCost } from "@plurnk/plurnk-contracts";
3
- import type { Reasoning, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
3
+ import type { JSONValue } from "ai";
4
+ import type { Reasoning, ReasoningResponseStyle } from "./env.ts";
4
5
  import type { LanguageModel } from "ai";
5
6
  import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
6
7
  export type ProviderFetch = typeof globalThis.fetch;
7
8
  export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "thinking_effort" | "template" | "anthropic";
8
9
  export type GrammarStyle = "none" | "llamacpp";
10
+ export type CacheAffinity = {
11
+ readonly target: "header" | "body";
12
+ readonly name: string;
13
+ } | {
14
+ readonly target: "provider-option";
15
+ readonly provider: string;
16
+ readonly name: string;
17
+ };
18
+ export type AiSdkProviderOptions = Record<string, Record<string, JSONValue | undefined>>;
9
19
  export type AiSdkProviderConfig = {
10
20
  model: string;
11
21
  url?: string;
12
22
  languageModel?: LanguageModel;
13
23
  attributions?: (context: PluginAttributionContext) => PluginAttribution;
14
24
  fetchTimeoutMs: number;
25
+ operationTimeoutMs: number;
26
+ firstContentTimeoutMs: number;
15
27
  streamIdleTimeoutMs?: number;
16
28
  headers?: Record<string, string>;
17
29
  fetch?: ProviderFetch;
18
30
  contextWindow?: number | null;
31
+ maxInputTokens?: number | null;
32
+ maxOutputTokens?: number | null;
33
+ outputBudget?: number | null;
34
+ reasoningBudget?: number | null;
35
+ additiveReasoningProvider?: "anthropic" | "bedrock";
19
36
  reasoningStyle?: ReasoningStyle;
20
37
  reasoningResponseStyle?: ReasoningResponseStyle;
21
38
  countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
22
- calculateCost?: (usage: ProviderUsage) => number;
23
- calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, {
24
- kind: "authoritative";
25
- }>;
26
- normalizeCharge?: AuthoritativeChargeNormalizer;
39
+ estimateCost?: (usage: ProviderUsage | undefined) => ProviderCost;
40
+ normalizeCost?: ProviderCostNormalizer;
27
41
  source?: string;
28
42
  grammarStyle?: GrammarStyle;
29
- promptCacheKey?: boolean;
43
+ cacheAffinity?: CacheAffinity;
44
+ systemCacheProviderOptions?: AiSdkProviderOptions;
45
+ reasoningResponseProviderOptions?: AiSdkProviderOptions;
30
46
  serviceTier?: string;
31
47
  gbnfDebug?: boolean;
32
48
  streaming?: boolean;
@@ -38,7 +54,7 @@ export type AiSdkProviderConfig = {
38
54
  tokenizeUrl?: string;
39
55
  promptTokensUrl?: string;
40
56
  servedModel?: string;
41
- requiresMaxTokens?: boolean;
57
+ requiresOutputBudget?: boolean;
42
58
  reasoning: Reasoning;
43
59
  temperature: number;
44
60
  repeatPenalty: number;
@@ -51,8 +67,6 @@ export type AiSdkProviderConfig = {
51
67
  errorDetailLimit?: number;
52
68
  topLogprobs?: number | null;
53
69
  rawBody?: boolean;
54
- reasoningReserve?: ReserveSpec;
55
- completionReserve?: ReserveSpec;
56
70
  tuningFloors?: boolean;
57
71
  };
58
72
  export declare const effortFromBudget: (budget: number) => "low" | "medium" | "high";
@@ -62,31 +76,17 @@ export default class AiSdkProvider implements Provider {
62
76
  tokenize?: (text: string) => Promise<number[]>;
63
77
  constructor(config: AiSdkProviderConfig);
64
78
  get contextWindow(): number | null;
65
- get reasoningReserve(): number | null;
66
- get completionReserve(): number | null;
79
+ get maxInputTokens(): number | null;
80
+ get maxOutputTokens(): number | null;
81
+ get outputBudget(): number | null;
82
+ get reasoningBudget(): number | null;
83
+ get inputCapacity(): number | null;
67
84
  get model(): string;
68
85
  get servedModel(): string | undefined;
69
- get requiresMaxTokens(): boolean | undefined;
86
+ get requiresOutputBudget(): boolean | undefined;
70
87
  get constrainsOutput(): boolean;
71
88
  countPromptTokens(messages: readonly ChatMessage[], signal?: AbortSignal): Promise<PromptTokenMeasurement>;
72
- calculateCost(usage: ProviderUsage): number;
73
- calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, {
74
- kind: "authoritative";
75
- }>;
76
- generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }: {
77
- messages: ChatMessage[];
78
- workerId: string;
79
- primaryWorkerId?: string;
80
- signal?: AbortSignal;
81
- grammar?: string;
82
- maxTokens?: number;
83
- attributions?: string[];
84
- client?: string;
85
- strikes?: number;
86
- workspaceId?: string;
87
- loop?: number;
88
- turn?: number;
89
- sampling?: Record<string, unknown>;
90
- }): Promise<ProviderResponse>;
89
+ assessRequestCapacity(messages: readonly ChatMessage[], maxOutputTokens?: number, signal?: AbortSignal): Promise<ProviderRequestCapacity>;
90
+ generate({ messages, workerId, primaryWorkerId, signal, grammar, maxOutputTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }: ProviderGenerateArgs): Promise<ProviderResponse>;
91
91
  }
92
92
  //# sourceMappingURL=AiSdkProvider.d.ts.map