@plurnk/plurnk-providers 1.5.0 → 1.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +41 -34
- package/README.md +15 -0
- package/SPEC.md +242 -89
- package/dist/AiSdkProvider.d.ts +33 -33
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +442 -133
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +10 -11
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +87 -25
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +9 -24
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +86 -25
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts +5 -2
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +100 -16
- package/dist/accounting.js.map +1 -1
- package/dist/accountingPublic.d.ts +5 -0
- package/dist/accountingPublic.d.ts.map +1 -0
- package/dist/accountingPublic.js +3 -0
- package/dist/accountingPublic.js.map +1 -0
- package/dist/aiSdkTransport.d.ts +9 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +160 -62
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/capacity.d.ts +26 -0
- package/dist/capacity.d.ts.map +1 -0
- package/dist/capacity.js +90 -0
- package/dist/capacity.js.map +1 -0
- package/dist/catalogProvider.d.ts +8 -3
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +45 -41
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +26 -12
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +10 -10
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +90 -42
- package/dist/cost.js.map +1 -1
- package/dist/env.d.ts +13 -11
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +83 -46
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +17 -3
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +91 -8
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +7 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5 -3
- package/dist/index.js.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/promptTokens.d.ts.map +1 -1
- package/dist/promptTokens.js +7 -4
- package/dist/promptTokens.js.map +1 -1
- package/dist/sdkModels.d.ts +7 -2
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +43 -13
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +55 -33
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +22 -5
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +169 -83
- package/dist/usage.js.map +1 -1
- package/package.json +18 -7
- package/src/AiSdkProvider.test.ts +964 -206
- package/src/AiSdkProvider.ts +545 -155
- package/src/Mock.test.ts +69 -30
- package/src/Mock.ts +99 -29
- package/src/Pool.test.ts +90 -19
- package/src/Pool.ts +96 -27
- package/src/ProviderRegistry.test.ts +16 -11
- package/src/accounting.test.ts +58 -22
- package/src/accounting.ts +119 -18
- package/src/accountingPublic.ts +9 -0
- package/src/aiSdkTransport.test.ts +42 -49
- package/src/aiSdkTransport.ts +174 -62
- package/src/boundaries.test.ts +2 -0
- package/src/capacity.test.ts +92 -0
- package/src/capacity.ts +140 -0
- package/src/catalogProvider.test.ts +339 -30
- package/src/catalogProvider.ts +65 -47
- package/src/compatibleProvider.test.ts +7 -5
- package/src/compatibleProvider.ts +29 -13
- package/src/cost.test.ts +86 -36
- package/src/cost.ts +111 -50
- package/src/defaults.test.ts +13 -3
- package/src/env.test.ts +103 -25
- package/src/env.ts +153 -65
- package/src/errors.test.ts +80 -2
- package/src/errors.ts +107 -8
- package/src/index.ts +26 -7
- package/src/ollama.test.ts +5 -3
- package/src/ollama.ts +3 -3
- package/src/promptTokens.ts +8 -5
- package/src/sdkModels.test.ts +77 -8
- package/src/sdkModels.ts +51 -15
- package/src/types.ts +112 -51
- package/src/usage.test.ts +112 -116
- package/src/usage.ts +214 -93
package/SPEC.md
CHANGED
|
@@ -9,8 +9,8 @@ ordinary provider protocols.
|
|
|
9
9
|
The provider stack has four owners:
|
|
10
10
|
|
|
11
11
|
1. Models.dev supplies a release-time snapshot of provider package, API
|
|
12
|
-
endpoint, credential names, models, context
|
|
13
|
-
|
|
12
|
+
endpoint, credential names, models, context/input/output limits, reasoning
|
|
13
|
+
capability, and USD rates including distinct reasoning rates when supplied.
|
|
14
14
|
2. Official AI SDK providers own vendor request and response protocols.
|
|
15
15
|
3. This package owns the PLURNK contract: aliases, envelopes, normalized usage
|
|
16
16
|
and errors, evidence, local capabilities, and first-party metadata.
|
|
@@ -22,6 +22,14 @@ model prefix, context window, price, or vendor request shape into a PLURNK
|
|
|
22
22
|
table. A missing or wrong catalog fact is fixed upstream, overridden through a
|
|
23
23
|
provider declaration, or left explicitly unknown.
|
|
24
24
|
|
|
25
|
+
§provider-runtime-neutral-accounting The public
|
|
26
|
+
`@plurnk/plurnk-providers/accounting` subpath exposes provider-request
|
|
27
|
+
aggregation and Models.dev cost estimation, plus their wire types, without
|
|
28
|
+
evaluating Node-only provider discovery, filesystem defaults, or runtime
|
|
29
|
+
construction. It re-exports the package's sole accounting implementation. The
|
|
30
|
+
package root remains the Node provider-runtime composition surface and is not a
|
|
31
|
+
Worker entrypoint.
|
|
32
|
+
|
|
25
33
|
## §2 Provider interface
|
|
26
34
|
|
|
27
35
|
§provider-interface `Provider` exposes immutable model facts and one generation
|
|
@@ -31,88 +39,139 @@ operation:
|
|
|
31
39
|
interface Provider {
|
|
32
40
|
readonly model: string;
|
|
33
41
|
readonly contextWindow: number | null;
|
|
42
|
+
readonly maxInputTokens: number | null;
|
|
43
|
+
readonly maxOutputTokens: number | null;
|
|
44
|
+
readonly outputBudget: number | null;
|
|
45
|
+
readonly reasoningBudget: number | null;
|
|
46
|
+
readonly inputCapacity: number | null;
|
|
34
47
|
readonly servedModel?: string;
|
|
35
48
|
readonly constrainsOutput?: boolean;
|
|
36
|
-
readonly
|
|
37
|
-
readonly reasoningReserve?: number | null;
|
|
38
|
-
readonly completionReserve?: number | null;
|
|
49
|
+
readonly requiresOutputBudget?: boolean;
|
|
39
50
|
|
|
40
51
|
countPromptTokens(
|
|
41
52
|
messages: readonly ChatMessage[],
|
|
42
53
|
signal?: AbortSignal,
|
|
43
54
|
): Promise<PromptTokenMeasurement>;
|
|
55
|
+
assessRequestCapacity(
|
|
56
|
+
messages: readonly ChatMessage[],
|
|
57
|
+
maxOutputTokens?: number,
|
|
58
|
+
signal?: AbortSignal,
|
|
59
|
+
): Promise<ProviderRequestCapacity>;
|
|
44
60
|
tokenize?(text: string): Promise<number[]>;
|
|
45
|
-
calculateCost(usage: ProviderUsage): number;
|
|
46
|
-
calculateCharge?(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }>;
|
|
47
61
|
generate(args: GenerateArgs): Promise<ProviderResponse>;
|
|
48
62
|
}
|
|
49
63
|
```
|
|
50
64
|
|
|
51
|
-
`contextWindow` is
|
|
52
|
-
{§model-fact-resolution}
|
|
53
|
-
|
|
54
|
-
|
|
65
|
+
`contextWindow` is the effective total context envelope resolved under
|
|
66
|
+
{§model-fact-resolution}: the minimum of known model capacity and any stricter
|
|
67
|
+
operator cap. `null` means genuinely unknown; a consumer MUST NOT invent a
|
|
68
|
+
stand-in. The context-window knob is a hard cap, never model-facing grinder
|
|
69
|
+
pressure.
|
|
55
70
|
|
|
56
|
-
`PromptTokenMeasurement` is a discriminated
|
|
71
|
+
§provider-prompt-measurement `PromptTokenMeasurement` is a discriminated
|
|
72
|
+
request-level result:
|
|
57
73
|
|
|
58
|
-
| `kind`
|
|
59
|
-
|
|
|
60
|
-
| `exact`
|
|
61
|
-
| `upper_bound` | Proven upper bound for the complete provider request.
|
|
62
|
-
| `estimate`
|
|
74
|
+
| `kind` | Meaning | Capacity authority |
|
|
75
|
+
| --- | --- | --- |
|
|
76
|
+
| `exact` | Exact count for the complete provider request. | May prove fit or overflow. |
|
|
77
|
+
| `upper_bound` | Proven upper bound for the complete provider request. | May prove fit; exceeding a limit does not prove overflow. |
|
|
78
|
+
| `estimate` | Empirical prediction with required causal `detail`. | Cannot admit or reject. |
|
|
79
|
+
| `unavailable` | No quantified measurement, with required causal `detail`. | Cannot admit or reject. |
|
|
63
80
|
|
|
64
|
-
Every result carries a non-
|
|
81
|
+
Every result carries a non-empty `source`; quantified kinds carry non-negative
|
|
82
|
+
integer `tokens`.
|
|
65
83
|
`countPromptTokens` receives the same messages supplied to `generate` and may
|
|
66
84
|
perform cancellable provider I/O. The common fallback is chars/2 over message
|
|
67
85
|
content; it is announced once and reported honestly as an estimate because it
|
|
68
86
|
knows neither the serving vocabulary nor provider-owned request framing.
|
|
69
|
-
An
|
|
70
|
-
|
|
71
|
-
|
|
87
|
+
An adapter may retain that estimate when an optional counting endpoint fails,
|
|
88
|
+
provided its detail names the cause; one unable to quantify anything returns
|
|
89
|
+
`unavailable`. A malformed measurement is a provider contract violation and
|
|
90
|
+
fails hard.
|
|
91
|
+
|
|
92
|
+
§provider-capacity-admission `assessRequestCapacity` intersects every known
|
|
93
|
+
physical input constraint: independent `maxInputTokens` and
|
|
94
|
+
`contextWindow - outputBudget`. Its result is `admit`, `reject`, or `defer` and
|
|
95
|
+
retains the complete limit and measurement evidence. Exact fit admits; exact
|
|
96
|
+
overflow rejects. A proven upper bound admits only when it fits. Unknown limits,
|
|
97
|
+
an upper bound above a limit, estimates, and unavailable measurements defer to
|
|
98
|
+
the upstream provider as capacity oracle. The same stable intersection is
|
|
99
|
+
exposed as `inputCapacity`; `null` means the available limits cannot establish
|
|
100
|
+
one. A known combined context and output budget must leave positive input
|
|
101
|
+
capacity. Consumers may display or use that fact as policy, but MUST NOT
|
|
102
|
+
substitute their own content heuristic for request-shaped admission.
|
|
72
103
|
|
|
73
104
|
`tokenize` is the separate content-token capability and exists only when the
|
|
74
105
|
endpoint exposes its real vocabulary. Content tokenization does not substitute
|
|
75
106
|
for complete-request measurement.
|
|
76
107
|
|
|
77
|
-
§provider-monetary-evidence One precedence path converts each provider
|
|
78
|
-
into {§provider-cost} evidence
|
|
108
|
+
§provider-monetary-evidence One precedence path converts each physical provider
|
|
109
|
+
request into {§provider-cost} evidence before the request leaves the provider
|
|
110
|
+
boundary:
|
|
79
111
|
|
|
80
112
|
| Precedence | Evidence | Result |
|
|
81
113
|
| --- | --- | --- |
|
|
82
|
-
| 1 | A documented monetary field on that response | The adapter
|
|
83
|
-
| 2 |
|
|
84
|
-
| 3 | Neither |
|
|
114
|
+
| 1 | A documented monetary field on that response or error | The adapter validates and preserves its documented `charged` or `estimated` character; its exact amount wins. |
|
|
115
|
+
| 2 | Known response usage and the exact model's Models.dev rates | The adapter returns an exact decimal USD `estimated` amount only when every differently-priced applicable category is known. |
|
|
116
|
+
| 3 | Neither | `unknown` with a concrete reason. |
|
|
85
117
|
|
|
86
|
-
Models.dev is the sole supported fallback rate table. Missing usage
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
118
|
+
Models.dev is the sole supported fallback rate table. Missing usage, a missing
|
|
119
|
+
applicable category, or missing rates never proves zero. An exact zero rate
|
|
120
|
+
produces an ordinary estimated amount of USD `0`. Rate calculation is internal
|
|
121
|
+
to the provider request; Core, digest, ping, and clients never call a parallel
|
|
122
|
+
pricing method.
|
|
91
123
|
|
|
92
124
|
### Generation
|
|
93
125
|
|
|
94
|
-
`generate` requires a non-empty, stable, opaque
|
|
126
|
+
§provider-cache-identity `generate` requires a non-empty, stable, opaque
|
|
127
|
+
`workerId`. A durable worker uses one globally unique value for its lifetime;
|
|
128
|
+
independent databases and processes cannot mint the same local sequence. A
|
|
129
|
+
`bare` call instead uses a fresh per-call value, preventing unrelated prompts
|
|
130
|
+
from acquiring affinity with either the parent worker or another BARE call.
|
|
131
|
+
Providers MUST NOT interpret either value.
|
|
132
|
+
|
|
133
|
+
`generate` accepts:
|
|
95
134
|
|
|
96
135
|
- `messages`: system, user, and assistant text messages;
|
|
97
136
|
- caller cancellation through `signal`;
|
|
98
|
-
- optional `grammar` and `
|
|
137
|
+
- optional `grammar` and call-specific `maxOutputTokens` tightening;
|
|
99
138
|
- standard `sampling` intent;
|
|
139
|
+
- the caller-owned `callKind` output contract when one applies;
|
|
100
140
|
- opaque attribution tags plus client, strike, workspace, loop, and turn metadata.
|
|
101
141
|
|
|
142
|
+
§provider-call-kind `callKind` is either `emission` (the response is a PLURNK
|
|
143
|
+
turn emission) or `bare` (the response is unconstrained answer text). The
|
|
144
|
+
consumer states this semantic fact explicitly; providers MUST NOT infer it from
|
|
145
|
+
message count, grammar presence, worker identity, or another incidental request
|
|
146
|
+
shape. The first-party adapter transports a supplied value as
|
|
147
|
+
`Plurnk-Call-Kind`; the metadata gate drops it for every third-party backend.
|
|
148
|
+
The signal is request metadata and never enters model-facing messages. Generic
|
|
149
|
+
provider callers MAY omit it; Core supplies it for every model call.
|
|
150
|
+
|
|
102
151
|
A successful return carries the model's raw content and reasoning, normalized
|
|
103
|
-
|
|
104
|
-
|
|
152
|
+
finish reason, model identity, its ordered {§provider-request-accounting}, the
|
|
153
|
+
request's `ProviderRequestCapacity`, opaque evidence, optional metadata, and
|
|
154
|
+
optional notices. A `ProviderError` carries the same available capacity and
|
|
155
|
+
accounting evidence. The provider transports and observes model
|
|
105
156
|
output; it never retries, discards, or repairs an otherwise completed exchange
|
|
106
157
|
because PLURNK grammar did not accept it.
|
|
107
158
|
|
|
108
|
-
|
|
159
|
+
§provider-request-observer When a consumer supplies the request observer, the
|
|
160
|
+
provider opens one durable identity through it immediately before each physical
|
|
161
|
+
I/O and settles that identity with the resulting
|
|
162
|
+
`ProviderRequestAccounting`. This applies to every automatic retry and capacity
|
|
163
|
+
failover request. The observer is a durability sink, not an alternate evidence
|
|
164
|
+
representation; the same ordered records remain on the final response or error.
|
|
165
|
+
|
|
166
|
+
Usage obeys {§provider-usage}:
|
|
109
167
|
|
|
110
168
|
```text
|
|
111
|
-
|
|
112
|
-
|
|
169
|
+
totalTokens = inputTokens + outputTokens
|
|
170
|
+
cacheReadTokens, cacheWriteTokens ⊆ inputTokens
|
|
171
|
+
reasoningTokens ⊆ outputTokens
|
|
113
172
|
```
|
|
114
173
|
|
|
115
|
-
|
|
174
|
+
Unknown fields remain absent. Ordinary vendor finish reasons normalize to
|
|
116
175
|
`stop`, `length`, `tool_calls`, or `content_filter`; an unknown value becomes
|
|
117
176
|
`null` and emits a warning. `resource_interrupted` is the distinct failed-attempt
|
|
118
177
|
disposition defined by {§provider-interrupted-attempt}.
|
|
@@ -132,9 +191,9 @@ explicit alias-scoped response style:
|
|
|
132
191
|
|
|
133
192
|
Tag projection never runs when readable structured reasoning is already
|
|
134
193
|
present. Streamed and buffered transports converge on this response boundary.
|
|
135
|
-
When the upstream reports only combined output usage,
|
|
136
|
-
|
|
137
|
-
|
|
194
|
+
When the upstream reports only combined output usage, that value remains
|
|
195
|
+
`outputTokens` and its unavailable text/reasoning detail stays absent. The
|
|
196
|
+
adapter never apportions tokens from character lengths.
|
|
138
197
|
|
|
139
198
|
Grammar evidence retains the exact pre-projection sentence and its Unicode
|
|
140
199
|
content offset. Response classification cannot rewrite what a transported GBNF
|
|
@@ -144,8 +203,10 @@ rail observed.
|
|
|
144
203
|
|
|
145
204
|
§provider-sdk-boundary Cataloged providers instantiate their
|
|
146
205
|
Models.dev-declared AI SDK package.
|
|
147
|
-
Standard request shaping, streaming,
|
|
148
|
-
and
|
|
206
|
+
Standard request shaping, streaming, usage, and vendor error parsing belong to
|
|
207
|
+
the SDK. PLURNK supplies cancellation and deadline signals and owns the sole
|
|
208
|
+
cross-attempt scheduler so every physical request remains observable and
|
|
209
|
+
accountable.
|
|
149
210
|
|
|
150
211
|
PLURNK maps its generic settings to AI SDK call settings:
|
|
151
212
|
|
|
@@ -153,19 +214,48 @@ PLURNK maps its generic settings to AI SDK call settings:
|
|
|
153
214
|
- presence and frequency penalties;
|
|
154
215
|
- stop sequences and seed;
|
|
155
216
|
- output-token ceiling;
|
|
156
|
-
- `off`, `adaptive`, or
|
|
217
|
+
- `off`, provider-default `adaptive`, or explicit `on` reasoning intent, with
|
|
218
|
+
an optional operator budget.
|
|
157
219
|
|
|
158
220
|
Provider-specific options are permitted only where they preserve a documented
|
|
159
221
|
PLURNK product contract the generic SDK surface cannot express.
|
|
160
222
|
|
|
223
|
+
§provider-readable-reasoning When the effective reasoning posture is not
|
|
224
|
+
`off`, a native adapter MUST request readable reasoning summaries if its
|
|
225
|
+
provider requires a separate response-visibility option. That option neither
|
|
226
|
+
activates reasoning nor selects its depth. The exact wire projection belongs to
|
|
227
|
+
the provider adapter; Models.dev's reasoning bit remains capability metadata.
|
|
228
|
+
|
|
229
|
+
The portable SDK surface has no boolean-enabled reasoning value. An unqualified
|
|
230
|
+
`on` therefore projects to its conventional `medium` enabled posture. This is a
|
|
231
|
+
wire activation value, not a reasoning budget or output-token ceiling.
|
|
232
|
+
|
|
233
|
+
§provider-cache-affinity **Cache affinity is route-owned request projection.**
|
|
234
|
+
When a provider documents a semantics-preserving conversation, session, or
|
|
235
|
+
prompt-cache routing key, its catalog adapter projects `workerId` through that
|
|
236
|
+
provider's documented header, body field, or native SDK option. The common
|
|
237
|
+
transport neither guesses from protocol resemblance nor sends a generic cache
|
|
238
|
+
field to an unknown provider. The operator may disable affinity globally or per
|
|
239
|
+
alias; automatic provider caching without an affinity control remains untouched.
|
|
240
|
+
|
|
241
|
+
§provider-cache-write-policy **Cache-write policy is separate from affinity.**
|
|
242
|
+
`PLURNK_PROVIDERS_CACHE_WRITE_POLICY` is `off` or `stable-system`. The latter
|
|
243
|
+
marks only the final leading system instruction as an explicit reusable cache
|
|
244
|
+
boundary, and only on routes whose native SDK documents that control. It does
|
|
245
|
+
not mark the changing user packet or enable an API-wide automatic cache mode.
|
|
246
|
+
Unsupported routes receive no invented option. The default five-minute
|
|
247
|
+
provider lifetime is used; a longer, differently priced lifetime is not an
|
|
248
|
+
implicit transport choice.
|
|
249
|
+
|
|
161
250
|
§deepseek-reasoning-request The direct DeepSeek catalog path maps the common
|
|
162
251
|
reasoning intent to its OpenAI-compatible controls:
|
|
163
252
|
|
|
164
|
-
| PLURNK
|
|
165
|
-
|
|
|
166
|
-
| `off`
|
|
167
|
-
| `adaptive`
|
|
168
|
-
| `on`
|
|
253
|
+
| PLURNK posture | `thinking` | `reasoning_effort` |
|
|
254
|
+
| --------------- | --------------------- | -------------------- |
|
|
255
|
+
| `off` | `{ type: disabled }` | omitted |
|
|
256
|
+
| `adaptive` | omitted | omitted |
|
|
257
|
+
| `on` | `{ type: enabled }` | omitted |
|
|
258
|
+
| `on` + budget | `{ type: enabled }` | budget-derived tier |
|
|
169
259
|
|
|
170
260
|
The compatible transport is deliberately retained for:
|
|
171
261
|
|
|
@@ -195,9 +285,10 @@ The universal groups are:
|
|
|
195
285
|
- reasoning activation and optional explicit budget;
|
|
196
286
|
- explicit reasoning response-content style;
|
|
197
287
|
- decode tuning;
|
|
198
|
-
-
|
|
288
|
+
- operation, physical-attempt, first-content, stream-idle, retry, and probe budgets;
|
|
199
289
|
- local GBNF and llama-server capability pins;
|
|
200
290
|
- context-window and generation-envelope overrides;
|
|
291
|
+
- provider-documented cache affinity and explicit cache-write policy;
|
|
201
292
|
- opt-in logprob and raw-body capture.
|
|
202
293
|
|
|
203
294
|
Operator secrets and machine-specific values never belong in committed
|
|
@@ -214,16 +305,20 @@ alias.
|
|
|
214
305
|
|
|
215
306
|
Provider and model facts resolve independently:
|
|
216
307
|
|
|
217
|
-
| Fact
|
|
218
|
-
|
|
|
219
|
-
| Context window
|
|
220
|
-
|
|
|
221
|
-
|
|
|
222
|
-
|
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
308
|
+
| Fact | Natural source | Operator source | Effective value |
|
|
309
|
+
| --- | --- | --- | --- |
|
|
310
|
+
| Context window | Catalog metadata or local endpoint probe. | `PLURNK_PROVIDERS_CONTEXT_WINDOW`. | Minimum when both exist; sole value otherwise. Cataloged cloud miss fails construction; compatible probe miss remains `null` with one warning. |
|
|
311
|
+
| Maximum input | Catalog `limit.input`; no generic live probe. | None. | Catalog value or `null`; never reconstructed from context and output. |
|
|
312
|
+
| Maximum output | Catalog `limit.output`; no generic live probe. | None. | Minimum of catalog value and effective context, or `null`. |
|
|
313
|
+
| Total output budget | None. | `PLURNK_PROVIDERS_OUTPUT_BUDGET`. | Percentage of effective context or absolute count, capped by known context/output limits; a call may only tighten it. |
|
|
314
|
+
| Reasoning budget | None. | Optional `PLURNK_PROVIDERS_REASONING_BUDGET`. | Percentage of effective context or absolute count; valid only as a strict subset of total output and effective only while reasoning is on or adaptive. |
|
|
315
|
+
| Reasoning capability | Catalog `reasoning: true`. | Runtime activation and adapter wire style. | Catalog bit remains informational; it neither activates nor blocks reasoning. |
|
|
316
|
+
| Estimated USD rates | Models.dev input, output, optional reasoning, and optional cache rates. | None. | Missing differently-priced usage or rates produces unknown; exact all-zero rates produces estimated USD zero. |
|
|
317
|
+
|
|
318
|
+
Models.dev cache-read and cache-write rates default to the input rate, and the
|
|
319
|
+
reasoning rate defaults to the output rate, when omitted. A differently priced
|
|
320
|
+
category requires its corresponding usage detail or the request estimate is
|
|
321
|
+
unknown.
|
|
227
322
|
|
|
228
323
|
`instantiateProvider` resolves in this order:
|
|
229
324
|
|
|
@@ -298,7 +393,7 @@ also expose:
|
|
|
298
393
|
- EOS marker removal;
|
|
299
394
|
- exact complete-request counting through `/v1/chat/completions/input_tokens`;
|
|
300
395
|
- exact content token IDs through `/tokenize`;
|
|
301
|
-
- the requirement that the
|
|
396
|
+
- the requirement that the adapter apply a finite output budget.
|
|
302
397
|
|
|
303
398
|
`PLURNK_PROVIDERS_LLAMA_SERVER` may force or disable detection. Probe attempts
|
|
304
399
|
and delay are knobs. A failed probe does not silently assert capabilities.
|
|
@@ -311,14 +406,14 @@ OpenAI-compatible generation endpoint through the SDK adapter.
|
|
|
311
406
|
For a detected llama-server, PLURNK sends the complete reasoning contract on
|
|
312
407
|
every request:
|
|
313
408
|
|
|
314
|
-
|
|
|
409
|
+
| Posture | Template activation | `thinking_budget_tokens` |
|
|
315
410
|
|---|---:|---:|
|
|
316
411
|
| `off` | false | `0` |
|
|
317
|
-
| `adaptive` | true |
|
|
318
|
-
| `on` | true |
|
|
412
|
+
| `adaptive` | true | configured reasoning subset, otherwise omitted |
|
|
413
|
+
| `on` | true | configured reasoning subset, otherwise omitted |
|
|
319
414
|
|
|
320
|
-
The
|
|
321
|
-
|
|
415
|
+
The allowance is contained by the request's total output budget. Template calls
|
|
416
|
+
normally use `reasoning_format: "auto"` for a separate
|
|
322
417
|
readable channel. A GBNF-bearing call uses `"none"` so the exact constrained
|
|
323
418
|
sentence survives response projection; the adapter separates its leading
|
|
324
419
|
reasoning enclosure only after preserving grammar evidence. Process-wide
|
|
@@ -342,7 +437,7 @@ intent. It cannot override:
|
|
|
342
437
|
- data-capture settings;
|
|
343
438
|
- tool, modality, or multi-choice behavior;
|
|
344
439
|
- the consumer-owned output envelope;
|
|
345
|
-
-
|
|
440
|
+
- cache affinity identity or cache-write policy.
|
|
346
441
|
|
|
347
442
|
Generic AI SDK calls accept only settings represented by the SDK's portable
|
|
348
443
|
surface. Compatible endpoints may carry additional sampling keys after reserved
|
|
@@ -366,16 +461,40 @@ normal value. Retry exhaustion is preserved as `attempts` and
|
|
|
366
461
|
`retryExhausted`, and the resulting Problem is not marked retryable after the
|
|
367
462
|
provider has consumed its automatic retry budget.
|
|
368
463
|
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
464
|
+
§provider-capacity-failure A proven exact preflight overflow and an upstream
|
|
465
|
+
context rejection normalize to `ProviderError(kind="capacity_exceeded")` and
|
|
466
|
+
an RFC 9457 status 413. `capacityStage` is `preflight` or `upstream`; a
|
|
467
|
+
non-413 upstream status remains `providerStatus`, while physical request
|
|
468
|
+
accounting retains the status actually received. Preflight rejection occurs
|
|
469
|
+
before provider I/O and therefore opens no request identity and creates no
|
|
470
|
+
request-accounting row. Capacity failures are not connectivity failures and are
|
|
471
|
+
never retried by the provider scheduler; bounded packet recovery belongs to the
|
|
472
|
+
consumer.
|
|
473
|
+
|
|
474
|
+
§provider-connectivity The provider adapter owns one attempt scheduler around
|
|
475
|
+
the complete generation exchange; SDK-internal retries are disabled.
|
|
476
|
+
`PLURNK_PROVIDERS_RETRY_ATTEMPTS=N` permits at most `N + 1` physical requests.
|
|
477
|
+
The layers are independent and a configured value of zero disables only that
|
|
478
|
+
deadline:
|
|
479
|
+
|
|
480
|
+
| Layer | Operator knob | Boundary | Expiry |
|
|
481
|
+
| --- | --- | --- | --- |
|
|
482
|
+
| Operation | `PLURNK_PROVIDERS_OPERATION_TIMEOUT` | Complete logical call, including every attempt and retry delay. | Final `deadline_exceeded` Problem at 504 with `timeoutPhase=operation`; never retried. |
|
|
483
|
+
| Attempt | `PLURNK_PROVIDERS_FETCH_TIMEOUT` | One physical generation request, including response consumption. | Retryable `network_failure` with `timeoutPhase=attempt`. |
|
|
484
|
+
| First content | `PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT` | Response-stream start through first semantic model content; metadata, empty deltas, and transport activity do not satisfy it. | Retryable `network_failure` with `timeoutPhase=first_content`. |
|
|
485
|
+
| Stream idle | `PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT` | Silence between semantic content chunks after content begins. | Retryable `network_failure` with `timeoutPhase=stream_idle`. |
|
|
486
|
+
|
|
487
|
+
Caller cancellation spans the operation and preserves the caller's reason.
|
|
488
|
+
Inner deadline failures consume the ordinary retry budget; retry exhaustion
|
|
489
|
+
adds `attempts` and `retryExhausted`, retains the inner `timeoutPhase` and
|
|
490
|
+
`timeoutMs`, and is final. Every scheduler iteration opens and settles exactly
|
|
491
|
+
one ordered {§provider-request-accounting} record, including response-less
|
|
492
|
+
network failures and timed-out attempts.
|
|
375
493
|
|
|
376
494
|
HTTP 408, 409, 429, and ordinary 5xx responses are retryable unless the endpoint
|
|
377
|
-
explicitly says otherwise
|
|
378
|
-
|
|
495
|
+
explicitly says otherwise through `X-Should-Retry`. That header is authoritative;
|
|
496
|
+
without an explicit directive, endpoint control responses 520–527 are final so
|
|
497
|
+
a router can prevent multiplicative retries behind its own policy.
|
|
379
498
|
|
|
380
499
|
### §provider-interrupted-attempt Provider-declared interruption
|
|
381
500
|
|
|
@@ -386,10 +505,10 @@ completed exchange.
|
|
|
386
505
|
| Concern | Contract |
|
|
387
506
|
| -------------------------- | ---------------------------------------------------------------------------------------------------------------------------- |
|
|
388
507
|
| Normalized finish reason | Exact `insufficient_system_resource` becomes `resource_interrupted`; the raw value remains in `assistantRaw`. |
|
|
389
|
-
| `generate` outcome | Throw `ProviderError(kind="resource_interrupted")` at local status 503 with the normalized attempt on `error.attempt`.
|
|
508
|
+
| `generate` outcome | Throw `ProviderError(kind="resource_interrupted")` at local status 503 with the normalized attempt on `error.attempt` and the same accounting on `error.accounting`. |
|
|
390
509
|
| Partial response | Preserve content, reasoning, usage, model, metadata, optional raw body, and other evidence without admitting it as success. |
|
|
391
510
|
| Automatic replay | None; the Problem has `retryable: false`, and AI SDK retry scheduling has already completed at the successful transport. |
|
|
392
|
-
| Capacity-pool overflow | None
|
|
511
|
+
| Capacity-pool overflow | None under the existing routing policy; when other overflow-eligible failures do reach a sibling, the pool concatenates their request accounting. |
|
|
393
512
|
| Consumer admission | Persist the evidence as an unaccepted attempt; never parse it into executable work, even when its frame looks complete. |
|
|
394
513
|
|
|
395
514
|
## §10 Grammar
|
|
@@ -398,8 +517,9 @@ GBNF is a local llama-server capability, not a generic provider expectation.
|
|
|
398
517
|
The consumer chooses whether to supply a grammar. The provider never creates or
|
|
399
518
|
rewrites one.
|
|
400
519
|
|
|
401
|
-
GBNF defines the accepted
|
|
402
|
-
|
|
520
|
+
GBNF defines the accepted sampled text; it runs before response reasoning is
|
|
521
|
+
separated from regular content. A generated rail may declare a response root
|
|
522
|
+
that composes a template-provided prefix for independent evidence grading. The shipped PLURNK sentence is owned by
|
|
403
523
|
`plurnk-contracts` {§gbnf-turn-shape} and {§gbnf-reasoning-boundary}; this package does
|
|
404
524
|
not restate or rewrite it.
|
|
405
525
|
|
|
@@ -418,7 +538,6 @@ The consumer validates `grammarEvidence.input` outside the enforcer's failure
|
|
|
418
538
|
domain. `PLURNK_PROVIDERS_GBNF_DEBUG` still validates grammar syntax before the
|
|
419
539
|
call and sets `transported: false` for the unconstrained comparison.
|
|
420
540
|
|
|
421
|
-
|
|
422
541
|
## §11 Evidence and metadata
|
|
423
542
|
|
|
424
543
|
§provider-evidence `assistantRaw` is an opaque normalized transport record.
|
|
@@ -449,11 +568,35 @@ correlate them to an entity it actually created rather than reusing `id`.
|
|
|
449
568
|
|
|
450
569
|
## §12 Generation envelopes
|
|
451
570
|
|
|
452
|
-
§provider-generation-envelope
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
571
|
+
§provider-generation-envelope Every request has at most one total output
|
|
572
|
+
budget. It includes visible output and hidden reasoning. An optional reasoning
|
|
573
|
+
budget is a strict subset of that total, never an additive reserve. The
|
|
574
|
+
configured total is a percentage of effective context or an absolute count;
|
|
575
|
+
percentages resolve to the nearest whole token with a one-token minimum. It is
|
|
576
|
+
capped by known context and model-output limits; `generate.maxOutputTokens` may
|
|
577
|
+
only tighten it for one call. The effective reasoning subset tightens with that
|
|
578
|
+
total and remains strictly smaller.
|
|
579
|
+
|
|
580
|
+
The adapter owns native projection. A backend whose generic SDK maximum already
|
|
581
|
+
includes reasoning receives the total directly. When a native SDK instead adds
|
|
582
|
+
an explicit reasoning allowance to its generic visible-output maximum, the
|
|
583
|
+
adapter sends `total - reasoning` through the generic field and the reasoning
|
|
584
|
+
subset through the documented provider option. Core and other callers never
|
|
585
|
+
reconstruct this arithmetic.
|
|
586
|
+
|
|
587
|
+
§provider-output-budget-conformance When a completed response reports
|
|
588
|
+
normalized output-token usage greater than its effective total output budget,
|
|
589
|
+
the exchange is an `invalid_response` at 502 rather than an admitted result or
|
|
590
|
+
a prompt-capacity 413. Its complete failed-attempt evidence and settled charged
|
|
591
|
+
request remain available. The violation is final and is never automatically
|
|
592
|
+
replayed. Missing output usage cannot prove a violation.
|
|
593
|
+
|
|
594
|
+
`PLURNK_PROVIDERS_OUTPUT_BUDGET` is required for standard providers and ships
|
|
595
|
+
as `35%`. `PLURNK_PROVIDERS_REASONING_BUDGET` is optional; leaving it unset
|
|
596
|
+
preserves provider-adaptive depth. A backend known to decode without a finite
|
|
597
|
+
limit advertises `requiresOutputBudget` and fails construction when no total can
|
|
598
|
+
be resolved. The retired additive reserve knobs fail hard rather than creating
|
|
599
|
+
a second envelope contract.
|
|
457
600
|
|
|
458
601
|
## §13 Capacity pool
|
|
459
602
|
|
|
@@ -466,8 +609,13 @@ availability and rate-limit failures that carry no normalized response attempt;
|
|
|
466
609
|
{§provider-interrupted-attempt} propagates without overflow.
|
|
467
610
|
|
|
468
611
|
Prompt measurement covers every backend that could receive the request. The
|
|
469
|
-
pool takes the largest result; differing exact counts or any proven
|
|
470
|
-
an `upper_bound`,
|
|
612
|
+
pool takes the largest quantified result; differing exact counts or any proven
|
|
613
|
+
bound yield an `upper_bound`, any estimate makes the aggregate an estimate, and
|
|
614
|
+
any unavailable backend makes it unavailable. Physical limits and budgets are
|
|
615
|
+
independent safe minima across the pool. `inputCapacity` is the minimum of each
|
|
616
|
+
backend's complete derived input capacity, never a synthetic subtraction across
|
|
617
|
+
minima from different backends; request-specific output tightening repeats the
|
|
618
|
+
complete-envelope derivation per backend before taking the minimum.
|
|
471
619
|
|
|
472
620
|
## §14 Conformance
|
|
473
621
|
|
|
@@ -479,7 +627,12 @@ Coverage MUST prove:
|
|
|
479
627
|
- compatible extension preservation;
|
|
480
628
|
- timeout, retry, cancellation, interrupted-attempt, and final-error behavior;
|
|
481
629
|
- local capability probes and pins;
|
|
482
|
-
- exact, bounded, and
|
|
630
|
+
- exact, bounded, estimated, and unavailable complete-request measurements;
|
|
631
|
+
- independent input/context/output limits, asymmetric admission, and normalized
|
|
632
|
+
local/upstream capacity failures;
|
|
633
|
+
- one total output budget and native additive-reasoning projection;
|
|
634
|
+
- provider-reported output beyond that budget failing once with complete
|
|
635
|
+
attempt and accounting evidence;
|
|
483
636
|
- local reasoning activation, response-wide allowance, and GBNF coexistence;
|
|
484
637
|
- explicit tagged-reasoning projection across streamed, buffered, capped, and
|
|
485
638
|
literal-tag responses;
|
package/dist/AiSdkProvider.d.ts
CHANGED
|
@@ -1,32 +1,48 @@
|
|
|
1
|
-
import type {
|
|
1
|
+
import type { ChatMessage, PromptTokenMeasurement, Provider, ProviderCostNormalizer, ProviderGenerateArgs, ProviderRequestCapacity, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
2
2
|
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
3
|
-
import type {
|
|
3
|
+
import type { JSONValue } from "ai";
|
|
4
|
+
import type { Reasoning, ReasoningResponseStyle } from "./env.ts";
|
|
4
5
|
import type { LanguageModel } from "ai";
|
|
5
6
|
import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
6
7
|
export type ProviderFetch = typeof globalThis.fetch;
|
|
7
8
|
export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "thinking_effort" | "template" | "anthropic";
|
|
8
9
|
export type GrammarStyle = "none" | "llamacpp";
|
|
10
|
+
export type CacheAffinity = {
|
|
11
|
+
readonly target: "header" | "body";
|
|
12
|
+
readonly name: string;
|
|
13
|
+
} | {
|
|
14
|
+
readonly target: "provider-option";
|
|
15
|
+
readonly provider: string;
|
|
16
|
+
readonly name: string;
|
|
17
|
+
};
|
|
18
|
+
export type AiSdkProviderOptions = Record<string, Record<string, JSONValue | undefined>>;
|
|
9
19
|
export type AiSdkProviderConfig = {
|
|
10
20
|
model: string;
|
|
11
21
|
url?: string;
|
|
12
22
|
languageModel?: LanguageModel;
|
|
13
23
|
attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
14
24
|
fetchTimeoutMs: number;
|
|
25
|
+
operationTimeoutMs: number;
|
|
26
|
+
firstContentTimeoutMs: number;
|
|
15
27
|
streamIdleTimeoutMs?: number;
|
|
16
28
|
headers?: Record<string, string>;
|
|
17
29
|
fetch?: ProviderFetch;
|
|
18
30
|
contextWindow?: number | null;
|
|
31
|
+
maxInputTokens?: number | null;
|
|
32
|
+
maxOutputTokens?: number | null;
|
|
33
|
+
outputBudget?: number | null;
|
|
34
|
+
reasoningBudget?: number | null;
|
|
35
|
+
additiveReasoningProvider?: "anthropic" | "bedrock";
|
|
19
36
|
reasoningStyle?: ReasoningStyle;
|
|
20
37
|
reasoningResponseStyle?: ReasoningResponseStyle;
|
|
21
38
|
countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
kind: "authoritative";
|
|
25
|
-
}>;
|
|
26
|
-
normalizeCharge?: AuthoritativeChargeNormalizer;
|
|
39
|
+
estimateCost?: (usage: ProviderUsage | undefined) => ProviderCost;
|
|
40
|
+
normalizeCost?: ProviderCostNormalizer;
|
|
27
41
|
source?: string;
|
|
28
42
|
grammarStyle?: GrammarStyle;
|
|
29
|
-
|
|
43
|
+
cacheAffinity?: CacheAffinity;
|
|
44
|
+
systemCacheProviderOptions?: AiSdkProviderOptions;
|
|
45
|
+
reasoningResponseProviderOptions?: AiSdkProviderOptions;
|
|
30
46
|
serviceTier?: string;
|
|
31
47
|
gbnfDebug?: boolean;
|
|
32
48
|
streaming?: boolean;
|
|
@@ -38,7 +54,7 @@ export type AiSdkProviderConfig = {
|
|
|
38
54
|
tokenizeUrl?: string;
|
|
39
55
|
promptTokensUrl?: string;
|
|
40
56
|
servedModel?: string;
|
|
41
|
-
|
|
57
|
+
requiresOutputBudget?: boolean;
|
|
42
58
|
reasoning: Reasoning;
|
|
43
59
|
temperature: number;
|
|
44
60
|
repeatPenalty: number;
|
|
@@ -51,8 +67,6 @@ export type AiSdkProviderConfig = {
|
|
|
51
67
|
errorDetailLimit?: number;
|
|
52
68
|
topLogprobs?: number | null;
|
|
53
69
|
rawBody?: boolean;
|
|
54
|
-
reasoningReserve?: ReserveSpec;
|
|
55
|
-
completionReserve?: ReserveSpec;
|
|
56
70
|
tuningFloors?: boolean;
|
|
57
71
|
};
|
|
58
72
|
export declare const effortFromBudget: (budget: number) => "low" | "medium" | "high";
|
|
@@ -62,31 +76,17 @@ export default class AiSdkProvider implements Provider {
|
|
|
62
76
|
tokenize?: (text: string) => Promise<number[]>;
|
|
63
77
|
constructor(config: AiSdkProviderConfig);
|
|
64
78
|
get contextWindow(): number | null;
|
|
65
|
-
get
|
|
66
|
-
get
|
|
79
|
+
get maxInputTokens(): number | null;
|
|
80
|
+
get maxOutputTokens(): number | null;
|
|
81
|
+
get outputBudget(): number | null;
|
|
82
|
+
get reasoningBudget(): number | null;
|
|
83
|
+
get inputCapacity(): number | null;
|
|
67
84
|
get model(): string;
|
|
68
85
|
get servedModel(): string | undefined;
|
|
69
|
-
get
|
|
86
|
+
get requiresOutputBudget(): boolean | undefined;
|
|
70
87
|
get constrainsOutput(): boolean;
|
|
71
88
|
countPromptTokens(messages: readonly ChatMessage[], signal?: AbortSignal): Promise<PromptTokenMeasurement>;
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
kind: "authoritative";
|
|
75
|
-
}>;
|
|
76
|
-
generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }: {
|
|
77
|
-
messages: ChatMessage[];
|
|
78
|
-
workerId: string;
|
|
79
|
-
primaryWorkerId?: string;
|
|
80
|
-
signal?: AbortSignal;
|
|
81
|
-
grammar?: string;
|
|
82
|
-
maxTokens?: number;
|
|
83
|
-
attributions?: string[];
|
|
84
|
-
client?: string;
|
|
85
|
-
strikes?: number;
|
|
86
|
-
workspaceId?: string;
|
|
87
|
-
loop?: number;
|
|
88
|
-
turn?: number;
|
|
89
|
-
sampling?: Record<string, unknown>;
|
|
90
|
-
}): Promise<ProviderResponse>;
|
|
89
|
+
assessRequestCapacity(messages: readonly ChatMessage[], maxOutputTokens?: number, signal?: AbortSignal): Promise<ProviderRequestCapacity>;
|
|
90
|
+
generate({ messages, workerId, primaryWorkerId, signal, grammar, maxOutputTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }: ProviderGenerateArgs): Promise<ProviderResponse>;
|
|
91
91
|
}
|
|
92
92
|
//# sourceMappingURL=AiSdkProvider.d.ts.map
|