@plurnk/plurnk-providers 1.21.0 → 1.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +110 -24
- package/README.md +19 -6
- package/SPEC.md +250 -97
- package/dist/AiSdkProvider.d.ts +19 -13
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +154 -73
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/AiSdkRequestBody.d.ts +7 -11
- package/dist/AiSdkRequestBody.d.ts.map +1 -1
- package/dist/AiSdkRequestBody.js +18 -90
- package/dist/AiSdkRequestBody.js.map +1 -1
- package/dist/InferenceAdmission.d.ts +7 -0
- package/dist/InferenceAdmission.d.ts.map +1 -0
- package/dist/InferenceAdmission.js +58 -0
- package/dist/InferenceAdmission.js.map +1 -0
- package/dist/Mock.d.ts +1 -1
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +2 -2
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -2
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +2 -2
- package/dist/Pool.js.map +1 -1
- package/dist/RepeatedLine.d.ts +9 -0
- package/dist/RepeatedLine.d.ts.map +1 -0
- package/dist/RepeatedLine.js +27 -0
- package/dist/RepeatedLine.js.map +1 -0
- package/dist/RequestFields.d.ts +25 -0
- package/dist/RequestFields.d.ts.map +1 -0
- package/dist/RequestFields.js +240 -0
- package/dist/RequestFields.js.map +1 -0
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +38 -7
- package/dist/accounting.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +8 -3
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +77 -24
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +4 -3
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +73 -170
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +11 -9
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/discover.d.ts +0 -1
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js +1 -10
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +19 -8
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +125 -23
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +1 -8
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +9 -13
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +8 -7
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +6 -5
- package/dist/index.js.map +1 -1
- package/dist/model-options.d.ts +3 -0
- package/dist/model-options.d.ts.map +1 -0
- package/dist/model-options.js +26 -0
- package/dist/model-options.js.map +1 -0
- package/dist/ollama.d.ts.map +1 -1
- package/dist/ollama.js +1 -0
- package/dist/ollama.js.map +1 -1
- package/dist/provider-env.d.ts +3 -0
- package/dist/provider-env.d.ts.map +1 -0
- package/dist/provider-env.js +8 -0
- package/dist/provider-env.js.map +1 -0
- package/dist/providerError.d.ts +2 -2
- package/dist/providerError.d.ts.map +1 -1
- package/dist/providerError.js +19 -22
- package/dist/providerError.js.map +1 -1
- package/dist/reasoning-effort.d.ts +4 -3
- package/dist/reasoning-effort.d.ts.map +1 -1
- package/dist/reasoning-effort.js +18 -2
- package/dist/reasoning-effort.js.map +1 -1
- package/dist/sdkModels.d.ts +3 -2
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +39 -66
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +8 -8
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +5 -5
- package/dist/types.js.map +1 -1
- package/dist/usage.d.ts +5 -1
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +62 -17
- package/dist/usage.js.map +1 -1
- package/package.json +6 -7
- package/src/AiSdkProvider.test.ts +0 -2841
- package/src/AiSdkProvider.ts +0 -1064
- package/src/AiSdkRequestBody.ts +0 -364
- package/src/LeadingReasoning.test.ts +0 -34
- package/src/LeadingReasoning.ts +0 -87
- package/src/Mock.test.ts +0 -209
- package/src/Mock.ts +0 -197
- package/src/Pool.test.ts +0 -267
- package/src/Pool.ts +0 -261
- package/src/ProviderRegistry.test.ts +0 -483
- package/src/ProviderRegistry.ts +0 -128
- package/src/accounting.test.ts +0 -139
- package/src/accounting.ts +0 -172
- package/src/accountingPublic.ts +0 -9
- package/src/aiSdkTransport.test.ts +0 -476
- package/src/aiSdkTransport.ts +0 -776
- package/src/boundaries.test.ts +0 -73
- package/src/capacity.test.ts +0 -124
- package/src/capacity.ts +0 -174
- package/src/catalogProvider.test.ts +0 -900
- package/src/catalogProvider.ts +0 -476
- package/src/compatibleProvider.test.ts +0 -157
- package/src/compatibleProvider.ts +0 -217
- package/src/cost.test.ts +0 -114
- package/src/cost.ts +0 -140
- package/src/defaults.test.ts +0 -30
- package/src/defaults.ts +0 -9
- package/src/discover.test.ts +0 -200
- package/src/discover.ts +0 -136
- package/src/env.test.ts +0 -335
- package/src/env.ts +0 -380
- package/src/errors.test.ts +0 -278
- package/src/errors.ts +0 -234
- package/src/index.ts +0 -100
- package/src/inputModalities.test.ts +0 -54
- package/src/lexicon-guard.test.ts +0 -58
- package/src/notices.ts +0 -22
- package/src/ollama.test.ts +0 -66
- package/src/ollama.ts +0 -66
- package/src/openai.ts +0 -15
- package/src/promptTokens.ts +0 -45
- package/src/providerDefaults.test.ts +0 -51
- package/src/providerError.ts +0 -143
- package/src/reasoning-effort.ts +0 -15
- package/src/sdkModels.test.ts +0 -293
- package/src/sdkModels.ts +0 -460
- package/src/types.ts +0 -337
- package/src/usage.test.ts +0 -149
- package/src/usage.ts +0 -266
- package/src/warnings.test.ts +0 -31
- package/src/warnings.ts +0 -0
package/SPEC.md
CHANGED
|
@@ -125,6 +125,23 @@ produces an ordinary estimated amount of USD `0`. Rate calculation is internal
|
|
|
125
125
|
to the provider request; Core, digest, ping, and clients never call a parallel
|
|
126
126
|
pricing method.
|
|
127
127
|
|
|
128
|
+
Router accounting uses the complete cost to the caller, not just the router's
|
|
129
|
+
fee. OpenRouter's response `usage` is authoritative; its SDK metadata projection
|
|
130
|
+
omits the BYOK discriminator. Apply the same rule to successful and failed
|
|
131
|
+
requests, independently of raw-body capture:
|
|
132
|
+
|
|
133
|
+
| OpenRouter evidence | Monetary result |
|
|
134
|
+
| --- | --- |
|
|
135
|
+
| `is_byok: false` and `cost` | `charged`: `cost`; upstream inference is already included. |
|
|
136
|
+
| `is_byok: true`, `cost`, and `cost_details.upstream_inference_cost` | `charged`: exact decimal sum of router fee and upstream charge. |
|
|
137
|
+
| A monetary field without a BYOK discriminator, or BYOK missing either component | `unknown`; neither zero nor catalog pricing completes the reported charge. |
|
|
138
|
+
| No reported charge and no BYOK indication | Ordinary Models.dev fallback above. |
|
|
139
|
+
|
|
140
|
+
Explicit zero components are valid. Malformed monetary values are contract
|
|
141
|
+
violations, never coerced. Raw routing and billing evidence remain available
|
|
142
|
+
under {§provider-evidence}; Core and clients consume the existing single cost.
|
|
143
|
+
See [OpenRouter usage accounting](https://openrouter.ai/docs/cookbook/administration/usage-accounting).
|
|
144
|
+
|
|
128
145
|
### Generation
|
|
129
146
|
|
|
130
147
|
§provider-cache-identity `generate` requires a non-empty, stable, opaque
|
|
@@ -231,14 +248,21 @@ the SDK. PLURNK supplies cancellation and deadline signals and owns the sole
|
|
|
231
248
|
cross-attempt scheduler so every physical request remains observable and
|
|
232
249
|
accountable.
|
|
233
250
|
|
|
251
|
+
| Usage input | Normalization |
|
|
252
|
+
| --- | --- |
|
|
253
|
+
| Native protocol | Preserve the SDK's inclusive totals and category counts; raw protocol counters are evidence, not interchangeable totals. |
|
|
254
|
+
| OpenAI-compatible wire | Retain extended counters and exact total identities the SDK may omit. |
|
|
255
|
+
| Compatible SDK input partition | Retain its uncached count when total input and an explicitly reported cache-read count agree with the wire; never promote SDK defaults over absent or contradictory wire counts. |
|
|
256
|
+
| Input total plus two partition counts | Derive the remaining count exactly; otherwise preserve absence. |
|
|
257
|
+
|
|
234
258
|
PLURNK maps its generic settings to AI SDK call settings:
|
|
235
259
|
|
|
236
260
|
- `temperature`, `top_p`, `top_k`;
|
|
237
261
|
- presence and frequency penalties;
|
|
238
262
|
- stop sequences and seed;
|
|
239
263
|
- output-token ceiling;
|
|
240
|
-
-
|
|
241
|
-
|
|
264
|
+
- the supported efforts and optional numeric control under
|
|
265
|
+
{§provider-effort}.
|
|
242
266
|
|
|
243
267
|
Provider-specific options are permitted only where they preserve a documented
|
|
244
268
|
PLURNK product contract the generic SDK surface cannot express.
|
|
@@ -253,28 +277,31 @@ catalog figure. Without catalog rates the override must declare `input` and
|
|
|
253
277
|
response cost still outranks any estimate. Unknown keys, repeats, and negative
|
|
254
278
|
or non-numeric rates refuse at construction.
|
|
255
279
|
|
|
256
|
-
§provider-
|
|
257
|
-
{§
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
280
|
+
§provider-effort The portable vocabulary comes from
|
|
281
|
+
{§effort-wire}. `adaptive` uses the first applicable projection:
|
|
282
|
+
|
|
283
|
+
| Condition | Projection |
|
|
284
|
+
| --- | --- |
|
|
285
|
+
| Explicit adaptive declaration: `REASONING_ADAPTIVE_BODY`, or a native model family's `ADAPTIVE_OPTIONS` ({§provider-model-options}) | Preserve that mechanism; an explicit `{}` retains the enabled endpoint's default. |
|
|
286
|
+
| The configured `PLURNK_PROVIDERS_EFFORT_FALLBACK` is supported by both model and transport | Send that exact effort. The shipped fallback is `high`, not the strongest available level. |
|
|
287
|
+
| No supported fallback, including an empty fallback setting | Retain reasoning activation and the provider's default effort; never invent a level or escalate to another one. |
|
|
288
|
+
|
|
289
|
+
Activation is distinct from effort: declared enable fields accompany active
|
|
290
|
+
reasoning, and a toggle-only route uses its documented enable mechanism.
|
|
291
|
+
Non-reasoning models receive no reasoning controls; `off` is the explicit opt-out.
|
|
292
|
+
A fixed policy retains its exact name and is rejected before provider I/O when
|
|
293
|
+
either the route or transport cannot represent it. Every provider exposes that
|
|
294
|
+
exact intersection. A numeric reasoning budget constrains
|
|
267
295
|
the generation envelope independently and never selects or changes policy. On
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
requiring credentials, or performing provider I/O.
|
|
296
|
+
routes whose controls are exclusive, fixed effort plus a numeric budget is
|
|
297
|
+
rejected before I/O. Under `adaptive`, an explicit budget selects the numeric
|
|
298
|
+
control; `off` suppresses it.
|
|
299
|
+
|
|
300
|
+
`catalogEfforts` projects that same admission calculation from catalog
|
|
301
|
+
facts and provider-wide environment declarations over the installed defaults,
|
|
302
|
+
without constructing a model, requiring credentials, or performing provider I/O.
|
|
276
303
|
Admission respects the installed projection: native SDK fixed efforts exclude
|
|
277
|
-
`max`;
|
|
304
|
+
`max`; declared transport vocabularies constrain per-call option projections.
|
|
278
305
|
|
|
279
306
|
Models.dev's route-specific `reasoning_options` is the capability authority for
|
|
280
307
|
what the daemon offers on its own; the daemon never adds vendor behaviour absent
|
|
@@ -291,7 +318,7 @@ explicit compatible adapter owns the wire projection:
|
|
|
291
318
|
| --- | --- | --- |
|
|
292
319
|
| `reasoning: false` | No reasoning request | `off` |
|
|
293
320
|
| `reasoning_options: []` | Provider default | None |
|
|
294
|
-
| `effort.values` | Native dynamic mechanism, otherwise
|
|
321
|
+
| `effort.values` | Native dynamic mechanism, otherwise the supported configured fallback or provider default | Transportable fixed members of {§effort-wire}; `off` only when `none` is transportable |
|
|
295
322
|
| `toggle` | Native or explicitly declared activation, otherwise provider default | `off` only when that transport owns the toggle wire |
|
|
296
323
|
| `budget_tokens` | Does not select policy | None; an adapter may use its bounds when projecting the independent budget |
|
|
297
324
|
| No catalog entry | Explicit adapter declaration | Only the declaration's exact subset |
|
|
@@ -321,34 +348,36 @@ provider's documented header, body field, or native SDK option. The common
|
|
|
321
348
|
transport neither guesses from protocol resemblance nor sends a generic cache
|
|
322
349
|
field to an unknown provider. The operator may disable affinity globally or per
|
|
323
350
|
alias; automatic provider caching without an affinity control remains untouched.
|
|
351
|
+
The environment's `CACHE_AFFINITY_FIELD` declares that placement as
|
|
352
|
+
`{"target":"header"|"body","name":"…"}` or
|
|
353
|
+
`{"target":"provider-option","provider":"…","name":"…"}`. It follows
|
|
354
|
+
the provider/route/alias precedence of {§provider-wire-declaration}; `null`
|
|
355
|
+
clears the declaration. Placement is independent of the enable switch, cannot
|
|
356
|
+
replace transport-owned fields, and does not imply support from a provider's name.
|
|
357
|
+
|
|
358
|
+
§provider-model-options **A model family's native options are a provider declaration.** Models.dev
|
|
359
|
+
does not say which models on a native SDK take adaptive reasoning or explicit cache writes, and the SDKs
|
|
360
|
+
do not export their tables, so a provider declares them as data: `ADAPTIVE_OPTIONS` and
|
|
361
|
+
`SYSTEM_CACHE_OPTIONS` are JSON arrays of `{"models":["<glob>",…],"options":{…}}` rules. The first rule
|
|
362
|
+
with a glob matching the route's model id (`path.matchesGlob`) supplies its per-call provider options;
|
|
363
|
+
without a match none apply. They follow the provider/route/alias precedence of
|
|
364
|
+
{§provider-wire-declaration}, an empty value declares none, and a malformed value fails construction.
|
|
365
|
+
None ships: Plurnk's shipped defaults serve open models only, and an operator who routes a closed model
|
|
366
|
+
family declares its options in their own configuration.
|
|
324
367
|
|
|
325
368
|
§provider-cache-write-policy **Cache-write policy is separate from affinity.**
|
|
326
369
|
`PLURNK_PROVIDERS_CACHE_WRITE_POLICY` is `off` or `stable-system`. The latter
|
|
327
370
|
marks only the final leading system instruction as an explicit reusable cache
|
|
328
|
-
boundary, and only on routes whose
|
|
371
|
+
boundary, and only on routes whose provider declares the control in
|
|
372
|
+
`SYSTEM_CACHE_OPTIONS` ({§provider-model-options}). It does
|
|
329
373
|
not mark the changing user packet or enable an API-wide automatic cache mode.
|
|
330
374
|
Unsupported routes receive no invented option. The default five-minute
|
|
331
375
|
provider lifetime is used; a longer, differently priced lifetime is not an
|
|
332
376
|
implicit transport choice.
|
|
333
377
|
|
|
334
|
-
§deepseek-reasoning-request
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
| PLURNK posture | `thinking` | `reasoning_effort` |
|
|
338
|
-
| --------------- | --------------------- | -------------------- |
|
|
339
|
-
| `off` | `{ type: disabled }` | omitted |
|
|
340
|
-
| `adaptive` | `{ type: enabled }` | omitted |
|
|
341
|
-
| `high` | `{ type: enabled }` | `high` |
|
|
342
|
-
|
|
343
|
-
The direct API does not distinguish portable `low` or `medium` intent and
|
|
344
|
-
therefore advertises only `off`, `adaptive`, and `high`.
|
|
345
|
-
|
|
346
|
-
§provider-reasoning-style A reasoning style names the wire a compatible route's
|
|
347
|
-
reasoning controls take. `PLURNK_PROVIDERS_PROVIDER_<PREFIX>_REASONING_STYLE`
|
|
348
|
-
declares it for every route of a provider; `PLURNK_PROVIDERS_REASONING_STYLE`,
|
|
349
|
-
alias-scopable like every bare knob, declares it for one route and wins. One
|
|
350
|
-
provider can serve models whose controls differ, so a route states its own style. The daemon never infers a
|
|
351
|
-
style from a model name.
|
|
378
|
+
§deepseek-reasoning-request Direct DeepSeek uses the panel's request-field
|
|
379
|
+
declarations under {§provider-wire-declaration}. Models.dev supplies each
|
|
380
|
+
model's effort vocabulary; it is not a second fixed list in this adapter.
|
|
352
381
|
|
|
353
382
|
The compatible transport is deliberately retained for:
|
|
354
383
|
|
|
@@ -359,6 +388,49 @@ The compatible transport is deliberately retained for:
|
|
|
359
388
|
It carries PLURNK-only fields and raw wire evidence without reimplementing the
|
|
360
389
|
SDK's ordinary transport.
|
|
361
390
|
|
|
391
|
+
### §provider-wire-declaration Request-field declarations
|
|
392
|
+
|
|
393
|
+
Models.dev owns model capabilities, effort vocabulary, and numeric budget bounds.
|
|
394
|
+
For endpoints whose field names it does not describe, the provider's
|
|
395
|
+
environment panel supplies a bounded projection, independent of provider identity:
|
|
396
|
+
|
|
397
|
+
| Declaration suffix | Meaning |
|
|
398
|
+
| --- | --- |
|
|
399
|
+
| `OPTIONS_NAMESPACE` | Native SDK's per-call `providerOptions` namespace. Without one, declared fields target the compatible request body. |
|
|
400
|
+
| `OUTPUT_PATH` | RFC 6901 object-member pointer for the **inclusive** output ceiling; absent uses the compatible SDK's `max_tokens` field. |
|
|
401
|
+
| `REASONING_EFFORT_PATH` / `REASONING_BUDGET_PATH` | Pointers for exact effort or numeric reasoning subset. No pointer means no such control. |
|
|
402
|
+
| `REASONING_EFFORTS` | Additional declared effort values, unioned with Models.dev. |
|
|
403
|
+
| `REASONING_TRANSPORT_EFFORTS` | Optional transport vocabulary, intersected with catalog/declaration efforts. It adds no model capability. |
|
|
404
|
+
| `REASONING_CONTROLS` | Required when both pointers exist: `exclusive` refuses fixed effort plus budget; `combined` sends both. |
|
|
405
|
+
| `REASONING_ON_BODY` | Static reasoning activation fields, merged with the selected control. |
|
|
406
|
+
| `REASONING_OFF_BODY` | Explicit disable fields; otherwise a declared `none` effort can disable. |
|
|
407
|
+
| `REASONING_ADAPTIVE_BODY` | Explicit adaptive fields, ahead of graded fallback; `{}` retains the enabled endpoint's default. |
|
|
408
|
+
| `REASONING_TOGGLE_BODY` | Reasoning enable fields when Models.dev declares a toggle and neither an explicit adaptive body nor a supported fallback effort applies. |
|
|
409
|
+
|
|
410
|
+
The existing provider declaration prefix is
|
|
411
|
+
`PLURNK_PROVIDERS_PROVIDER_<NAME>_`; `PLURNK_PROVIDERS_<suffix>` overrides
|
|
412
|
+
it for the route and accepts the ordinary alias suffix. These are data, not
|
|
413
|
+
executable transformations. Static bodies cannot replace transport, sampling,
|
|
414
|
+
or managed numeric controls; pointers cannot overlap. Configuration fails at
|
|
415
|
+
construction before inference when a requested policy or numeric control is
|
|
416
|
+
unrepresentable. A tighter per-call envelope reprojects the same declaration.
|
|
417
|
+
Discovery and generation share the resolved policy set. A cataloged
|
|
418
|
+
non-reasoning model receives no reasoning fields.
|
|
419
|
+
Native SDKs retain their own output field; `OUTPUT_PATH` is incompatible with an
|
|
420
|
+
options namespace. Request-local options are projected after the effective
|
|
421
|
+
envelope is known, never frozen into SDK model construction. Streaming and
|
|
422
|
+
non-streaming calls use the same projection.
|
|
423
|
+
Retired `REASONING_STYLE` selectors fail at the selected provider/alias boundary;
|
|
424
|
+
they neither select a preset nor silently coexist with these declarations.
|
|
425
|
+
|
|
426
|
+
A native SDK's portable reasoning setting has no `max`, so a native route without
|
|
427
|
+
a namespace cannot send an effort the catalog documents beyond it; construction
|
|
428
|
+
refuses it and names the declaration that would. Native SDKs built on
|
|
429
|
+
openai-compatible (DeepInfra, Together) write `reasoning_effort` from their own
|
|
430
|
+
`reasoningEffort` option after spreading the others, overwriting a raw
|
|
431
|
+
`reasoning_effort`; their shipped declarations therefore point at
|
|
432
|
+
`/reasoningEffort`, with `{"reasoningEffort":"none"}` as the off body.
|
|
433
|
+
|
|
362
434
|
## §4 Operator configuration
|
|
363
435
|
|
|
364
436
|
§provider-configuration Every operational value is an environment knob
|
|
@@ -434,8 +506,8 @@ Provider and model facts resolve independently:
|
|
|
434
506
|
| Context window | Catalog metadata or local endpoint probe. | `PLURNK_PROVIDERS_CONTEXT_WINDOW`. | Minimum when both exist; sole value otherwise. Cataloged cloud miss fails construction; compatible probe miss remains `null` with one warning. |
|
|
435
507
|
| Maximum input | Catalog `limit.input`; no generic live probe. | None. | Catalog value or `null`; never reconstructed from context and output. |
|
|
436
508
|
| Maximum output | Catalog `limit.output`; no generic live probe. | None. | Minimum of catalog value and effective context, or `null`. |
|
|
437
|
-
| Total output budget | None. | `PLURNK_PROVIDERS_OUTPUT_BUDGET`. |
|
|
438
|
-
|
|
|
509
|
+
| Total output budget | None. | `PLURNK_PROVIDERS_OUTPUT_BUDGET`. | Curation reservation: percentage of effective context or absolute count, capped by known context/output limits; a call may only tighten it. The response grant may expand under {§provider-flexed-allowance}. |
|
|
510
|
+
| Effort | Catalog `reasoning_options` intersected with the installed adapter; explicit adapter declaration for uncataloged routes. | `PLURNK_PROVIDERS_EFFORT`, initially; durable worker selection thereafter. | A supported member of {§effort-wire}, projected under {§provider-effort}. The shipped selection is `adaptive`. |
|
|
439
511
|
| Reasoning budget | None. | Optional `PLURNK_PROVIDERS_REASONING_BUDGET`. | Percentage of effective context or absolute count; valid only as a strict subset of total output and effective unless reasoning is `off`. |
|
|
440
512
|
| Cost override | None. | Optional `PLURNK_PROVIDERS_COST`. | {§operator-cost-override} — comma-separated `key=value` per-1M-token USD rates over `input, output, reasoning, cacheRead, cacheWrite`; merges over the Models.dev catalog block (the catalog is the starting point), alias-scoped like every knob. Without catalog rates the override must declare `input` and `output`. The cost estimate's `source` names the override; a provider-reported response cost still outranks any estimate. |
|
|
441
513
|
| Reasoning capability | Catalog `reasoning` and route-specific `reasoning_options`. | Adapter wire style only where the catalog cannot name the native field. | Catalog controls determine admissible policy; the adapter determines its wire projection. |
|
|
@@ -477,9 +549,9 @@ forcing a vendor-owned resource prefix into PLURNK aliases. Ambiguous suffixes
|
|
|
477
549
|
fail to resolve.
|
|
478
550
|
|
|
479
551
|
The catalog package identifies the protocol family, not a mandatory client
|
|
480
|
-
implementation. OpenRouter and DeepInfra
|
|
481
|
-
|
|
482
|
-
|
|
552
|
+
implementation. OpenRouter charges and DeepInfra estimates use cost normalizers
|
|
553
|
+
over documented response fields, retaining their distinct monetary character
|
|
554
|
+
under {§provider-monetary-evidence}.
|
|
483
555
|
|
|
484
556
|
§provider-fact-authority Provider declarations configure facts, not
|
|
485
557
|
credentials, and Models.dev is authoritative for cataloged providers: package
|
|
@@ -530,8 +602,7 @@ PLURNK `Provider`, read PLURNK tuning knobs, or reproduce transport policy.
|
|
|
530
602
|
Discovery is scope-agnostic and memoized per process. Duplicate names fail hard.
|
|
531
603
|
The common plugin trust gate applies before import ({§plugin-trust-boundary}). A plugin absent from
|
|
532
604
|
Models.dev requires an explicit context-window pin because PLURNK will not guess
|
|
533
|
-
model physics. `Discovery.packageAttributions` carries the canonical package map
|
|
534
|
-
the published name-keyed `Discovery.attributions` remains its 1.x projection.
|
|
605
|
+
model physics. `Discovery.packageAttributions` carries the canonical package map.
|
|
535
606
|
|
|
536
607
|
## §7 Local capabilities
|
|
537
608
|
|
|
@@ -575,6 +646,16 @@ reasoning enclosure only after preserving grammar evidence. Process-wide
|
|
|
575
646
|
llama-server flags are fallback server configuration, not part of the PLURNK
|
|
576
647
|
contract and need not be synchronized with an alias.
|
|
577
648
|
|
|
649
|
+
§repetition-stop **A response that repeats itself is stopped.** A degenerate model can stream one line
|
|
650
|
+
without end; content keeps arriving, so no silence deadline fires, and the call runs to its output
|
|
651
|
+
allowance. The transport counts the complete lines of each streamed channel, text and reasoning, and
|
|
652
|
+
when one line of 16 or more characters has appeared `PLURNK_PROVIDERS_REPEATED_LINE_LIMIT` times it stops
|
|
653
|
+
reading: the stream is cancelled, the physical request settles with whatever usage the chunks carried,
|
|
654
|
+
and the call fails as `repetition` (422, not retryable) carrying the partial attempt and the sentence
|
|
655
|
+
"The response repeated one line N times and was stopped: `<line>`". The consumer reads it as an invalid
|
|
656
|
+
emission, never as a provider failure to recover. Measured: a granite-4.2-8b run repeated one line 358
|
|
657
|
+
times across 45,875 tokens and 9 min 42 s; a limit of 32 stops it 14.6% of the way in (#876).
|
|
658
|
+
|
|
578
659
|
§llama-reasoning-request The allowance is cumulative across the complete response. Opening a second or
|
|
579
660
|
later reasoning block does not replenish it. Template parsing, the reasoning
|
|
580
661
|
sampler, normalized usage, and the returned reasoning channel MUST agree on that
|
|
@@ -598,14 +679,14 @@ Generic AI SDK calls accept only settings represented by the SDK's portable
|
|
|
598
679
|
surface. Compatible endpoints may carry additional sampling keys after reserved
|
|
599
680
|
keys are removed.
|
|
600
681
|
|
|
601
|
-
§openrouter-app-attribution **
|
|
602
|
-
|
|
603
|
-
`
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
682
|
+
§openrouter-app-attribution **A route on the OpenRouter SDK identifies the calling
|
|
683
|
+
application only as its provider declares.** `APP_URL`, an absolute HTTP(S) URL, and
|
|
684
|
+
the optional `APP_NAME` are provider declarations (`PLURNK_PROVIDERS_PROVIDER_<NAME>_`,
|
|
685
|
+
overridable per route or alias); the SDK sends them as `HTTP-Referer` and
|
|
686
|
+
`X-OpenRouter-Title`. The shipped floor declares them for `openrouter` only, so
|
|
687
|
+
another provider on the same SDK package sends none; an empty `APP_URL` suppresses
|
|
688
|
+
both. The retired `OPENROUTER_HTTP_REFERER`, `OPENROUTER_APP_TITLE` and
|
|
689
|
+
`OPENROUTER_X_TITLE` fail construction.
|
|
609
690
|
|
|
610
691
|
## §9 Failures, retries, and cancellation
|
|
611
692
|
|
|
@@ -618,8 +699,15 @@ into an empty model turn or reduced to a message plus a generic status.
|
|
|
618
699
|
Upstream diagnostic text is bounded by
|
|
619
700
|
`PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT`; the committed `.env.defaults` owns its
|
|
620
701
|
normal value. Retry exhaustion is preserved as `attempts` and
|
|
621
|
-
`retryExhausted
|
|
622
|
-
|
|
702
|
+
`retryExhausted`; it does not change the Problem's `retryable`.
|
|
703
|
+
|
|
704
|
+
§provider-retryable-truth **`retryable` states what happens next.** A Problem's
|
|
705
|
+
`retryable` is exactly membership of its kind in the exported
|
|
706
|
+
`RETRYABLE_PROVIDER_KINDS` — `rate_limit`, `network_failure`, `deadline_exceeded`,
|
|
707
|
+
`resource_interrupted`, `output_dropped` — and Core's provider recovery re-issues
|
|
708
|
+
exactly those kinds by reading the same set. It never copies the transport's own
|
|
709
|
+
retry policy: a 502 or 524 that the transport will not replay is still re-issued by
|
|
710
|
+
the consumer, so it reports `retryable: true`. Every other kind reports `false`.
|
|
623
711
|
|
|
624
712
|
§provider-failure-cause **The wrapper's cause is evidence, bounded.** The SDK
|
|
625
713
|
reports every processing failure of a 2xx body with one message ("Failed to
|
|
@@ -660,20 +748,29 @@ and the notice names the true per-call grant (`capacity.responseMax` on the
|
|
|
660
748
|
response) — the tolerance's honest edge, never worse than the fixed allowance
|
|
661
749
|
it forgives.
|
|
662
750
|
|
|
663
|
-
§provider-sampling-passthrough **
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
|
|
676
|
-
|
|
751
|
+
§provider-sampling-passthrough **Unconfigured sampling retains endpoint defaults.**
|
|
752
|
+
The provider panel owns tuning; Models.dev capabilities are not recommended
|
|
753
|
+
sampling values. Every knob accepts the ordinary alias override
|
|
754
|
+
({§provider-configuration}). Both cataloged and local providers apply the same
|
|
755
|
+
validation and precedence: configured values, then caller `sampling`, then
|
|
756
|
+
transport-owned fields ({§provider-request-authority}).
|
|
757
|
+
|
|
758
|
+
| `PLURNK_PROVIDERS_` suffix | Compatible wire / native SDK | Accepted configuration |
|
|
759
|
+
| --- | --- | --- |
|
|
760
|
+
| `TEMPERATURE` | `temperature` / `temperature` | Finite non-negative number; empty/unset omits. |
|
|
761
|
+
| `TOP_P` | `top_p` / `topP` | Finite number in `[0,1]`; empty/unset omits. |
|
|
762
|
+
| `TOP_K` | `top_k` / `topK` | Non-negative safe integer; empty/unset omits. Zero semantics are endpoint-owned. |
|
|
763
|
+
| `PRESENCE_PENALTY` | `presence_penalty` / `presencePenalty` | Finite number in `[-2,2]`; empty/unset omits. |
|
|
764
|
+
| `FREQUENCY_PENALTY` | `frequency_penalty` / `frequencyPenalty` | Required finite number in `[-2,2]`; the panel's zero selects no configured override. Negative values survive unchanged. |
|
|
765
|
+
| `SEED` | `seed` / `seed` | Safe integer; empty/unset omits. No guarantee of reproducibility. |
|
|
766
|
+
|
|
767
|
+
Configured zero is preserved except for the frequency knob's explicit
|
|
768
|
+
no-override sentinel; a caller-supplied zero always overrides a configured value.
|
|
769
|
+
Invalid configuration fails by knob name before inference. The SDK/endpoint
|
|
770
|
+
owns narrower model restrictions and unsupported-setting diagnostics; Plurnk
|
|
771
|
+
does not clamp values or invent model-specific sampling profiles.
|
|
772
|
+
`REPEAT_PENALTY` and DRY remain explicit llama-server extensions, omitted when
|
|
773
|
+
unconfigured. Their grammar protection remains at {§provider-grammar-transport}.
|
|
677
774
|
|
|
678
775
|
§provider-connectivity The provider adapter owns one attempt scheduler around
|
|
679
776
|
the complete generation exchange; SDK-internal retries are disabled.
|
|
@@ -683,20 +780,29 @@ deadline:
|
|
|
683
780
|
|
|
684
781
|
| Layer | Operator knob | Boundary | Expiry |
|
|
685
782
|
| --- | --- | --- | --- |
|
|
686
|
-
| Operation | `PLURNK_PROVIDERS_OPERATION_TIMEOUT` | Complete logical call, including every attempt and retry delay. | Final `deadline_exceeded` Problem at 504 with `timeoutPhase=operation`;
|
|
687
|
-
| Attempt | `PLURNK_PROVIDERS_FETCH_TIMEOUT` | One physical generation request. A non-streamed request is bounded through response consumption; a streamed one only until its first semantic content
|
|
688
|
-
| First content | `PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT` |
|
|
783
|
+
| Operation | `PLURNK_PROVIDERS_OPERATION_TIMEOUT` | Complete logical call, including admission queueing, every attempt and retry delay. | Final `deadline_exceeded` Problem at 504 with `timeoutPhase=operation`; not retried inside this operation. Consumer recovery is separate. Enforced as a race, not only the advisory signal, so a wedged transport that never observes the abort cannot hang the loop past the deadline (#505); a well-behaved transport unwinds within a short grace and settles its own attempt evidence first. |
|
|
784
|
+
| Attempt | `PLURNK_PROVIDERS_FETCH_TIMEOUT` | One physical generation request. A non-streamed request is bounded through response consumption; a streamed one only until its first semantic content. After content begins, stream-idle bounds silence and the operation deadline still bounds the whole call. | Surfaced `network_failure` with `timeoutPhase=attempt`; never transport-retried (#479) — the consumer's recovery owns re-issue. |
|
|
785
|
+
| First content | `PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT` | Dispatch through first semantic model content ({§provider-first-content-at-dispatch}); metadata, empty deltas, and transport activity do not satisfy it. | Surfaced `network_failure` with `timeoutPhase=first_content`; never transport-retried (#479). |
|
|
689
786
|
| Stream idle | `PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT` | Silence between semantic content chunks after content begins. | Surfaced `network_failure` with `timeoutPhase=stream_idle`; never transport-retried (#479). |
|
|
787
|
+
| Repeated line | `PLURNK_PROVIDERS_REPEATED_LINE_LIMIT` | A streamed line of 16 or more characters repeated this many times; `0` disables. | Stopped as `repetition` ({§repetition-stop}); never transport-retried. |
|
|
788
|
+
|
|
789
|
+
§provider-first-content-at-dispatch The first-content deadline is armed when the
|
|
790
|
+
admitted attempt is dispatched, before response headers, so an endpoint that accepts
|
|
791
|
+
the request and never answers fails at `timeoutPhase=first_content` instead of holding
|
|
792
|
+
the attempt for the whole attempt deadline. It still begins only after admission
|
|
793
|
+
({§provider-inference-admission}).
|
|
794
|
+
|
|
795
|
+
Settled calls remove their deadline timers and cancellation subscriptions.
|
|
690
796
|
|
|
691
797
|
Caller cancellation spans the operation and preserves the caller's reason.
|
|
692
798
|
A 2xx exchange whose body cannot be processed (a provider invalid-response)
|
|
693
799
|
classifies as the non-retryable 502 on the first failure unless an explicit
|
|
694
|
-
`x-should-retry` directive says otherwise (#479
|
|
695
|
-
promotion). Inner deadline failures surface on the first failure; when a
|
|
800
|
+
`x-should-retry` directive says otherwise (#479). Inner deadline failures surface on the first failure; when a
|
|
696
801
|
directive-driven retry sequence exhausts, `attempts` and `retryExhausted`
|
|
697
|
-
are added and the classification is final. Every
|
|
698
|
-
one ordered {§provider-request-accounting} record, including
|
|
699
|
-
network failures and timed-out attempts.
|
|
802
|
+
are added and the classification is final. Every admitted physical attempt opens
|
|
803
|
+
and settles exactly one ordered {§provider-request-accounting} record, including
|
|
804
|
+
response-less network failures and timed-out attempts. A cancelled or expired
|
|
805
|
+
admission wait opens no physical request.
|
|
700
806
|
|
|
701
807
|
Each streamed physical request assembles its own response. Failed partial answer
|
|
702
808
|
bytes never enter a later request's completed `ProviderResponse`; recovery is a
|
|
@@ -715,6 +821,21 @@ ordinary 5xx surface on the first failure as consumer-recoverable kinds; one
|
|
|
715
821
|
retry authority — the consumer's own provider-recovery machinery — owns
|
|
716
822
|
re-issue, backoff, and park above the transport.
|
|
717
823
|
|
|
824
|
+
§provider-output-dropped **A completed exchange whose own evidence proves its
|
|
825
|
+
text never arrived is `output_dropped`**, a 502 carrying the complete response as
|
|
826
|
+
`error.attempt` and `stage: "provider-response"`; it is never admitted as an empty or
|
|
827
|
+
truncated turn, and the consumer re-issues it ({§provider-retryable-truth}):
|
|
828
|
+
|
|
829
|
+
| Evidence | Detail | Facts |
|
|
830
|
+
| --- | --- | --- |
|
|
831
|
+
| `finish_reason` `tool_calls` (Plurnk never declares tools) | `The provider ended the response with tool calls although no tools were declared (N native tool call(s)); the response text was not delivered as text.` | `toolCallCount` |
|
|
832
|
+
| Billed output tokens exceed the code points streamed across every channel — text and reasoning, before projection — by at least `PLURNK_PROVIDERS_DROPPED_OUTPUT_TOKENS` (`0` disables; an output token decodes to at least one character) | `The provider billed T output tokens but streamed C characters of text and reasoning; at least T−C tokens of output never arrived.` | `billedOutputTokens`, `streamedCharacters`, `droppedOutputTokens` |
|
|
833
|
+
|
|
834
|
+
The token rule needs a billed output count, and it does not apply when the response bills
|
|
835
|
+
reasoning tokens but streamed no reasoning in any channel: that is hidden or summarized
|
|
836
|
+
reasoning, which the characters cannot account for. Text-only comparisons are not made —
|
|
837
|
+
a route may bill reasoning inside its text count.
|
|
838
|
+
|
|
718
839
|
§provider-request-rejection An HTTP 4xx rejection not classified as authorization,
|
|
719
840
|
quota, capacity, grammar, rate limit, or transient transport failure is
|
|
720
841
|
`request_rejected`: preserve the upstream status and detail, with `retryable: false`.
|
|
@@ -733,7 +854,7 @@ completed exchange.
|
|
|
733
854
|
| Normalized finish reason | Exact `insufficient_system_resource` becomes `resource_interrupted`; the raw value remains in `assistantRaw`. |
|
|
734
855
|
| `generate` outcome | Throw `ProviderError(kind="resource_interrupted")` at local status 503 with the normalized attempt on `error.attempt` and the same accounting on `error.accounting`. |
|
|
735
856
|
| Partial response | Preserve content, reasoning, usage, model, metadata, optional raw body, and other evidence without admitting it as success. |
|
|
736
|
-
| Automatic replay | None
|
|
857
|
+
| Automatic replay | None inside the provider; AI SDK retry scheduling has already completed. The Problem has `retryable: true`: the consumer re-issues it ({§provider-retryable-truth}). |
|
|
737
858
|
| Capacity-pool overflow | None under the existing routing policy; when other overflow-eligible failures do reach a sibling, the pool concatenates their request accounting. |
|
|
738
859
|
| Consumer admission | Persist the evidence as an unaccepted attempt; never parse it into executable work, even when its frame looks complete. |
|
|
739
860
|
|
|
@@ -777,19 +898,26 @@ by index with their arguments as streamed, the finish reasons seen, and the verb
|
|
|
777
898
|
of every string field the record does not already hold. Nothing is opt-in and nothing is
|
|
778
899
|
reinterpreted: a blank emission reads as thirteen chunks that carried nothing, or as a
|
|
779
900
|
tool-call section the model emitted into a channel the protocol does not read, instead of
|
|
780
|
-
being guessed at from token counts.
|
|
901
|
+
being guessed at from token counts. Frames outside the recognized OpenAI-compatible
|
|
902
|
+
envelope remain verbatim in `wire.unmappedChunks`; the recorder never discards an
|
|
903
|
+
unrecognized/native payload or invents another provider parser. This required
|
|
904
|
+
evidence is independent of the optional complete `rawBody` dataset capture.
|
|
905
|
+
Covered by `aiSdkTransport.test.ts` and core's `Digest.wire-evidence.test.ts`.
|
|
781
906
|
|
|
782
907
|
§provider-usage-refusal **The provider's bookkeeping is not the exchange.** Usage
|
|
783
|
-
normalization is exact
|
|
784
|
-
|
|
785
|
-
|
|
786
|
-
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
|
|
790
|
-
|
|
791
|
-
|
|
792
|
-
the
|
|
908
|
+
normalization is exact; a refusal never fails or retries the response. The same
|
|
909
|
+
rules apply to usage inside failure evidence:
|
|
910
|
+
|
|
911
|
+
| Contradiction | Normalized evidence |
|
|
912
|
+
| --- | --- |
|
|
913
|
+
| Invalid aggregate or total inconsistent with its parts | No `usage`. |
|
|
914
|
+
| Invalid optional counter | Omit that counter; retain independently valid quantities. |
|
|
915
|
+
| Cache or reasoning breakdown inconsistent with its aggregate | Omit that breakdown; retain valid aggregates and the other breakdown. |
|
|
916
|
+
|
|
917
|
+
Every refusal retains `usageRefusal`: the reason and the original wire counters.
|
|
918
|
+
No value is clamped, zeroed, or estimated. Cost uses authoritative charges when
|
|
919
|
+
available; catalog estimation requires all quantities its rates need, otherwise
|
|
920
|
+
cost is `unknown`. Covered by `usage.test.ts`, `aiSdkTransport.test.ts`, and
|
|
793
921
|
`AiSdkProvider.test.ts`.
|
|
794
922
|
|
|
795
923
|
§provider-open-reasoning **Plurnk reads the model's own reasoning.** The service is
|
|
@@ -812,8 +940,9 @@ budget is a strict subset of that total, never an additive reserve. The
|
|
|
812
940
|
configured total is a percentage of effective context or an absolute count;
|
|
813
941
|
percentages resolve to the nearest whole token with a one-token minimum. It is
|
|
814
942
|
capped by known context and model-output limits; `generate.maxOutputTokens` may
|
|
815
|
-
only tighten
|
|
816
|
-
total and remains strictly smaller.
|
|
943
|
+
only tighten this reservation for one call. The effective reasoning subset tightens
|
|
944
|
+
with that total and remains strictly smaller. Exact prompt measurements may expand
|
|
945
|
+
the response grant beyond the reservation under {§provider-flexed-allowance}.
|
|
817
946
|
|
|
818
947
|
The adapter owns native projection. A backend whose generic SDK maximum already
|
|
819
948
|
includes reasoning receives the total directly. When a native SDK instead adds
|
|
@@ -827,7 +956,7 @@ and provider minimum; an envelope too small to represent the minimum fails
|
|
|
827
956
|
before provider I/O.
|
|
828
957
|
|
|
829
958
|
§provider-output-budget-conformance When a completed response reports
|
|
830
|
-
normalized output-token usage greater than its
|
|
959
|
+
normalized output-token usage greater than its response grant ({§provider-flexed-allowance}),
|
|
831
960
|
the exchange is an `invalid_response` at 502 rather than an admitted result or
|
|
832
961
|
a prompt-capacity 413. Its complete failed-attempt evidence and settled charged
|
|
833
962
|
request remain available. The violation is final and is never automatically
|
|
@@ -840,7 +969,29 @@ limit advertises `requiresOutputBudget` and fails construction when no total can
|
|
|
840
969
|
be resolved. The retired additive reserve knobs fail hard rather than creating
|
|
841
970
|
a second envelope contract.
|
|
842
971
|
|
|
843
|
-
## §13
|
|
972
|
+
## §13 Inference capacity
|
|
973
|
+
|
|
974
|
+
### Admission
|
|
975
|
+
|
|
976
|
+
§provider-inference-admission `PLURNK_PROVIDERS_MAX_CONCURRENCY` controls physical
|
|
977
|
+
generation attempts per resolved endpoint within one process: `-1` is unrestricted;
|
|
978
|
+
a positive safe integer admits that many concurrent attempts. It follows ordinary
|
|
979
|
+
alias scoping. Zero, other negative values and non-integers are invalid.
|
|
980
|
+
|
|
981
|
+
| Boundary | Contract |
|
|
982
|
+
| --- | --- |
|
|
983
|
+
| Identity | Provider instances and aliases sharing the resolved API base URL share one allowance. SDK/plugin-owned endpoints without a resolved URL share their provider identity. Conflicting limits for one identity fail construction, naming the identity and both values; they never create independent queues. The daemon reads its environment once at boot, so an identity's limit cannot change within a process: reconstructing a provider from that environment reuses its allowance, and in-flight leases keep counting. |
|
|
984
|
+
| Admission | FIFO among live waiters. The lease begins before the physical request observer and ends after the complete response or transport failure settles, including streamed bodies. |
|
|
985
|
+
| Cancellation | A queued abort removes that waiter and preserves the caller's reason. It opens no physical request or accounting row. In-flight cancellation signals the transport; capacity is released when that attempt unwinds. A transport still running despite abort does not authorize exceeding the limit. |
|
|
986
|
+
| Retries | Backoff holds no lease. Each retry rejoins admission as a new physical attempt. |
|
|
987
|
+
| Deadlines | The existing operation deadline includes queueing. Attempt, first-content and stream-idle deadlines begin only after admission; queued work is not a stalled stream. |
|
|
988
|
+
| Scope | Workers, tools, messages, waits, token measurement and endpoint discovery are not serialized by inference admission. Independent endpoints progress independently. No cross-process or machine-wide capacity guarantee is implied. |
|
|
989
|
+
|
|
990
|
+
Admission neither changes worker lifecycle nor selects another model or endpoint.
|
|
991
|
+
Core consumes the same asynchronous Provider contract. A waiting worker holds no
|
|
992
|
+
inference lease; BARE and ordinary generation use the same physical boundary.
|
|
993
|
+
|
|
994
|
+
### Pool
|
|
844
995
|
|
|
845
996
|
§provider-capacity-pool `Pool` fronts interchangeable `Provider` instances. It
|
|
846
997
|
keeps workers sticky for
|
|
@@ -868,6 +1019,8 @@ Coverage MUST prove:
|
|
|
868
1019
|
- native SDK request mapping and normalized responses;
|
|
869
1020
|
- compatible extension preservation;
|
|
870
1021
|
- timeout, retry, cancellation, interrupted-attempt, and final-error behavior;
|
|
1022
|
+
- shared endpoint inference admission, FIFO queueing, queued/in-flight cancellation,
|
|
1023
|
+
full-stream lease lifetime, retry release, and independent endpoint progress;
|
|
871
1024
|
- local capability probes and pins;
|
|
872
1025
|
- exact, bounded, estimated, and unavailable complete-request measurements;
|
|
873
1026
|
- independent input/context/output limits, asymmetric admission, and normalized
|
package/dist/AiSdkProvider.d.ts
CHANGED
|
@@ -1,14 +1,15 @@
|
|
|
1
|
-
import type { ChatMessage, PromptTokenMeasurement, Provider, ProviderCostNormalizer, ProviderGenerateArgs, ProviderRequestCapacity, ProviderResponse, ProviderUsage,
|
|
1
|
+
import type { ChatMessage, PromptTokenMeasurement, Provider, ProviderCostNormalizer, ProviderGenerateArgs, ProviderRequestCapacity, ProviderResponse, ProviderUsage, Effort } from "./types.ts";
|
|
2
2
|
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
3
3
|
import type { JSONValue } from "ai";
|
|
4
|
-
import {
|
|
4
|
+
import type { EffortSetting, ReasoningResponseStyle } from "./env.ts";
|
|
5
5
|
import type { InputModality } from "./types.ts";
|
|
6
6
|
import type { LanguageModel } from "ai";
|
|
7
7
|
import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
8
|
+
import type RequestFields from "./RequestFields.ts";
|
|
9
|
+
import type InferenceAdmission from "./InferenceAdmission.ts";
|
|
8
10
|
export type ProviderFetch = typeof globalThis.fetch;
|
|
9
|
-
export type ReasoningStyle = "none" | "think" | "
|
|
10
|
-
export type
|
|
11
|
-
export type CompatibleReasoningEffort = NativeReasoningEffort | "max";
|
|
11
|
+
export type ReasoningStyle = "none" | "think" | "template";
|
|
12
|
+
export type NativeEffort = "minimal" | "low" | "medium" | "high" | "xhigh";
|
|
12
13
|
export type GrammarStyle = "none" | "llamacpp";
|
|
13
14
|
export type CacheAffinity = {
|
|
14
15
|
readonly target: "header" | "body";
|
|
@@ -20,6 +21,7 @@ export type CacheAffinity = {
|
|
|
20
21
|
};
|
|
21
22
|
export type AiSdkProviderOptions = Record<string, Record<string, JSONValue | undefined>>;
|
|
22
23
|
export type AiSdkProviderConfig = {
|
|
24
|
+
inferenceAdmission?: InferenceAdmission;
|
|
23
25
|
model: string;
|
|
24
26
|
url?: string;
|
|
25
27
|
languageModel?: LanguageModel;
|
|
@@ -28,6 +30,8 @@ export type AiSdkProviderConfig = {
|
|
|
28
30
|
operationTimeoutMs: number;
|
|
29
31
|
firstContentTimeoutMs: number;
|
|
30
32
|
streamIdleTimeoutMs?: number;
|
|
33
|
+
repeatedLineLimit?: number;
|
|
34
|
+
droppedOutputTokens?: number;
|
|
31
35
|
headers?: Record<string, string>;
|
|
32
36
|
fetch?: ProviderFetch;
|
|
33
37
|
contextWindow?: number | null;
|
|
@@ -36,14 +40,12 @@ export type AiSdkProviderConfig = {
|
|
|
36
40
|
maxOutputTokens?: number | null;
|
|
37
41
|
outputBudget?: number | null;
|
|
38
42
|
reasoningBudget?: number | null;
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
compatibleAdaptiveReasoning?: CompatibleReasoningEffort | "provider-default";
|
|
43
|
-
compatibleOffReasoning?: "none";
|
|
44
|
-
adaptiveReasoningProviderOptions?: AiSdkProviderOptions;
|
|
43
|
+
supportedEfforts?: readonly Effort[];
|
|
44
|
+
adaptiveEffort?: NativeEffort | "provider-default";
|
|
45
|
+
adaptiveEffortProviderOptions?: AiSdkProviderOptions;
|
|
45
46
|
additiveReasoningProvider?: "anthropic" | "bedrock";
|
|
46
47
|
reasoningStyle?: ReasoningStyle;
|
|
48
|
+
requestFields?: RequestFields;
|
|
47
49
|
reasoningResponseStyle?: ReasoningResponseStyle;
|
|
48
50
|
countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
|
|
49
51
|
estimateCost?: (usage: ProviderUsage | undefined) => ProviderCost;
|
|
@@ -63,8 +65,12 @@ export type AiSdkProviderConfig = {
|
|
|
63
65
|
promptTokensUrl?: string;
|
|
64
66
|
servedModel?: string;
|
|
65
67
|
requiresOutputBudget?: boolean;
|
|
66
|
-
|
|
68
|
+
effort: EffortSetting;
|
|
67
69
|
temperature: number | null;
|
|
70
|
+
topP?: number;
|
|
71
|
+
topK?: number;
|
|
72
|
+
presencePenalty?: number;
|
|
73
|
+
seed?: number;
|
|
68
74
|
repeatPenalty: number | null;
|
|
69
75
|
frequencyPenalty?: number;
|
|
70
76
|
dryMultiplier?: number;
|
|
@@ -87,7 +93,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
87
93
|
get maxOutputTokens(): number | null;
|
|
88
94
|
get outputBudget(): number | null;
|
|
89
95
|
get reasoningBudget(): number | null;
|
|
90
|
-
get
|
|
96
|
+
get supportedEfforts(): readonly Effort[];
|
|
91
97
|
get inputCapacity(): number | null;
|
|
92
98
|
get model(): string;
|
|
93
99
|
get servedModel(): string | undefined;
|