@plurnk/plurnk-providers 1.21.0 → 1.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (144) hide show
  1. package/.env.defaults +110 -24
  2. package/README.md +19 -6
  3. package/SPEC.md +250 -97
  4. package/dist/AiSdkProvider.d.ts +19 -13
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +154 -73
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/AiSdkRequestBody.d.ts +7 -11
  9. package/dist/AiSdkRequestBody.d.ts.map +1 -1
  10. package/dist/AiSdkRequestBody.js +18 -90
  11. package/dist/AiSdkRequestBody.js.map +1 -1
  12. package/dist/InferenceAdmission.d.ts +7 -0
  13. package/dist/InferenceAdmission.d.ts.map +1 -0
  14. package/dist/InferenceAdmission.js +58 -0
  15. package/dist/InferenceAdmission.js.map +1 -0
  16. package/dist/Mock.d.ts +1 -1
  17. package/dist/Mock.d.ts.map +1 -1
  18. package/dist/Mock.js +2 -2
  19. package/dist/Mock.js.map +1 -1
  20. package/dist/Pool.d.ts +2 -2
  21. package/dist/Pool.d.ts.map +1 -1
  22. package/dist/Pool.js +2 -2
  23. package/dist/Pool.js.map +1 -1
  24. package/dist/RepeatedLine.d.ts +9 -0
  25. package/dist/RepeatedLine.d.ts.map +1 -0
  26. package/dist/RepeatedLine.js +27 -0
  27. package/dist/RepeatedLine.js.map +1 -0
  28. package/dist/RequestFields.d.ts +25 -0
  29. package/dist/RequestFields.d.ts.map +1 -0
  30. package/dist/RequestFields.js +240 -0
  31. package/dist/RequestFields.js.map +1 -0
  32. package/dist/accounting.d.ts.map +1 -1
  33. package/dist/accounting.js +38 -7
  34. package/dist/accounting.js.map +1 -1
  35. package/dist/aiSdkTransport.d.ts +8 -3
  36. package/dist/aiSdkTransport.d.ts.map +1 -1
  37. package/dist/aiSdkTransport.js +77 -24
  38. package/dist/aiSdkTransport.js.map +1 -1
  39. package/dist/catalogProvider.d.ts +4 -3
  40. package/dist/catalogProvider.d.ts.map +1 -1
  41. package/dist/catalogProvider.js +73 -170
  42. package/dist/catalogProvider.js.map +1 -1
  43. package/dist/compatibleProvider.d.ts.map +1 -1
  44. package/dist/compatibleProvider.js +11 -9
  45. package/dist/compatibleProvider.js.map +1 -1
  46. package/dist/discover.d.ts +0 -1
  47. package/dist/discover.d.ts.map +1 -1
  48. package/dist/discover.js +1 -10
  49. package/dist/discover.js.map +1 -1
  50. package/dist/env.d.ts +19 -8
  51. package/dist/env.d.ts.map +1 -1
  52. package/dist/env.js +125 -23
  53. package/dist/env.js.map +1 -1
  54. package/dist/errors.d.ts +1 -8
  55. package/dist/errors.d.ts.map +1 -1
  56. package/dist/errors.js +9 -13
  57. package/dist/errors.js.map +1 -1
  58. package/dist/index.d.ts +8 -7
  59. package/dist/index.d.ts.map +1 -1
  60. package/dist/index.js +6 -5
  61. package/dist/index.js.map +1 -1
  62. package/dist/model-options.d.ts +3 -0
  63. package/dist/model-options.d.ts.map +1 -0
  64. package/dist/model-options.js +26 -0
  65. package/dist/model-options.js.map +1 -0
  66. package/dist/ollama.d.ts.map +1 -1
  67. package/dist/ollama.js +1 -0
  68. package/dist/ollama.js.map +1 -1
  69. package/dist/provider-env.d.ts +3 -0
  70. package/dist/provider-env.d.ts.map +1 -0
  71. package/dist/provider-env.js +8 -0
  72. package/dist/provider-env.js.map +1 -0
  73. package/dist/providerError.d.ts +2 -2
  74. package/dist/providerError.d.ts.map +1 -1
  75. package/dist/providerError.js +19 -22
  76. package/dist/providerError.js.map +1 -1
  77. package/dist/reasoning-effort.d.ts +4 -3
  78. package/dist/reasoning-effort.d.ts.map +1 -1
  79. package/dist/reasoning-effort.js +18 -2
  80. package/dist/reasoning-effort.js.map +1 -1
  81. package/dist/sdkModels.d.ts +3 -2
  82. package/dist/sdkModels.d.ts.map +1 -1
  83. package/dist/sdkModels.js +39 -66
  84. package/dist/sdkModels.js.map +1 -1
  85. package/dist/types.d.ts +8 -8
  86. package/dist/types.d.ts.map +1 -1
  87. package/dist/types.js +5 -5
  88. package/dist/types.js.map +1 -1
  89. package/dist/usage.d.ts +5 -1
  90. package/dist/usage.d.ts.map +1 -1
  91. package/dist/usage.js +62 -17
  92. package/dist/usage.js.map +1 -1
  93. package/package.json +6 -7
  94. package/src/AiSdkProvider.test.ts +0 -2841
  95. package/src/AiSdkProvider.ts +0 -1064
  96. package/src/AiSdkRequestBody.ts +0 -364
  97. package/src/LeadingReasoning.test.ts +0 -34
  98. package/src/LeadingReasoning.ts +0 -87
  99. package/src/Mock.test.ts +0 -209
  100. package/src/Mock.ts +0 -197
  101. package/src/Pool.test.ts +0 -267
  102. package/src/Pool.ts +0 -261
  103. package/src/ProviderRegistry.test.ts +0 -483
  104. package/src/ProviderRegistry.ts +0 -128
  105. package/src/accounting.test.ts +0 -139
  106. package/src/accounting.ts +0 -172
  107. package/src/accountingPublic.ts +0 -9
  108. package/src/aiSdkTransport.test.ts +0 -476
  109. package/src/aiSdkTransport.ts +0 -776
  110. package/src/boundaries.test.ts +0 -73
  111. package/src/capacity.test.ts +0 -124
  112. package/src/capacity.ts +0 -174
  113. package/src/catalogProvider.test.ts +0 -900
  114. package/src/catalogProvider.ts +0 -476
  115. package/src/compatibleProvider.test.ts +0 -157
  116. package/src/compatibleProvider.ts +0 -217
  117. package/src/cost.test.ts +0 -114
  118. package/src/cost.ts +0 -140
  119. package/src/defaults.test.ts +0 -30
  120. package/src/defaults.ts +0 -9
  121. package/src/discover.test.ts +0 -200
  122. package/src/discover.ts +0 -136
  123. package/src/env.test.ts +0 -335
  124. package/src/env.ts +0 -380
  125. package/src/errors.test.ts +0 -278
  126. package/src/errors.ts +0 -234
  127. package/src/index.ts +0 -100
  128. package/src/inputModalities.test.ts +0 -54
  129. package/src/lexicon-guard.test.ts +0 -58
  130. package/src/notices.ts +0 -22
  131. package/src/ollama.test.ts +0 -66
  132. package/src/ollama.ts +0 -66
  133. package/src/openai.ts +0 -15
  134. package/src/promptTokens.ts +0 -45
  135. package/src/providerDefaults.test.ts +0 -51
  136. package/src/providerError.ts +0 -143
  137. package/src/reasoning-effort.ts +0 -15
  138. package/src/sdkModels.test.ts +0 -293
  139. package/src/sdkModels.ts +0 -460
  140. package/src/types.ts +0 -337
  141. package/src/usage.test.ts +0 -149
  142. package/src/usage.ts +0 -266
  143. package/src/warnings.test.ts +0 -31
  144. package/src/warnings.ts +0 -0
package/SPEC.md CHANGED
@@ -125,6 +125,23 @@ produces an ordinary estimated amount of USD `0`. Rate calculation is internal
125
125
  to the provider request; Core, digest, ping, and clients never call a parallel
126
126
  pricing method.
127
127
 
128
+ Router accounting uses the complete cost to the caller, not just the router's
129
+ fee. OpenRouter's response `usage` is authoritative; its SDK metadata projection
130
+ omits the BYOK discriminator. Apply the same rule to successful and failed
131
+ requests, independently of raw-body capture:
132
+
133
+ | OpenRouter evidence | Monetary result |
134
+ | --- | --- |
135
+ | `is_byok: false` and `cost` | `charged`: `cost`; upstream inference is already included. |
136
+ | `is_byok: true`, `cost`, and `cost_details.upstream_inference_cost` | `charged`: exact decimal sum of router fee and upstream charge. |
137
+ | A monetary field without a BYOK discriminator, or BYOK missing either component | `unknown`; neither zero nor catalog pricing completes the reported charge. |
138
+ | No reported charge and no BYOK indication | Ordinary Models.dev fallback above. |
139
+
140
+ Explicit zero components are valid. Malformed monetary values are contract
141
+ violations, never coerced. Raw routing and billing evidence remain available
142
+ under {§provider-evidence}; Core and clients consume the existing single cost.
143
+ See [OpenRouter usage accounting](https://openrouter.ai/docs/cookbook/administration/usage-accounting).
144
+
128
145
  ### Generation
129
146
 
130
147
  §provider-cache-identity `generate` requires a non-empty, stable, opaque
@@ -231,14 +248,21 @@ the SDK. PLURNK supplies cancellation and deadline signals and owns the sole
231
248
  cross-attempt scheduler so every physical request remains observable and
232
249
  accountable.
233
250
 
251
+ | Usage input | Normalization |
252
+ | --- | --- |
253
+ | Native protocol | Preserve the SDK's inclusive totals and category counts; raw protocol counters are evidence, not interchangeable totals. |
254
+ | OpenAI-compatible wire | Retain extended counters and exact total identities the SDK may omit. |
255
+ | Compatible SDK input partition | Retain its uncached count when total input and an explicitly reported cache-read count agree with the wire; never promote SDK defaults over absent or contradictory wire counts. |
256
+ | Input total plus two partition counts | Derive the remaining count exactly; otherwise preserve absence. |
257
+
234
258
  PLURNK maps its generic settings to AI SDK call settings:
235
259
 
236
260
  - `temperature`, `top_p`, `top_k`;
237
261
  - presence and frequency penalties;
238
262
  - stop sequences and seed;
239
263
  - output-token ceiling;
240
- - `off`, `adaptive`, or fixed `low`, `medium`, or `high` reasoning policy, with
241
- an independent optional operator budget.
264
+ - the supported efforts and optional numeric control under
265
+ {§provider-effort}.
242
266
 
243
267
  Provider-specific options are permitted only where they preserve a documented
244
268
  PLURNK product contract the generic SDK surface cannot express.
@@ -253,28 +277,31 @@ catalog figure. Without catalog rates the override must declare `input` and
253
277
  response cost still outranks any estimate. Unknown keys, repeats, and negative
254
278
  or non-numeric rates refuse at construction.
255
279
 
256
- §provider-reasoning-policy The portable vocabulary comes from
257
- {§reasoning-policy-wire}. `adaptive` requests the provider's
258
- documented dynamic mechanism where one exists. Otherwise, a cataloged graded
259
- route receives its strongest positive Models.dev effort that the installed
260
- transport can represent; a route that declares a toggle control receives an
261
- explicit enable on transports that document one (OpenRouter `reasoning.enabled`,
262
- Fireworks Boolean `reasoning_effort`) — an unconfigured reasoning-capable route
263
- must not run reasoning-off, and `off` stays the explicit opt-out; otherwise the
264
- provider default holds. A fixed policy retains its exact name and is rejected before
265
- provider I/O when either the route or transport cannot represent it. Every
266
- provider exposes that exact intersection. A numeric reasoning budget constrains
280
+ §provider-effort The portable vocabulary comes from
281
+ {§effort-wire}. `adaptive` uses the first applicable projection:
282
+
283
+ | Condition | Projection |
284
+ | --- | --- |
285
+ | Explicit adaptive declaration: `REASONING_ADAPTIVE_BODY`, or a native model family's `ADAPTIVE_OPTIONS` ({§provider-model-options}) | Preserve that mechanism; an explicit `{}` retains the enabled endpoint's default. |
286
+ | The configured `PLURNK_PROVIDERS_EFFORT_FALLBACK` is supported by both model and transport | Send that exact effort. The shipped fallback is `high`, not the strongest available level. |
287
+ | No supported fallback, including an empty fallback setting | Retain reasoning activation and the provider's default effort; never invent a level or escalate to another one. |
288
+
289
+ Activation is distinct from effort: declared enable fields accompany active
290
+ reasoning, and a toggle-only route uses its documented enable mechanism.
291
+ Non-reasoning models receive no reasoning controls; `off` is the explicit opt-out.
292
+ A fixed policy retains its exact name and is rejected before provider I/O when
293
+ either the route or transport cannot represent it. Every provider exposes that
294
+ exact intersection. A numeric reasoning budget constrains
267
295
  the generation envelope independently and never selects or changes policy. On
268
- the OpenRouter transport a resolved budget under `adaptive` reaches the wire as
269
- the `reasoning.max_tokens` form — the only dial a budget_tokens-only route
270
- understands; a fixed policy keeps the `effort` form, and `off` suppresses the
271
- budget.
272
-
273
- `catalogReasoningPolicies` projects that same admission calculation from catalog
274
- facts and provider-wide environment declarations without constructing a model,
275
- requiring credentials, or performing provider I/O.
296
+ routes whose controls are exclusive, fixed effort plus a numeric budget is
297
+ rejected before I/O. Under `adaptive`, an explicit budget selects the numeric
298
+ control; `off` suppresses it.
299
+
300
+ `catalogEfforts` projects that same admission calculation from catalog
301
+ facts and provider-wide environment declarations over the installed defaults,
302
+ without constructing a model, requiring credentials, or performing provider I/O.
276
303
  Admission respects the installed projection: native SDK fixed efforts exclude
277
- `max`; the OpenRouter model-settings projection also excludes `xhigh`.
304
+ `max`; declared transport vocabularies constrain per-call option projections.
278
305
 
279
306
  Models.dev's route-specific `reasoning_options` is the capability authority for
280
307
  what the daemon offers on its own; the daemon never adds vendor behaviour absent
@@ -291,7 +318,7 @@ explicit compatible adapter owns the wire projection:
291
318
  | --- | --- | --- |
292
319
  | `reasoning: false` | No reasoning request | `off` |
293
320
  | `reasoning_options: []` | Provider default | None |
294
- | `effort.values` | Native dynamic mechanism, otherwise strongest transportable positive value | Transportable fixed members of {§reasoning-policy-wire}; `off` only when `none` is transportable |
321
+ | `effort.values` | Native dynamic mechanism, otherwise the supported configured fallback or provider default | Transportable fixed members of {§effort-wire}; `off` only when `none` is transportable |
295
322
  | `toggle` | Native or explicitly declared activation, otherwise provider default | `off` only when that transport owns the toggle wire |
296
323
  | `budget_tokens` | Does not select policy | None; an adapter may use its bounds when projecting the independent budget |
297
324
  | No catalog entry | Explicit adapter declaration | Only the declaration's exact subset |
@@ -321,34 +348,36 @@ provider's documented header, body field, or native SDK option. The common
321
348
  transport neither guesses from protocol resemblance nor sends a generic cache
322
349
  field to an unknown provider. The operator may disable affinity globally or per
323
350
  alias; automatic provider caching without an affinity control remains untouched.
351
+ The environment's `CACHE_AFFINITY_FIELD` declares that placement as
352
+ `{"target":"header"|"body","name":"…"}` or
353
+ `{"target":"provider-option","provider":"…","name":"…"}`. It follows
354
+ the provider/route/alias precedence of {§provider-wire-declaration}; `null`
355
+ clears the declaration. Placement is independent of the enable switch, cannot
356
+ replace transport-owned fields, and does not imply support from a provider's name.
357
+
358
+ §provider-model-options **A model family's native options are a provider declaration.** Models.dev
359
+ does not say which models on a native SDK take adaptive reasoning or explicit cache writes, and the SDKs
360
+ do not export their tables, so a provider declares them as data: `ADAPTIVE_OPTIONS` and
361
+ `SYSTEM_CACHE_OPTIONS` are JSON arrays of `{"models":["<glob>",…],"options":{…}}` rules. The first rule
362
+ with a glob matching the route's model id (`path.matchesGlob`) supplies its per-call provider options;
363
+ without a match none apply. They follow the provider/route/alias precedence of
364
+ {§provider-wire-declaration}, an empty value declares none, and a malformed value fails construction.
365
+ None ships: Plurnk's shipped defaults serve open models only, and an operator who routes a closed model
366
+ family declares its options in their own configuration.
324
367
 
325
368
  §provider-cache-write-policy **Cache-write policy is separate from affinity.**
326
369
  `PLURNK_PROVIDERS_CACHE_WRITE_POLICY` is `off` or `stable-system`. The latter
327
370
  marks only the final leading system instruction as an explicit reusable cache
328
- boundary, and only on routes whose native SDK documents that control. It does
371
+ boundary, and only on routes whose provider declares the control in
372
+ `SYSTEM_CACHE_OPTIONS` ({§provider-model-options}). It does
329
373
  not mark the changing user packet or enable an API-wide automatic cache mode.
330
374
  Unsupported routes receive no invented option. The default five-minute
331
375
  provider lifetime is used; a longer, differently priced lifetime is not an
332
376
  implicit transport choice.
333
377
 
334
- §deepseek-reasoning-request The direct DeepSeek catalog path maps the common
335
- reasoning intent to its OpenAI-compatible controls:
336
-
337
- | PLURNK posture | `thinking` | `reasoning_effort` |
338
- | --------------- | --------------------- | -------------------- |
339
- | `off` | `{ type: disabled }` | omitted |
340
- | `adaptive` | `{ type: enabled }` | omitted |
341
- | `high` | `{ type: enabled }` | `high` |
342
-
343
- The direct API does not distinguish portable `low` or `medium` intent and
344
- therefore advertises only `off`, `adaptive`, and `high`.
345
-
346
- §provider-reasoning-style A reasoning style names the wire a compatible route's
347
- reasoning controls take. `PLURNK_PROVIDERS_PROVIDER_<PREFIX>_REASONING_STYLE`
348
- declares it for every route of a provider; `PLURNK_PROVIDERS_REASONING_STYLE`,
349
- alias-scopable like every bare knob, declares it for one route and wins. One
350
- provider can serve models whose controls differ, so a route states its own style. The daemon never infers a
351
- style from a model name.
378
+ §deepseek-reasoning-request Direct DeepSeek uses the panel's request-field
379
+ declarations under {§provider-wire-declaration}. Models.dev supplies each
380
+ model's effort vocabulary; it is not a second fixed list in this adapter.
352
381
 
353
382
  The compatible transport is deliberately retained for:
354
383
 
@@ -359,6 +388,49 @@ The compatible transport is deliberately retained for:
359
388
  It carries PLURNK-only fields and raw wire evidence without reimplementing the
360
389
  SDK's ordinary transport.
361
390
 
391
+ ### §provider-wire-declaration Request-field declarations
392
+
393
+ Models.dev owns model capabilities, effort vocabulary, and numeric budget bounds.
394
+ For endpoints whose field names it does not describe, the provider's
395
+ environment panel supplies a bounded projection, independent of provider identity:
396
+
397
+ | Declaration suffix | Meaning |
398
+ | --- | --- |
399
+ | `OPTIONS_NAMESPACE` | Native SDK's per-call `providerOptions` namespace. Without one, declared fields target the compatible request body. |
400
+ | `OUTPUT_PATH` | RFC 6901 object-member pointer for the **inclusive** output ceiling; absent uses the compatible SDK's `max_tokens` field. |
401
+ | `REASONING_EFFORT_PATH` / `REASONING_BUDGET_PATH` | Pointers for exact effort or numeric reasoning subset. No pointer means no such control. |
402
+ | `REASONING_EFFORTS` | Additional declared effort values, unioned with Models.dev. |
403
+ | `REASONING_TRANSPORT_EFFORTS` | Optional transport vocabulary, intersected with catalog/declaration efforts. It adds no model capability. |
404
+ | `REASONING_CONTROLS` | Required when both pointers exist: `exclusive` refuses fixed effort plus budget; `combined` sends both. |
405
+ | `REASONING_ON_BODY` | Static reasoning activation fields, merged with the selected control. |
406
+ | `REASONING_OFF_BODY` | Explicit disable fields; otherwise a declared `none` effort can disable. |
407
+ | `REASONING_ADAPTIVE_BODY` | Explicit adaptive fields, ahead of graded fallback; `{}` retains the enabled endpoint's default. |
408
+ | `REASONING_TOGGLE_BODY` | Reasoning enable fields when Models.dev declares a toggle and neither an explicit adaptive body nor a supported fallback effort applies. |
409
+
410
+ The existing provider declaration prefix is
411
+ `PLURNK_PROVIDERS_PROVIDER_<NAME>_`; `PLURNK_PROVIDERS_<suffix>` overrides
412
+ it for the route and accepts the ordinary alias suffix. These are data, not
413
+ executable transformations. Static bodies cannot replace transport, sampling,
414
+ or managed numeric controls; pointers cannot overlap. Configuration fails at
415
+ construction before inference when a requested policy or numeric control is
416
+ unrepresentable. A tighter per-call envelope reprojects the same declaration.
417
+ Discovery and generation share the resolved policy set. A cataloged
418
+ non-reasoning model receives no reasoning fields.
419
+ Native SDKs retain their own output field; `OUTPUT_PATH` is incompatible with an
420
+ options namespace. Request-local options are projected after the effective
421
+ envelope is known, never frozen into SDK model construction. Streaming and
422
+ non-streaming calls use the same projection.
423
+ Retired `REASONING_STYLE` selectors fail at the selected provider/alias boundary;
424
+ they neither select a preset nor silently coexist with these declarations.
425
+
426
+ A native SDK's portable reasoning setting has no `max`, so a native route without
427
+ a namespace cannot send an effort the catalog documents beyond it; construction
428
+ refuses it and names the declaration that would. Native SDKs built on
429
+ openai-compatible (DeepInfra, Together) write `reasoning_effort` from their own
430
+ `reasoningEffort` option after spreading the others, overwriting a raw
431
+ `reasoning_effort`; their shipped declarations therefore point at
432
+ `/reasoningEffort`, with `{"reasoningEffort":"none"}` as the off body.
433
+
362
434
  ## §4 Operator configuration
363
435
 
364
436
  §provider-configuration Every operational value is an environment knob
@@ -434,8 +506,8 @@ Provider and model facts resolve independently:
434
506
  | Context window | Catalog metadata or local endpoint probe. | `PLURNK_PROVIDERS_CONTEXT_WINDOW`. | Minimum when both exist; sole value otherwise. Cataloged cloud miss fails construction; compatible probe miss remains `null` with one warning. |
435
507
  | Maximum input | Catalog `limit.input`; no generic live probe. | None. | Catalog value or `null`; never reconstructed from context and output. |
436
508
  | Maximum output | Catalog `limit.output`; no generic live probe. | None. | Minimum of catalog value and effective context, or `null`. |
437
- | Total output budget | None. | `PLURNK_PROVIDERS_OUTPUT_BUDGET`. | Percentage of effective context or absolute count, capped by known context/output limits; a call may only tighten it. |
438
- | Reasoning policy | Catalog `reasoning_options` intersected with the installed adapter; explicit adapter declaration for uncataloged routes. | `PLURNK_PROVIDERS_REASONING`, initially; durable worker selection thereafter. | One supported member of `off`, `adaptive`, `low`, `medium`, or `high`; `adaptive` is the default. |
509
+ | Total output budget | None. | `PLURNK_PROVIDERS_OUTPUT_BUDGET`. | Curation reservation: percentage of effective context or absolute count, capped by known context/output limits; a call may only tighten it. The response grant may expand under {§provider-flexed-allowance}. |
510
+ | Effort | Catalog `reasoning_options` intersected with the installed adapter; explicit adapter declaration for uncataloged routes. | `PLURNK_PROVIDERS_EFFORT`, initially; durable worker selection thereafter. | A supported member of {§effort-wire}, projected under {§provider-effort}. The shipped selection is `adaptive`. |
439
511
  | Reasoning budget | None. | Optional `PLURNK_PROVIDERS_REASONING_BUDGET`. | Percentage of effective context or absolute count; valid only as a strict subset of total output and effective unless reasoning is `off`. |
440
512
  | Cost override | None. | Optional `PLURNK_PROVIDERS_COST`. | {§operator-cost-override} — comma-separated `key=value` per-1M-token USD rates over `input, output, reasoning, cacheRead, cacheWrite`; merges over the Models.dev catalog block (the catalog is the starting point), alias-scoped like every knob. Without catalog rates the override must declare `input` and `output`. The cost estimate's `source` names the override; a provider-reported response cost still outranks any estimate. |
441
513
  | Reasoning capability | Catalog `reasoning` and route-specific `reasoning_options`. | Adapter wire style only where the catalog cannot name the native field. | Catalog controls determine admissible policy; the adapter determines its wire projection. |
@@ -477,9 +549,9 @@ forcing a vendor-owned resource prefix into PLURNK aliases. Ambiguous suffixes
477
549
  fail to resolve.
478
550
 
479
551
  The catalog package identifies the protocol family, not a mandatory client
480
- implementation. OpenRouter and DeepInfra carry exact charges their AI SDK
481
- projection omits, so each supplies a cost normalizer that reads the documented
482
- response rather than the projection.
552
+ implementation. OpenRouter charges and DeepInfra estimates use cost normalizers
553
+ over documented response fields, retaining their distinct monetary character
554
+ under {§provider-monetary-evidence}.
483
555
 
484
556
  §provider-fact-authority Provider declarations configure facts, not
485
557
  credentials, and Models.dev is authoritative for cataloged providers: package
@@ -530,8 +602,7 @@ PLURNK `Provider`, read PLURNK tuning knobs, or reproduce transport policy.
530
602
  Discovery is scope-agnostic and memoized per process. Duplicate names fail hard.
531
603
  The common plugin trust gate applies before import ({§plugin-trust-boundary}). A plugin absent from
532
604
  Models.dev requires an explicit context-window pin because PLURNK will not guess
533
- model physics. `Discovery.packageAttributions` carries the canonical package map;
534
- the published name-keyed `Discovery.attributions` remains its 1.x projection.
605
+ model physics. `Discovery.packageAttributions` carries the canonical package map.
535
606
 
536
607
  ## §7 Local capabilities
537
608
 
@@ -575,6 +646,16 @@ reasoning enclosure only after preserving grammar evidence. Process-wide
575
646
  llama-server flags are fallback server configuration, not part of the PLURNK
576
647
  contract and need not be synchronized with an alias.
577
648
 
649
+ §repetition-stop **A response that repeats itself is stopped.** A degenerate model can stream one line
650
+ without end; content keeps arriving, so no silence deadline fires, and the call runs to its output
651
+ allowance. The transport counts the complete lines of each streamed channel, text and reasoning, and
652
+ when one line of 16 or more characters has appeared `PLURNK_PROVIDERS_REPEATED_LINE_LIMIT` times it stops
653
+ reading: the stream is cancelled, the physical request settles with whatever usage the chunks carried,
654
+ and the call fails as `repetition` (422, not retryable) carrying the partial attempt and the sentence
655
+ "The response repeated one line N times and was stopped: `<line>`". The consumer reads it as an invalid
656
+ emission, never as a provider failure to recover. Measured: a granite-4.2-8b run repeated one line 358
657
+ times across 45,875 tokens and 9 min 42 s; a limit of 32 stops it 14.6% of the way in (#876).
658
+
578
659
  §llama-reasoning-request The allowance is cumulative across the complete response. Opening a second or
579
660
  later reasoning block does not replenish it. Template parsing, the reasoning
580
661
  sampler, normalized usage, and the returned reasoning channel MUST agree on that
@@ -598,14 +679,14 @@ Generic AI SDK calls accept only settings represented by the SDK's portable
598
679
  surface. Compatible endpoints may carry additional sampling keys after reserved
599
680
  keys are removed.
600
681
 
601
- §openrouter-app-attribution **The cataloged OpenRouter route identifies the
602
- calling application through OpenRouter's current app-attribution headers.**
603
- `HTTP-Referer` is the absolute HTTP(S) application URL and
604
- `X-OpenRouter-Title` is its optional display title. The shipped floor identifies
605
- the public Plurnk repository and may be replaced by operator configuration; an
606
- explicitly empty `OPENROUTER_HTTP_REFERER` suppresses both headers. Attribution
607
- applies only to the cataloged `openrouter` route and never leaks to another
608
- provider merely because it uses the same SDK package.
682
+ §openrouter-app-attribution **A route on the OpenRouter SDK identifies the calling
683
+ application only as its provider declares.** `APP_URL`, an absolute HTTP(S) URL, and
684
+ the optional `APP_NAME` are provider declarations (`PLURNK_PROVIDERS_PROVIDER_<NAME>_`,
685
+ overridable per route or alias); the SDK sends them as `HTTP-Referer` and
686
+ `X-OpenRouter-Title`. The shipped floor declares them for `openrouter` only, so
687
+ another provider on the same SDK package sends none; an empty `APP_URL` suppresses
688
+ both. The retired `OPENROUTER_HTTP_REFERER`, `OPENROUTER_APP_TITLE` and
689
+ `OPENROUTER_X_TITLE` fail construction.
609
690
 
610
691
  ## §9 Failures, retries, and cancellation
611
692
 
@@ -618,8 +699,15 @@ into an empty model turn or reduced to a message plus a generic status.
618
699
  Upstream diagnostic text is bounded by
619
700
  `PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT`; the committed `.env.defaults` owns its
620
701
  normal value. Retry exhaustion is preserved as `attempts` and
621
- `retryExhausted`, and the resulting Problem is not marked retryable after the
622
- provider has consumed its automatic retry budget.
702
+ `retryExhausted`; it does not change the Problem's `retryable`.
703
+
704
+ §provider-retryable-truth **`retryable` states what happens next.** A Problem's
705
+ `retryable` is exactly membership of its kind in the exported
706
+ `RETRYABLE_PROVIDER_KINDS` — `rate_limit`, `network_failure`, `deadline_exceeded`,
707
+ `resource_interrupted`, `output_dropped` — and Core's provider recovery re-issues
708
+ exactly those kinds by reading the same set. It never copies the transport's own
709
+ retry policy: a 502 or 524 that the transport will not replay is still re-issued by
710
+ the consumer, so it reports `retryable: true`. Every other kind reports `false`.
623
711
 
624
712
  §provider-failure-cause **The wrapper's cause is evidence, bounded.** The SDK
625
713
  reports every processing failure of a 2xx body with one message ("Failed to
@@ -660,20 +748,29 @@ and the notice names the true per-call grant (`capacity.responseMax` on the
660
748
  response) — the tolerance's honest edge, never worse than the fixed allowance
661
749
  it forgives.
662
750
 
663
- §provider-sampling-passthrough **The service imposes no sampling opinion.** With
664
- no configured value, a request carries no `temperature` and no repetition
665
- control — the provider's or model's own defaults govern, as a normal agentic
666
- service behaves. `PLURNK_PROVIDERS_TEMPERATURE` and
667
- `PLURNK_PROVIDERS_REPEAT_PENALTY` parse empty as null and exist as per-alias
668
- measurement instruments; caller `sampling` always outranks a configured value,
669
- and router-owned-tuning providers suppress configured floors entirely. The
670
- justified exception the penalty knob exists for: grammar-constrained decoding
671
- can loop under the mask — greedy continuation of an exactly-repeatable sequence
672
- the grammar keeps legal (measured on the July local rail work, #9/#30; and
673
- unconstrained on firefast, where 4/86 bench turns ran straight to the token cap
674
- on pure repetition, run52) — and a per-alias `repeat_penalty` (llama.cpp
675
- backends) or `frequency_penalty` (cloud backends, #426) is the measured remedy
676
- for a model that exhibits it, never a service-wide floor.
751
+ §provider-sampling-passthrough **Unconfigured sampling retains endpoint defaults.**
752
+ The provider panel owns tuning; Models.dev capabilities are not recommended
753
+ sampling values. Every knob accepts the ordinary alias override
754
+ ({§provider-configuration}). Both cataloged and local providers apply the same
755
+ validation and precedence: configured values, then caller `sampling`, then
756
+ transport-owned fields ({§provider-request-authority}).
757
+
758
+ | `PLURNK_PROVIDERS_` suffix | Compatible wire / native SDK | Accepted configuration |
759
+ | --- | --- | --- |
760
+ | `TEMPERATURE` | `temperature` / `temperature` | Finite non-negative number; empty/unset omits. |
761
+ | `TOP_P` | `top_p` / `topP` | Finite number in `[0,1]`; empty/unset omits. |
762
+ | `TOP_K` | `top_k` / `topK` | Non-negative safe integer; empty/unset omits. Zero semantics are endpoint-owned. |
763
+ | `PRESENCE_PENALTY` | `presence_penalty` / `presencePenalty` | Finite number in `[-2,2]`; empty/unset omits. |
764
+ | `FREQUENCY_PENALTY` | `frequency_penalty` / `frequencyPenalty` | Required finite number in `[-2,2]`; the panel's zero selects no configured override. Negative values survive unchanged. |
765
+ | `SEED` | `seed` / `seed` | Safe integer; empty/unset omits. No guarantee of reproducibility. |
766
+
767
+ Configured zero is preserved except for the frequency knob's explicit
768
+ no-override sentinel; a caller-supplied zero always overrides a configured value.
769
+ Invalid configuration fails by knob name before inference. The SDK/endpoint
770
+ owns narrower model restrictions and unsupported-setting diagnostics; Plurnk
771
+ does not clamp values or invent model-specific sampling profiles.
772
+ `REPEAT_PENALTY` and DRY remain explicit llama-server extensions, omitted when
773
+ unconfigured. Their grammar protection remains at {§provider-grammar-transport}.
677
774
 
678
775
  §provider-connectivity The provider adapter owns one attempt scheduler around
679
776
  the complete generation exchange; SDK-internal retries are disabled.
@@ -683,20 +780,29 @@ deadline:
683
780
 
684
781
  | Layer | Operator knob | Boundary | Expiry |
685
782
  | --- | --- | --- | --- |
686
- | Operation | `PLURNK_PROVIDERS_OPERATION_TIMEOUT` | Complete logical call, including every attempt and retry delay. | Final `deadline_exceeded` Problem at 504 with `timeoutPhase=operation`; never retried. Enforced as a race, not only the advisory signal, so a wedged transport that never observes the abort cannot hang the loop past the deadline (#505); a well-behaved transport unwinds within a short grace and settles its own attempt evidence first. |
687
- | Attempt | `PLURNK_PROVIDERS_FETCH_TIMEOUT` | One physical generation request. A non-streamed request is bounded through response consumption; a streamed one only until its first semantic content, after which stream-idle catches a stall and the operation deadline bounds the whole, so a stream still producing (long reasoning) is never cut off. | Surfaced `network_failure` with `timeoutPhase=attempt`; never transport-retried (#479) — the consumer's recovery owns re-issue. |
688
- | First content | `PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT` | Response-stream start through first semantic model content; metadata, empty deltas, and transport activity do not satisfy it. | Surfaced `network_failure` with `timeoutPhase=first_content`; never transport-retried (#479). |
783
+ | Operation | `PLURNK_PROVIDERS_OPERATION_TIMEOUT` | Complete logical call, including admission queueing, every attempt and retry delay. | Final `deadline_exceeded` Problem at 504 with `timeoutPhase=operation`; not retried inside this operation. Consumer recovery is separate. Enforced as a race, not only the advisory signal, so a wedged transport that never observes the abort cannot hang the loop past the deadline (#505); a well-behaved transport unwinds within a short grace and settles its own attempt evidence first. |
784
+ | Attempt | `PLURNK_PROVIDERS_FETCH_TIMEOUT` | One physical generation request. A non-streamed request is bounded through response consumption; a streamed one only until its first semantic content. After content begins, stream-idle bounds silence and the operation deadline still bounds the whole call. | Surfaced `network_failure` with `timeoutPhase=attempt`; never transport-retried (#479) — the consumer's recovery owns re-issue. |
785
+ | First content | `PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT` | Dispatch through first semantic model content ({§provider-first-content-at-dispatch}); metadata, empty deltas, and transport activity do not satisfy it. | Surfaced `network_failure` with `timeoutPhase=first_content`; never transport-retried (#479). |
689
786
  | Stream idle | `PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT` | Silence between semantic content chunks after content begins. | Surfaced `network_failure` with `timeoutPhase=stream_idle`; never transport-retried (#479). |
787
+ | Repeated line | `PLURNK_PROVIDERS_REPEATED_LINE_LIMIT` | A streamed line of 16 or more characters repeated this many times; `0` disables. | Stopped as `repetition` ({§repetition-stop}); never transport-retried. |
788
+
789
+ §provider-first-content-at-dispatch The first-content deadline is armed when the
790
+ admitted attempt is dispatched, before response headers, so an endpoint that accepts
791
+ the request and never answers fails at `timeoutPhase=first_content` instead of holding
792
+ the attempt for the whole attempt deadline. It still begins only after admission
793
+ ({§provider-inference-admission}).
794
+
795
+ Settled calls remove their deadline timers and cancellation subscriptions.
690
796
 
691
797
  Caller cancellation spans the operation and preserves the caller's reason.
692
798
  A 2xx exchange whose body cannot be processed (a provider invalid-response)
693
799
  classifies as the non-retryable 502 on the first failure unless an explicit
694
- `x-should-retry` directive says otherwise (#479 supersedes #446's budgeted
695
- promotion). Inner deadline failures surface on the first failure; when a
800
+ `x-should-retry` directive says otherwise (#479). Inner deadline failures surface on the first failure; when a
696
801
  directive-driven retry sequence exhausts, `attempts` and `retryExhausted`
697
- are added and the classification is final. Every scheduler iteration opens and settles exactly
698
- one ordered {§provider-request-accounting} record, including response-less
699
- network failures and timed-out attempts.
802
+ are added and the classification is final. Every admitted physical attempt opens
803
+ and settles exactly one ordered {§provider-request-accounting} record, including
804
+ response-less network failures and timed-out attempts. A cancelled or expired
805
+ admission wait opens no physical request.
700
806
 
701
807
  Each streamed physical request assembles its own response. Failed partial answer
702
808
  bytes never enter a later request's completed `ProviderResponse`; recovery is a
@@ -715,6 +821,21 @@ ordinary 5xx surface on the first failure as consumer-recoverable kinds; one
715
821
  retry authority — the consumer's own provider-recovery machinery — owns
716
822
  re-issue, backoff, and park above the transport.
717
823
 
824
+ §provider-output-dropped **A completed exchange whose own evidence proves its
825
+ text never arrived is `output_dropped`**, a 502 carrying the complete response as
826
+ `error.attempt` and `stage: "provider-response"`; it is never admitted as an empty or
827
+ truncated turn, and the consumer re-issues it ({§provider-retryable-truth}):
828
+
829
+ | Evidence | Detail | Facts |
830
+ | --- | --- | --- |
831
+ | `finish_reason` `tool_calls` (Plurnk never declares tools) | `The provider ended the response with tool calls although no tools were declared (N native tool call(s)); the response text was not delivered as text.` | `toolCallCount` |
832
+ | Billed output tokens exceed the code points streamed across every channel — text and reasoning, before projection — by at least `PLURNK_PROVIDERS_DROPPED_OUTPUT_TOKENS` (`0` disables; an output token decodes to at least one character) | `The provider billed T output tokens but streamed C characters of text and reasoning; at least T−C tokens of output never arrived.` | `billedOutputTokens`, `streamedCharacters`, `droppedOutputTokens` |
833
+
834
+ The token rule needs a billed output count, and it does not apply when the response bills
835
+ reasoning tokens but streamed no reasoning in any channel: that is hidden or summarized
836
+ reasoning, which the characters cannot account for. Text-only comparisons are not made —
837
+ a route may bill reasoning inside its text count.
838
+
718
839
  §provider-request-rejection An HTTP 4xx rejection not classified as authorization,
719
840
  quota, capacity, grammar, rate limit, or transient transport failure is
720
841
  `request_rejected`: preserve the upstream status and detail, with `retryable: false`.
@@ -733,7 +854,7 @@ completed exchange.
733
854
  | Normalized finish reason | Exact `insufficient_system_resource` becomes `resource_interrupted`; the raw value remains in `assistantRaw`. |
734
855
  | `generate` outcome | Throw `ProviderError(kind="resource_interrupted")` at local status 503 with the normalized attempt on `error.attempt` and the same accounting on `error.accounting`. |
735
856
  | Partial response | Preserve content, reasoning, usage, model, metadata, optional raw body, and other evidence without admitting it as success. |
736
- | Automatic replay | None; the Problem has `retryable: false`, and AI SDK retry scheduling has already completed at the successful transport. |
857
+ | Automatic replay | None inside the provider; AI SDK retry scheduling has already completed. The Problem has `retryable: true`: the consumer re-issues it ({§provider-retryable-truth}). |
737
858
  | Capacity-pool overflow | None under the existing routing policy; when other overflow-eligible failures do reach a sibling, the pool concatenates their request accounting. |
738
859
  | Consumer admission | Persist the evidence as an unaccepted attempt; never parse it into executable work, even when its frame looks complete. |
739
860
 
@@ -777,19 +898,26 @@ by index with their arguments as streamed, the finish reasons seen, and the verb
777
898
  of every string field the record does not already hold. Nothing is opt-in and nothing is
778
899
  reinterpreted: a blank emission reads as thirteen chunks that carried nothing, or as a
779
900
  tool-call section the model emitted into a channel the protocol does not read, instead of
780
- being guessed at from token counts. Covered by `aiSdkTransport.test.ts`.
901
+ being guessed at from token counts. Frames outside the recognized OpenAI-compatible
902
+ envelope remain verbatim in `wire.unmappedChunks`; the recorder never discards an
903
+ unrecognized/native payload or invents another provider parser. This required
904
+ evidence is independent of the optional complete `rawBody` dataset capture.
905
+ Covered by `aiSdkTransport.test.ts` and core's `Digest.wire-evidence.test.ts`.
781
906
 
782
907
  §provider-usage-refusal **The provider's bookkeeping is not the exchange.** Usage
783
- normalization is exact and refuses counters that contradict each other (a
784
- reasoning detail larger than its output aggregate, a total that disagrees with
785
- its parts). A refusal is not a request failure: the response is delivered, its
786
- accounting row carries no `usage` and a cost of kind `unknown` with the reason
787
- (never an invented, clamped, or zero counter), and the transport record keeps
788
- `usageRefusal` — the normalizer's reason beside the counters exactly as the
789
- wire reported them — so the digest can read what the provider said. The same
790
- holds for the usage inside failure evidence. Before this rule a contradictory
791
- counter turned a complete response into a retryable `network_failure` and lost
792
- the response (#580). Covered by `aiSdkTransport.test.ts` and
908
+ normalization is exact; a refusal never fails or retries the response. The same
909
+ rules apply to usage inside failure evidence:
910
+
911
+ | Contradiction | Normalized evidence |
912
+ | --- | --- |
913
+ | Invalid aggregate or total inconsistent with its parts | No `usage`. |
914
+ | Invalid optional counter | Omit that counter; retain independently valid quantities. |
915
+ | Cache or reasoning breakdown inconsistent with its aggregate | Omit that breakdown; retain valid aggregates and the other breakdown. |
916
+
917
+ Every refusal retains `usageRefusal`: the reason and the original wire counters.
918
+ No value is clamped, zeroed, or estimated. Cost uses authoritative charges when
919
+ available; catalog estimation requires all quantities its rates need, otherwise
920
+ cost is `unknown`. Covered by `usage.test.ts`, `aiSdkTransport.test.ts`, and
793
921
  `AiSdkProvider.test.ts`.
794
922
 
795
923
  §provider-open-reasoning **Plurnk reads the model's own reasoning.** The service is
@@ -812,8 +940,9 @@ budget is a strict subset of that total, never an additive reserve. The
812
940
  configured total is a percentage of effective context or an absolute count;
813
941
  percentages resolve to the nearest whole token with a one-token minimum. It is
814
942
  capped by known context and model-output limits; `generate.maxOutputTokens` may
815
- only tighten it for one call. The effective reasoning subset tightens with that
816
- total and remains strictly smaller.
943
+ only tighten this reservation for one call. The effective reasoning subset tightens
944
+ with that total and remains strictly smaller. Exact prompt measurements may expand
945
+ the response grant beyond the reservation under {§provider-flexed-allowance}.
817
946
 
818
947
  The adapter owns native projection. A backend whose generic SDK maximum already
819
948
  includes reasoning receives the total directly. When a native SDK instead adds
@@ -827,7 +956,7 @@ and provider minimum; an envelope too small to represent the minimum fails
827
956
  before provider I/O.
828
957
 
829
958
  §provider-output-budget-conformance When a completed response reports
830
- normalized output-token usage greater than its effective total output budget,
959
+ normalized output-token usage greater than its response grant ({§provider-flexed-allowance}),
831
960
  the exchange is an `invalid_response` at 502 rather than an admitted result or
832
961
  a prompt-capacity 413. Its complete failed-attempt evidence and settled charged
833
962
  request remain available. The violation is final and is never automatically
@@ -840,7 +969,29 @@ limit advertises `requiresOutputBudget` and fails construction when no total can
840
969
  be resolved. The retired additive reserve knobs fail hard rather than creating
841
970
  a second envelope contract.
842
971
 
843
- ## §13 Capacity pool
972
+ ## §13 Inference capacity
973
+
974
+ ### Admission
975
+
976
+ §provider-inference-admission `PLURNK_PROVIDERS_MAX_CONCURRENCY` controls physical
977
+ generation attempts per resolved endpoint within one process: `-1` is unrestricted;
978
+ a positive safe integer admits that many concurrent attempts. It follows ordinary
979
+ alias scoping. Zero, other negative values and non-integers are invalid.
980
+
981
+ | Boundary | Contract |
982
+ | --- | --- |
983
+ | Identity | Provider instances and aliases sharing the resolved API base URL share one allowance. SDK/plugin-owned endpoints without a resolved URL share their provider identity. Conflicting limits for one identity fail construction, naming the identity and both values; they never create independent queues. The daemon reads its environment once at boot, so an identity's limit cannot change within a process: reconstructing a provider from that environment reuses its allowance, and in-flight leases keep counting. |
984
+ | Admission | FIFO among live waiters. The lease begins before the physical request observer and ends after the complete response or transport failure settles, including streamed bodies. |
985
+ | Cancellation | A queued abort removes that waiter and preserves the caller's reason. It opens no physical request or accounting row. In-flight cancellation signals the transport; capacity is released when that attempt unwinds. A transport still running despite abort does not authorize exceeding the limit. |
986
+ | Retries | Backoff holds no lease. Each retry rejoins admission as a new physical attempt. |
987
+ | Deadlines | The existing operation deadline includes queueing. Attempt, first-content and stream-idle deadlines begin only after admission; queued work is not a stalled stream. |
988
+ | Scope | Workers, tools, messages, waits, token measurement and endpoint discovery are not serialized by inference admission. Independent endpoints progress independently. No cross-process or machine-wide capacity guarantee is implied. |
989
+
990
+ Admission neither changes worker lifecycle nor selects another model or endpoint.
991
+ Core consumes the same asynchronous Provider contract. A waiting worker holds no
992
+ inference lease; BARE and ordinary generation use the same physical boundary.
993
+
994
+ ### Pool
844
995
 
845
996
  §provider-capacity-pool `Pool` fronts interchangeable `Provider` instances. It
846
997
  keeps workers sticky for
@@ -868,6 +1019,8 @@ Coverage MUST prove:
868
1019
  - native SDK request mapping and normalized responses;
869
1020
  - compatible extension preservation;
870
1021
  - timeout, retry, cancellation, interrupted-attempt, and final-error behavior;
1022
+ - shared endpoint inference admission, FIFO queueing, queued/in-flight cancellation,
1023
+ full-stream lease lifetime, retry release, and independent endpoint progress;
871
1024
  - local capability probes and pins;
872
1025
  - exact, bounded, estimated, and unavailable complete-request measurements;
873
1026
  - independent input/context/output limits, asymmetric admission, and normalized
@@ -1,14 +1,15 @@
1
- import type { ChatMessage, PromptTokenMeasurement, Provider, ProviderCostNormalizer, ProviderGenerateArgs, ProviderRequestCapacity, ProviderResponse, ProviderUsage, ReasoningPolicy } from "./types.ts";
1
+ import type { ChatMessage, PromptTokenMeasurement, Provider, ProviderCostNormalizer, ProviderGenerateArgs, ProviderRequestCapacity, ProviderResponse, ProviderUsage, Effort } from "./types.ts";
2
2
  import type { ProviderCost } from "@plurnk/plurnk-contracts";
3
3
  import type { JSONValue } from "ai";
4
- import { type Reasoning, type ReasoningResponseStyle } from "./env.ts";
4
+ import type { EffortSetting, ReasoningResponseStyle } from "./env.ts";
5
5
  import type { InputModality } from "./types.ts";
6
6
  import type { LanguageModel } from "ai";
7
7
  import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
8
+ import type RequestFields from "./RequestFields.ts";
9
+ import type InferenceAdmission from "./InferenceAdmission.ts";
8
10
  export type ProviderFetch = typeof globalThis.fetch;
9
- export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "effort_required" | "thinking_effort" | "template" | "anthropic";
10
- export type NativeReasoningEffort = "minimal" | "low" | "medium" | "high" | "xhigh";
11
- export type CompatibleReasoningEffort = NativeReasoningEffort | "max";
11
+ export type ReasoningStyle = "none" | "think" | "template";
12
+ export type NativeEffort = "minimal" | "low" | "medium" | "high" | "xhigh";
12
13
  export type GrammarStyle = "none" | "llamacpp";
13
14
  export type CacheAffinity = {
14
15
  readonly target: "header" | "body";
@@ -20,6 +21,7 @@ export type CacheAffinity = {
20
21
  };
21
22
  export type AiSdkProviderOptions = Record<string, Record<string, JSONValue | undefined>>;
22
23
  export type AiSdkProviderConfig = {
24
+ inferenceAdmission?: InferenceAdmission;
23
25
  model: string;
24
26
  url?: string;
25
27
  languageModel?: LanguageModel;
@@ -28,6 +30,8 @@ export type AiSdkProviderConfig = {
28
30
  operationTimeoutMs: number;
29
31
  firstContentTimeoutMs: number;
30
32
  streamIdleTimeoutMs?: number;
33
+ repeatedLineLimit?: number;
34
+ droppedOutputTokens?: number;
31
35
  headers?: Record<string, string>;
32
36
  fetch?: ProviderFetch;
33
37
  contextWindow?: number | null;
@@ -36,14 +40,12 @@ export type AiSdkProviderConfig = {
36
40
  maxOutputTokens?: number | null;
37
41
  outputBudget?: number | null;
38
42
  reasoningBudget?: number | null;
39
- supportedReasoningPolicies?: readonly ReasoningPolicy[];
40
- adaptiveReasoning?: NativeReasoningEffort | "provider-default";
41
- reasoningToggle?: boolean;
42
- compatibleAdaptiveReasoning?: CompatibleReasoningEffort | "provider-default";
43
- compatibleOffReasoning?: "none";
44
- adaptiveReasoningProviderOptions?: AiSdkProviderOptions;
43
+ supportedEfforts?: readonly Effort[];
44
+ adaptiveEffort?: NativeEffort | "provider-default";
45
+ adaptiveEffortProviderOptions?: AiSdkProviderOptions;
45
46
  additiveReasoningProvider?: "anthropic" | "bedrock";
46
47
  reasoningStyle?: ReasoningStyle;
48
+ requestFields?: RequestFields;
47
49
  reasoningResponseStyle?: ReasoningResponseStyle;
48
50
  countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
49
51
  estimateCost?: (usage: ProviderUsage | undefined) => ProviderCost;
@@ -63,8 +65,12 @@ export type AiSdkProviderConfig = {
63
65
  promptTokensUrl?: string;
64
66
  servedModel?: string;
65
67
  requiresOutputBudget?: boolean;
66
- reasoning: Reasoning;
68
+ effort: EffortSetting;
67
69
  temperature: number | null;
70
+ topP?: number;
71
+ topK?: number;
72
+ presencePenalty?: number;
73
+ seed?: number;
68
74
  repeatPenalty: number | null;
69
75
  frequencyPenalty?: number;
70
76
  dryMultiplier?: number;
@@ -87,7 +93,7 @@ export default class AiSdkProvider implements Provider {
87
93
  get maxOutputTokens(): number | null;
88
94
  get outputBudget(): number | null;
89
95
  get reasoningBudget(): number | null;
90
- get supportedReasoningPolicies(): readonly ReasoningPolicy[];
96
+ get supportedEfforts(): readonly Effort[];
91
97
  get inputCapacity(): number | null;
92
98
  get model(): string;
93
99
  get servedModel(): string | undefined;