@plurnk/plurnk-providers 1.14.2 → 1.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/.env.defaults +1 -1
  2. package/SPEC.md +16 -2
  3. package/dist/AiSdkProvider.d.ts +3 -0
  4. package/dist/AiSdkProvider.d.ts.map +1 -1
  5. package/dist/AiSdkProvider.js +50 -339
  6. package/dist/AiSdkProvider.js.map +1 -1
  7. package/dist/AiSdkRequestBody.d.ts +40 -0
  8. package/dist/AiSdkRequestBody.d.ts.map +1 -0
  9. package/dist/AiSdkRequestBody.js +361 -0
  10. package/dist/AiSdkRequestBody.js.map +1 -0
  11. package/dist/Mock.d.ts +5 -1
  12. package/dist/Mock.d.ts.map +1 -1
  13. package/dist/Mock.js +9 -2
  14. package/dist/Mock.js.map +1 -1
  15. package/dist/Pool.d.ts +2 -1
  16. package/dist/Pool.d.ts.map +1 -1
  17. package/dist/Pool.js +5 -0
  18. package/dist/Pool.js.map +1 -1
  19. package/dist/aiSdkTransport.d.ts +1 -1
  20. package/dist/aiSdkTransport.d.ts.map +1 -1
  21. package/dist/aiSdkTransport.js +21 -2
  22. package/dist/aiSdkTransport.js.map +1 -1
  23. package/dist/capacity.d.ts +0 -1
  24. package/dist/capacity.d.ts.map +1 -1
  25. package/dist/capacity.js +1 -1
  26. package/dist/capacity.js.map +1 -1
  27. package/dist/catalogProvider.d.ts +2 -1
  28. package/dist/catalogProvider.d.ts.map +1 -1
  29. package/dist/catalogProvider.js +6 -0
  30. package/dist/catalogProvider.js.map +1 -1
  31. package/dist/index.d.ts +3 -0
  32. package/dist/index.d.ts.map +1 -1
  33. package/dist/index.js +2 -0
  34. package/dist/index.js.map +1 -1
  35. package/dist/promptTokens.d.ts.map +1 -1
  36. package/dist/promptTokens.js +2 -1
  37. package/dist/promptTokens.js.map +1 -1
  38. package/dist/reasoning-effort.d.ts +4 -0
  39. package/dist/reasoning-effort.d.ts.map +1 -0
  40. package/dist/reasoning-effort.js +14 -0
  41. package/dist/reasoning-effort.js.map +1 -0
  42. package/dist/types.d.ts +17 -1
  43. package/dist/types.d.ts.map +1 -1
  44. package/dist/types.js +5 -0
  45. package/dist/types.js.map +1 -1
  46. package/package.json +6 -6
  47. package/src/AiSdkProvider.test.ts +22 -3
  48. package/src/AiSdkProvider.ts +51 -375
  49. package/src/AiSdkRequestBody.ts +419 -0
  50. package/src/Mock.ts +10 -2
  51. package/src/Pool.test.ts +1 -0
  52. package/src/Pool.ts +6 -1
  53. package/src/aiSdkTransport.ts +23 -4
  54. package/src/boundaries.test.ts +2 -0
  55. package/src/capacity.ts +1 -1
  56. package/src/catalogProvider.ts +9 -1
  57. package/src/index.ts +3 -0
  58. package/src/inputModalities.test.ts +53 -0
  59. package/src/promptTokens.ts +2 -1
  60. package/src/reasoning-effort.ts +15 -0
  61. package/src/types.ts +18 -1
package/.env.defaults CHANGED
@@ -125,7 +125,7 @@ PLURNK_PROVIDERS_PROBE_DELAY=250
125
125
  # Rails on a weak model need a recency reminder or turns death-march to the
126
126
  # token ceiling ({§gbnf-forced-march}): put in the Recap footer (core
127
127
  # PLURNK_SERVICE_RECAP) a line such as
128
- # YOU MUST begin with `# PLAN0` and end with `## SEND0 [status code]`
128
+ # YOU MUST begin with `## PLAN0` and end with `### SEND0 [status code]`
129
129
  # PLURNK_PROVIDERS_GBNF=plurnk.qwen.gbnf
130
130
  # Debug toggle: validate but withhold a configured local GBNF, then report the
131
131
  # unconstrained output's divergence. Development aid; leave unset in production.
package/SPEC.md CHANGED
@@ -412,6 +412,20 @@ names without values. Construction rejects the same missing requirements at
412
412
  the provider boundary instead of deferring a known configuration failure to a
413
413
  model request.
414
414
 
415
+ ### §provider-input-modalities Native input parts
416
+
417
+ A provider declares `inputModalities`, the set of native non-text inputs its model
418
+ accepts, from the catalog's input modalities (Models.dev `modalities.input` minus
419
+ `text`, kept to the vocabulary `image`, `pdf`, `audio`, `video`; empty when the
420
+ model is unknown). A user `ChatMessage` may then carry content parts: text beside
421
+ `{ type: "image", image: bytes, mediaType }` and `{ type: "file", data: bytes, mediaType }`,
422
+ which the AI SDK transport forwards as the model's native image and file input;
423
+ system and assistant messages stay text, and prompt-token estimates count text
424
+ only, the provider's reported usage owning each part's cost. A pool declares a
425
+ modality only when every backend does; the Mock declares them by option and
426
+ records every request it receives. Which parts actually ride a request is the
427
+ service's decision per attachment ({§packet-attachment-parts} in the core specification).
428
+
415
429
  ### §model-fact-resolution Model fact precedence
416
430
 
417
431
  Provider and model facts resolve independently:
@@ -660,7 +674,7 @@ deadline:
660
674
 
661
675
  | Layer | Operator knob | Boundary | Expiry |
662
676
  | --- | --- | --- | --- |
663
- | Operation | `PLURNK_PROVIDERS_OPERATION_TIMEOUT` | Complete logical call, including every attempt and retry delay. | Final `deadline_exceeded` Problem at 504 with `timeoutPhase=operation`; never retried. |
677
+ | Operation | `PLURNK_PROVIDERS_OPERATION_TIMEOUT` | Complete logical call, including every attempt and retry delay. | Final `deadline_exceeded` Problem at 504 with `timeoutPhase=operation`; never retried. Enforced as a race, not only the advisory signal, so a wedged transport that never observes the abort cannot hang the loop past the deadline (#505); a well-behaved transport unwinds within a short grace and settles its own attempt evidence first. |
664
678
  | Attempt | `PLURNK_PROVIDERS_FETCH_TIMEOUT` | One physical generation request, including response consumption. | Surfaced `network_failure` with `timeoutPhase=attempt`; never transport-retried (#479) — the consumer's recovery owns re-issue. Embedding requests share the same per-physical-request deadline, enforced as a race so a wedged adapter cannot hang its awaiters (#463). |
665
679
  | First content | `PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT` | Response-stream start through first semantic model content; metadata, empty deltas, and transport activity do not satisfy it. | Surfaced `network_failure` with `timeoutPhase=first_content`; never transport-retried (#479). |
666
680
  | Stream idle | `PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT` | Silence between semantic content chunks after content begins. | Surfaced `network_failure` with `timeoutPhase=stream_idle`; never transport-retried (#479). |
@@ -732,7 +746,7 @@ is `content` and `contentStart` is zero.
732
746
  §gbnf-forced-march **A grammar masks end-of-generation until the sentence completes — pair rails on a weak model with a recency reminder.** llama-server admits EOG only in a grammar-accepting state, so a model whose intended emission diverges from the required turn shape cannot stop: each masked substitution drifts it further from the terminal, and the request runs to the token ceiling and dies as a length cut (measured 2026-09-01, #477 — the thought budget fires and throughput is unaffected; the march is the grammar's one failure mode, and the weaker the model, the likelier the divergence that triggers it). When enabling a GBNF rail for a model that does not reliably hold the frame shape, put the shape command where recency helps — the operator's Recap footer (core `PLURNK_SERVICE_RECAP`) — for example:
733
747
 
734
748
  ```text
735
- YOU MUST begin with `# PLAN0` and end with `## SEND0 [status code]`
749
+ YOU MUST begin with `## PLAN0` and end with `### SEND0 [status code]`
736
750
  ```
737
751
 
738
752
  A model that emits valid frames freehand needs neither the rail nor the reminder.
@@ -2,6 +2,7 @@ import type { ChatMessage, PromptTokenMeasurement, Provider, ProviderCostNormali
2
2
  import type { ProviderCost } from "@plurnk/plurnk-contracts";
3
3
  import type { JSONValue } from "ai";
4
4
  import { type Reasoning, type ReasoningResponseStyle } from "./env.ts";
5
+ import type { InputModality } from "./types.ts";
5
6
  import type { LanguageModel } from "ai";
6
7
  import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
7
8
  export type ProviderFetch = typeof globalThis.fetch;
@@ -30,6 +31,7 @@ export type AiSdkProviderConfig = {
30
31
  headers?: Record<string, string>;
31
32
  fetch?: ProviderFetch;
32
33
  contextWindow?: number | null;
34
+ inputModalities?: ReadonlySet<InputModality>;
33
35
  maxInputTokens?: number | null;
34
36
  maxOutputTokens?: number | null;
35
37
  outputBudget?: number | null;
@@ -83,6 +85,7 @@ export default class AiSdkProvider implements Provider {
83
85
  tokenize?: (text: string) => Promise<number[]>;
84
86
  constructor(config: AiSdkProviderConfig);
85
87
  get contextWindow(): number | null;
88
+ get inputModalities(): ReadonlySet<InputModality>;
86
89
  get maxInputTokens(): number | null;
87
90
  get maxOutputTokens(): number | null;
88
91
  get outputBudget(): number | null;
@@ -1 +1 @@
1
- {"version":3,"file":"AiSdkProvider.d.ts","sourceRoot":"","sources":["../src/AiSdkProvider.ts"],"names":[],"mappings":"AAQA,OAAO,KAAK,EACR,WAAW,EAEX,sBAAsB,EACtB,QAAQ,EAER,sBAAsB,EAEtB,oBAAoB,EAEpB,uBAAuB,EAEvB,gBAAgB,EAChB,aAAa,EACb,eAAe,EAClB,MAAM,YAAY,CAAC;AACpB,OAAO,KAAK,EAAE,YAAY,EAAE,MAAM,0BAA0B,CAAC;AAE7D,OAAO,KAAK,EAAe,SAAS,EAAE,MAAM,IAAI,CAAC;AACjD,OAAO,EAA2B,KAAK,SAAS,EAAE,KAAK,sBAAsB,EAAE,MAAM,UAAU,CAAC;AAQhG,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,IAAI,CAAC;AAOxC,OAAO,KAAK,EAAE,iBAAiB,EAAE,wBAAwB,EAAE,MAAM,qBAAqB,CAAC;AAMvF,MAAM,MAAM,aAAa,GAAG,OAAO,UAAU,CAAC,KAAK,CAAC;AAIpD,MAAM,MAAM,cAAc,GAAG,MAAM,GAAG,OAAO,GAAG,mBAAmB,GAAG,QAAQ,GAAG,iBAAiB,GAAG,iBAAiB,GAAG,iBAAiB,GAAG,UAAU,GAAG,WAAW,CAAC;AAEtK,MAAM,MAAM,qBAAqB,GAAG,SAAS,GAAG,KAAK,GAAG,QAAQ,GAAG,MAAM,GAAG,OAAO,CAAC;AACpF,MAAM,MAAM,yBAAyB,GAAG,qBAAqB,GAAG,KAAK,CAAC;AAItE,MAAM,MAAM,YAAY,GAAG,MAAM,GAAG,UAAU,CAAC;AAE/C,MAAM,MAAM,aAAa,GACnB;IAAE,QAAQ,CAAC,MAAM,EAAE,QAAQ,GAAG,MAAM,CAAC;IAAC,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GAC7D;IAAE,QAAQ,CAAC,MAAM,EAAE,iBAAiB,CAAC;IAAC,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAAC,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAA;CAAE,CAAC;AAE/F,MAAM,MAAM,oBAAoB,GAAG,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,MAAM,EAAE,SAAS,GAAG,SAAS,CAAC,CAAC,CAAC;AAqBzF,MAAM,MAAM,mBAAmB,GAAG;IAC9B,KAAK,EAAE,MAAM,CAAC;IACd,GAAG,CAAC,EAAE,MAAM,CAAC;IACb,aAAa,CAAC,EAAE,aAAa,CAAC;IAC9B,YAAY,CAAC,EAAE,CAAC,OAAO,EAAE,wBAAwB,KAAK,iBAAiB,CAAC;IACxE,cAAc,EAAE,MAAM,CAAC;IACvB,kBAAkB,EAAE,MAAM,CAAC;IAC3B,qBAAqB,EAAE,MAAM,CAAC;IAC9B,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACjC,KAAK,CAAC,EAAE,aAAa,CAAC;IACtB,aAAa,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC9B,cAAc,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC/B,eAAe,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAChC,YAAY,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC7B,eAAe,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAChC,0BAA0B,CAAC,EAAE,SAAS,eAAe,EAAE,CAAC;IAIxD,iBAAiB,CAAC,EAAE,qBAAqB,GAAG,kBAAkB,CAAC;IAE/D,eAAe,CAAC,EAAE,OAAO,CAAC;IAI1B,2BAA2B,CAAC,EAAE,yBAAyB,GAAG,kBAAkB,CAAC;IAC7E,sBAAsB,CAAC,EAAE,MAAM,CAAC;IAChC,gCAAgC,CAAC,EAAE,oBAAoB,CAAC;IAKxD,yBAAyB,CAAC,EAAE,WAAW,GAAG,SAAS,CAAC;IACpD,cAAc,CAAC,EAAE,cAAc,CAAC;IAChC,sBAAsB,CAAC,EAAE,sBAAsB,CAAC;IAChD,iBAAiB,CAAC,EAAE,CAAC,QAAQ,EAAE,SAAS,WAAW,EAAE,EAAE,MAAM,CAAC,EAAE,WAAW,KAAK,sBAAsB,GAAG,OAAO,CAAC,sBAAsB,CAAC,CAAC;IACzI,YAAY,CAAC,EAAE,CAAC,KAAK,EAAE,aAAa,GAAG,SAAS,KAAK,YAAY,CAAC;IAClE,aAAa,CAAC,EAAE,sBAAsB,CAAC;IACvC,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,YAAY,CAAC,EAAE,YAAY,CAAC;IAG5B,aAAa,CAAC,EAAE,aAAa,CAAC;IAG9B,0BAA0B,CAAC,EAAE,oBAAoB,CAAC;IAGlD,gCAAgC,CAAC,EAAE,oBAAoB,CAAC;IAGxD,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,SAAS,CAAC,EAAE,OAAO,CAAC;IACpB,SAAS,CAAC,EAAE,OAAO,CAAC;IACpB,kBAAkB,CAAC,EAAE,OAAO,CAAC;IAC7B,qBAAqB,CAAC,EAAE,MAAM,CAAC;IAC/B,OAAO,CAAC,EAAE,MAAM,CAAC;IAEjB,mBAAmB,CAAC,EAAE,OAAO,CAAC;IAC9B,SAAS,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAI1B,WAAW,CAAC,EAAE,MAAM,CAAC;IAGrB,eAAe,CAAC,EAAE,MAAM,CAAC;IAKzB,WAAW,CAAC,EAAE,MAAM,CAAC;IAIrB,oBAAoB,CAAC,EAAE,OAAO,CAAC;IAM/B,SAAS,EAAE,SAAS,CAAC;IAOrB,WAAW,EAAE,MAAM,GAAG,IAAI,CAAC;IAC3B,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;IAK7B,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAK1B,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,WAAW,CAAC,EAAE,MAAM,CAAC;IAKrB,aAAa,EAAE,MAAM,CAAC;IAItB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAQ1B,WAAW,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC5B,OAAO,CAAC,EAAE,OAAO,CAAC;IAUlB,YAAY,CAAC,EAAE,OAAO,CAAC;CAC1B,CAAC;AAyJF,MAAM,CAAC,OAAO,OAAO,aAAc,YAAW,QAAQ;;IAyDlD,QAAQ,CAAC,YAAY,CAAC,EAAE,CAAC,OAAO,EAAE,wBAAwB,KAAK,iBAAiB,CAAC;IAMjF,QAAQ,CAAC,EAAE,CAAC,IAAI,EAAE,MAAM,KAAK,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;IAC/C,YAAY,MAAM,EAAE,mBAAmB,EAmKtC;IAED,IAAI,aAAa,IAAI,MAAM,GAAG,IAAI,CAAgC;IAClE,IAAI,cAAc,IAAI,MAAM,GAAG,IAAI,CAAiC;IACpE,IAAI,eAAe,IAAI,MAAM,GAAG,IAAI,CAAkC;IACtE,IAAI,YAAY,IAAI,MAAM,GAAG,IAAI,CAA+B;IAChE,IAAI,eAAe,IAAI,MAAM,GAAG,IAAI,CAAkC;IACtE,IAAI,0BAA0B,IAAI,SAAS,eAAe,EAAE,CAA6C;IACzG,IAAI,aAAa,IAAI,MAAM,GAAG,IAAI,CAMjC;IACD,IAAI,KAAK,IAAI,MAAM,CAAwB;IAE3C,IAAI,WAAW,IAAI,MAAM,GAAG,SAAS,CAA8B;IAEnE,IAAI,oBAAoB,IAAI,OAAO,GAAG,SAAS,CAAuC;IAItF,IAAI,gBAAgB,IAAI,OAAO,CAA0C;IAEnE,iBAAiB,CACnB,QAAQ,EAAE,SAAS,WAAW,EAAE,EAChC,MAAM,CAAC,EAAE,WAAW,GACrB,OAAO,CAAC,sBAAsB,CAAC,CAqDjC;IAEK,qBAAqB,CACvB,QAAQ,EAAE,SAAS,WAAW,EAAE,EAChC,eAAe,CAAC,EAAE,MAAM,EACxB,MAAM,CAAC,EAAE,WAAW,GACrB,OAAO,CAAC,uBAAuB,CAAC,CAmBlC;IAgTK,QAAQ,CAAC,EAAE,QAAQ,EAAE,QAAQ,EAAE,eAAe,EAAE,MAAM,EAAE,OAAO,EAAE,eAAe,EAAE,YAAY,EAAE,MAAM,EAAE,OAAO,EAAE,WAAW,EAAE,IAAI,EAAE,IAAI,EAAE,QAAQ,EAAE,cAAc,EAAE,gBAAgB,EAAE,QAAQ,EAAE,EAAE,oBAAoB,GAAG,OAAO,CAAC,gBAAgB,CAAC,CAkcvP;CAEJ"}
1
+ {"version":3,"file":"AiSdkProvider.d.ts","sourceRoot":"","sources":["../src/AiSdkProvider.ts"],"names":[],"mappings":"AAQA,OAAO,KAAK,EAAE,WAAW,EAAmB,sBAAsB,EAAE,QAAQ,EAAmB,sBAAsB,EAAE,oBAAoB,EAA6B,uBAAuB,EAA6B,gBAAgB,EAAE,aAAa,EAAE,eAAe,EAAE,MAAM,YAAY,CAAC;AACjS,OAAO,KAAK,EAAE,YAAY,EAAE,MAAM,0BAA0B,CAAC;AAE7D,OAAO,KAAK,EAAe,SAAS,EAAE,MAAM,IAAI,CAAC;AACjD,OAAO,EAA2B,KAAK,SAAS,EAAE,KAAK,sBAAsB,EAAE,MAAM,UAAU,CAAC;AAEhG,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,YAAY,CAAC;AAEhD,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,IAAI,CAAC;AAMxC,OAAO,KAAK,EAAE,iBAAiB,EAAE,wBAAwB,EAAE,MAAM,qBAAqB,CAAC;AAQvF,MAAM,MAAM,aAAa,GAAG,OAAO,UAAU,CAAC,KAAK,CAAC;AA0BpD,MAAM,MAAM,cAAc,GAAG,MAAM,GAAG,OAAO,GAAG,mBAAmB,GAAG,QAAQ,GAAG,iBAAiB,GAAG,iBAAiB,GAAG,iBAAiB,GAAG,UAAU,GAAG,WAAW,CAAC;AAEtK,MAAM,MAAM,qBAAqB,GAAG,SAAS,GAAG,KAAK,GAAG,QAAQ,GAAG,MAAM,GAAG,OAAO,CAAC;AACpF,MAAM,MAAM,yBAAyB,GAAG,qBAAqB,GAAG,KAAK,CAAC;AAItE,MAAM,MAAM,YAAY,GAAG,MAAM,GAAG,UAAU,CAAC;AAE/C,MAAM,MAAM,aAAa,GACnB;IAAE,QAAQ,CAAC,MAAM,EAAE,QAAQ,GAAG,MAAM,CAAC;IAAC,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GAC7D;IAAE,QAAQ,CAAC,MAAM,EAAE,iBAAiB,CAAC;IAAC,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAAC,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAA;CAAE,CAAC;AAE/F,MAAM,MAAM,oBAAoB,GAAG,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,MAAM,EAAE,SAAS,GAAG,SAAS,CAAC,CAAC,CAAC;AAIzF,MAAM,MAAM,mBAAmB,GAAG;IAC9B,KAAK,EAAE,MAAM,CAAC;IACd,GAAG,CAAC,EAAE,MAAM,CAAC;IACb,aAAa,CAAC,EAAE,aAAa,CAAC;IAC9B,YAAY,CAAC,EAAE,CAAC,OAAO,EAAE,wBAAwB,KAAK,iBAAiB,CAAC;IACxE,cAAc,EAAE,MAAM,CAAC;IACvB,kBAAkB,EAAE,MAAM,CAAC;IAC3B,qBAAqB,EAAE,MAAM,CAAC;IAC9B,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACjC,KAAK,CAAC,EAAE,aAAa,CAAC;IACtB,aAAa,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC9B,eAAe,CAAC,EAAE,WAAW,CAAC,aAAa,CAAC,CAAC;IAC7C,cAAc,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC/B,eAAe,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAChC,YAAY,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC7B,eAAe,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAChC,0BAA0B,CAAC,EAAE,SAAS,eAAe,EAAE,CAAC;IAIxD,iBAAiB,CAAC,EAAE,qBAAqB,GAAG,kBAAkB,CAAC;IAE/D,eAAe,CAAC,EAAE,OAAO,CAAC;IAI1B,2BAA2B,CAAC,EAAE,yBAAyB,GAAG,kBAAkB,CAAC;IAC7E,sBAAsB,CAAC,EAAE,MAAM,CAAC;IAChC,gCAAgC,CAAC,EAAE,oBAAoB,CAAC;IAKxD,yBAAyB,CAAC,EAAE,WAAW,GAAG,SAAS,CAAC;IACpD,cAAc,CAAC,EAAE,cAAc,CAAC;IAChC,sBAAsB,CAAC,EAAE,sBAAsB,CAAC;IAChD,iBAAiB,CAAC,EAAE,CAAC,QAAQ,EAAE,SAAS,WAAW,EAAE,EAAE,MAAM,CAAC,EAAE,WAAW,KAAK,sBAAsB,GAAG,OAAO,CAAC,sBAAsB,CAAC,CAAC;IACzI,YAAY,CAAC,EAAE,CAAC,KAAK,EAAE,aAAa,GAAG,SAAS,KAAK,YAAY,CAAC;IAClE,aAAa,CAAC,EAAE,sBAAsB,CAAC;IACvC,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,YAAY,CAAC,EAAE,YAAY,CAAC;IAG5B,aAAa,CAAC,EAAE,aAAa,CAAC;IAG9B,0BAA0B,CAAC,EAAE,oBAAoB,CAAC;IAGlD,gCAAgC,CAAC,EAAE,oBAAoB,CAAC;IAGxD,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,SAAS,CAAC,EAAE,OAAO,CAAC;IACpB,SAAS,CAAC,EAAE,OAAO,CAAC;IACpB,kBAAkB,CAAC,EAAE,OAAO,CAAC;IAC7B,qBAAqB,CAAC,EAAE,MAAM,CAAC;IAC/B,OAAO,CAAC,EAAE,MAAM,CAAC;IAEjB,mBAAmB,CAAC,EAAE,OAAO,CAAC;IAC9B,SAAS,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAI1B,WAAW,CAAC,EAAE,MAAM,CAAC;IAGrB,eAAe,CAAC,EAAE,MAAM,CAAC;IAKzB,WAAW,CAAC,EAAE,MAAM,CAAC;IAIrB,oBAAoB,CAAC,EAAE,OAAO,CAAC;IAM/B,SAAS,EAAE,SAAS,CAAC;IAOrB,WAAW,EAAE,MAAM,GAAG,IAAI,CAAC;IAC3B,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;IAK7B,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAK1B,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,WAAW,CAAC,EAAE,MAAM,CAAC;IAKrB,aAAa,EAAE,MAAM,CAAC;IAItB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAQ1B,WAAW,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC5B,OAAO,CAAC,EAAE,OAAO,CAAC;IAUlB,YAAY,CAAC,EAAE,OAAO,CAAC;CAC1B,CAAC;AA0GF,MAAM,CAAC,OAAO,OAAO,aAAc,YAAW,QAAQ;;IA0DlD,QAAQ,CAAC,YAAY,CAAC,EAAE,CAAC,OAAO,EAAE,wBAAwB,KAAK,iBAAiB,CAAC;IAMjF,QAAQ,CAAC,EAAE,CAAC,IAAI,EAAE,MAAM,KAAK,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;IAE/C,YAAY,MAAM,EAAE,mBAAmB,EAqKtC;IAED,IAAI,aAAa,IAAI,MAAM,GAAG,IAAI,CAAgC;IAClE,IAAI,eAAe,IAAI,WAAW,CAAC,aAAa,CAAC,CAAkC;IACnF,IAAI,cAAc,IAAI,MAAM,GAAG,IAAI,CAAiC;IACpE,IAAI,eAAe,IAAI,MAAM,GAAG,IAAI,CAAkC;IACtE,IAAI,YAAY,IAAI,MAAM,GAAG,IAAI,CAA+B;IAChE,IAAI,eAAe,IAAI,MAAM,GAAG,IAAI,CAAkC;IACtE,IAAI,0BAA0B,IAAI,SAAS,eAAe,EAAE,CAA6C;IACzG,IAAI,aAAa,IAAI,MAAM,GAAG,IAAI,CAMjC;IACD,IAAI,KAAK,IAAI,MAAM,CAAwB;IAE3C,IAAI,WAAW,IAAI,MAAM,GAAG,SAAS,CAA8B;IAEnE,IAAI,oBAAoB,IAAI,OAAO,GAAG,SAAS,CAAuC;IAItF,IAAI,gBAAgB,IAAI,OAAO,CAA0C;IAEnE,iBAAiB,CACnB,QAAQ,EAAE,SAAS,WAAW,EAAE,EAChC,MAAM,CAAC,EAAE,WAAW,GACrB,OAAO,CAAC,sBAAsB,CAAC,CAqDjC;IAEK,qBAAqB,CACvB,QAAQ,EAAE,SAAS,WAAW,EAAE,EAChC,eAAe,CAAC,EAAE,MAAM,EACxB,MAAM,CAAC,EAAE,WAAW,GACrB,OAAO,CAAC,uBAAuB,CAAC,CAmBlC;IA4BK,QAAQ,CAAC,EAAE,QAAQ,EAAE,QAAQ,EAAE,eAAe,EAAE,MAAM,EAAE,OAAO,EAAE,eAAe,EAAE,YAAY,EAAE,MAAM,EAAE,OAAO,EAAE,WAAW,EAAE,IAAI,EAAE,IAAI,EAAE,QAAQ,EAAE,cAAc,EAAE,gBAAgB,EAAE,QAAQ,EAAE,EAAE,oBAAoB,GAAG,OAAO,CAAC,gBAAgB,CAAC,CAwcvP;CAEJ"}
@@ -8,27 +8,41 @@
8
8
  import { REASONING_POLICIES } from "@plurnk/plurnk-contracts";
9
9
  import { MAX_PROVIDER_TIMEOUT_MS } from "./env.js";
10
10
  import { UnsupportedReasoningPolicyError } from "./types.js";
11
- import { executeAiSdkModel, executeOpenAICompatible, transportFailureOutputObserved, transportFailureEvidence, } from "./aiSdkTransport.js";
11
+ import { executeAiSdkModel, executeOpenAICompatible, transportFailureOutputObserved, transportFailureEvidence } from "./aiSdkTransport.js";
12
12
  import { prepareRetries } from "ai/internal";
13
13
  import { toProviderError, ProviderError, ProviderTimeoutError } from "./errors.js";
14
- import { validateGbnf } from "@plurnk/gbnf";
15
14
  import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.js";
16
15
  import { emitWarningOnce } from "./warnings.js";
17
16
  import { resolveProviderCost } from "./cost.js";
18
17
  import { validateProviderRequestAccounting } from "./accounting.js";
19
18
  import { validateProviderUsage } from "./usage.js";
20
19
  import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.js";
21
- const isJsonObject = (value) => typeof value === "object" && value !== null && !Array.isArray(value);
22
- const mergeJsonObjects = (left, right) => Object.fromEntries([...new Set([...Object.keys(left), ...Object.keys(right)])].map((key) => {
23
- const leftValue = left[key];
24
- const rightValue = right[key];
25
- return [
26
- key,
27
- isJsonObject(leftValue) && isJsonObject(rightValue)
28
- ? mergeJsonObjects(leftValue, rightValue)
29
- : rightValue ?? leftValue,
30
- ];
31
- }));
20
+ import { nativeFixedEffort } from "./reasoning-effort.js";
21
+ import AiSdkRequestBody from "./AiSdkRequestBody.js";
22
+ // {§provider-connectivity} — an AbortSignal is advisory: a wedged transport that never observes it can
23
+ // hang the await past the deadline (#505). This backstops the deadline the same way an embedding request
24
+ // bounds a wedged adapter (#463): once the signal fires, a well-behaved transport unwinds and settles its
25
+ // own attempt (and accounting) within a short grace, and that path wins the race untouched; only a
26
+ // transport still wedged after the grace is force-rejected with the signal's own reason, so the existing
27
+ // operation/cancellation classification is unchanged. The grace is unwind slack, not a second deadline.
28
+ const OPERATION_DEADLINE_UNWIND_GRACE_MS = 1_000;
29
+ const raceAgainstDeadline = async (work, signal) => {
30
+ let timer;
31
+ const backstop = new Promise((_resolve, reject) => {
32
+ const arm = () => { timer = setTimeout(() => reject(signal.reason), OPERATION_DEADLINE_UNWIND_GRACE_MS); };
33
+ if (signal.aborted)
34
+ arm();
35
+ else
36
+ signal.addEventListener("abort", arm, { once: true });
37
+ });
38
+ try {
39
+ return await Promise.race([work, backstop]);
40
+ }
41
+ finally {
42
+ if (timer !== undefined)
43
+ clearTimeout(timer);
44
+ }
45
+ };
32
46
  class ProviderRequestObserverError extends Error {
33
47
  constructor(cause) {
34
48
  super("provider request accounting could not be durably settled", { cause });
@@ -101,32 +115,6 @@ const projectTemplateReasoning = (content) => {
101
115
  }
102
116
  return { content, reasoning: "", projected: false, contentStart: 0 };
103
117
  };
104
- const fixedEffort = (mode) => {
105
- if (mode === "low" || mode === "medium" || mode === "high" || mode === "xhigh" || mode === "max")
106
- return mode;
107
- throw new TypeError(`reasoning policy '${mode}' is not a fixed effort`);
108
- };
109
- // The native SDK effort surface tops at xhigh; admission never grants a native
110
- // route "max", so reaching it here is a contract violation, not a fallback site.
111
- const nativeFixedEffort = (mode) => {
112
- const effort = fixedEffort(mode);
113
- if (effort === "max")
114
- throw new TypeError(`reasoning policy 'max' has no native SDK effort surface`);
115
- return effort;
116
- };
117
- // Anthropic's older manual-reasoning protocol needs an absolute allowance while
118
- // PLURNK's durable contract names an effort. These fractions match the native
119
- // SDK's policy projection, but apply to PLURNK's total envelope rather than the
120
- // model's physical maximum. The minimum is imposed by the provider protocol.
121
- const MANUAL_REASONING_FRACTIONS = Object.freeze({
122
- adaptive: 0.6,
123
- low: 0.1,
124
- medium: 0.3,
125
- high: 0.6,
126
- xhigh: 0.75,
127
- max: 0.85,
128
- });
129
- const MANUAL_REASONING_MINIMUM = 1024;
130
118
  const providerWarningMessage = (warning) => {
131
119
  switch (warning.type) {
132
120
  case "unsupported":
@@ -136,27 +124,6 @@ const providerWarningMessage = (warning) => {
136
124
  case "other": return warning.message;
137
125
  }
138
126
  };
139
- // Body keys the provider owns — a caller's `sampling` passthrough may not set
140
- // these. Two families:
141
- // transport/managed — grammar transport, the stream/JSON choice, slot pinning,
142
- // data capture ({§provider-evidence}: backend-specific fields never cross the contract);
143
- // contract invariants — `n` (atomic single completion: choices[0] is the
144
- // response; n>1 = paid, dropped output), the tool-calling family (tools-in-
145
- // body doctrine, §2: native tool_calls return null content = a broken turn),
146
- // modalities/audio (text-only contract), prediction (decode semantics, not
147
- // sampling), and the token caps (the envelope is the managed maxOutputTokens —
148
- // sampling must not bypass the consumer's cap).
149
- // Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
150
- // platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
151
- // metadata, store, verbosity) pass through; the managed floors spread UNDER
152
- // sampling stay deliberately caller-overridable.
153
- const RESERVED_BODY_KEYS = new Set([
154
- "model", "messages", "stream", "stream_options", "grammar", "response_format", "id_slot", "logprobs", "top_logprobs",
155
- "reasoning_format", "reasoning_effort", "thinking", "think", "include_reasoning", "chat_template_kwargs", "thinking_budget_tokens", // lexicon-allow: backend wire fields
156
- "n", "tools", "tool_choice", "functions", "function_call", "parallel_tool_calls",
157
- "modalities", "audio", "prediction", "max_tokens", "max_completion_tokens",
158
- "prompt_cache_key",
159
- ]);
160
127
  export default class AiSdkProvider {
161
128
  #model;
162
129
  #url;
@@ -171,6 +138,7 @@ export default class AiSdkProvider {
171
138
  #apiKeyRejectedMessage;
172
139
  #eosText;
173
140
  #contextWindow;
141
+ #inputModalities;
174
142
  #maxInputTokens;
175
143
  #maxOutputTokens;
176
144
  #outputBudget;
@@ -220,6 +188,7 @@ export default class AiSdkProvider {
220
188
  // tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
221
189
  // the honest capability signal for every other backend.
222
190
  tokenize;
191
+ #requestBody;
223
192
  constructor(config) {
224
193
  this.#model = config.model;
225
194
  this.#url = config.url;
@@ -245,6 +214,7 @@ export default class AiSdkProvider {
245
214
  this.#headers = config.headers ?? {};
246
215
  this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
247
216
  this.#contextWindow = config.contextWindow ?? null;
217
+ this.#inputModalities = config.inputModalities ?? new Set();
248
218
  this.#maxInputTokens = config.maxInputTokens ?? null;
249
219
  this.#maxOutputTokens = config.maxOutputTokens ?? null;
250
220
  this.#outputBudget = config.outputBudget ?? null;
@@ -377,8 +347,10 @@ export default class AiSdkProvider {
377
347
  return tokens;
378
348
  };
379
349
  }
350
+ this.#requestBody = new AiSdkRequestBody({ reasoningBudget: this.#reasoningBudget, additiveReasoningProvider: this.#additiveReasoningProvider, reasoning: this.#reasoning, reasoningToggle: this.#reasoningToggle, compatibleAdaptiveReasoning: this.#compatibleAdaptiveReasoning, compatibleOffReasoning: this.#compatibleOffReasoning, adaptiveReasoningProviderOptions: this.#adaptiveReasoningProviderOptions, repeatPenalty: this.#repeatPenalty, frequencyPenalty: this.#frequencyPenalty, dryMultiplier: this.#dryMultiplier, dryBase: this.#dryBase, dryAllowedLength: this.#dryAllowedLength, repeatLastN: this.#repeatLastN, reasoningStyle: this.#reasoningStyle, source: this.#source, grammarStyle: this.#grammarStyle, cacheAffinity: this.#cacheAffinity, reasoningResponseProviderOptions: this.#reasoningResponseProviderOptions, firstPartyMetadata: this.#firstPartyMetadata, supportsSlotPinning: this.#supportsSlotPinning, slotCount: this.#slotCount });
380
351
  }
381
352
  get contextWindow() { return this.#contextWindow; }
353
+ get inputModalities() { return this.#inputModalities; }
382
354
  get maxInputTokens() { return this.#maxInputTokens; }
383
355
  get maxOutputTokens() { return this.#maxOutputTokens; }
384
356
  get outputBudget() { return this.#outputBudget; }
@@ -420,7 +392,7 @@ export default class AiSdkProvider {
420
392
  body: JSON.stringify({
421
393
  model: this.#model,
422
394
  messages,
423
- ...this.#reasoningBody(),
395
+ ...this.#requestBody.reasoningBody(),
424
396
  }),
425
397
  ...(requestSignal === undefined ? {} : { signal: requestSignal }),
426
398
  });
@@ -462,208 +434,6 @@ export default class AiSdkProvider {
462
434
  measurement: await this.countPromptTokens(messages, signal),
463
435
  });
464
436
  }
465
- // Reasoning activation and allowance are independent of grammar transport;
466
- // only the response representation becomes lossless when evidence is needed.
467
- // The llama-server template mapping is owned by {§llama-reasoning-request}.
468
- #reasoningBody(preserveGrammarSentence = false, reasoningBudget = this.#reasoningBudget) {
469
- const { mode } = this.#reasoning;
470
- const budget = reasoningBudget;
471
- const on = mode !== "off";
472
- switch (this.#reasoningStyle) {
473
- case "template": {
474
- const allowance = mode === "off"
475
- ? 0
476
- : budget;
477
- // A fixed effort rides into the template as its own variable; adaptive
478
- // and off send none and leave the template's default in force.
479
- const templateEffort = mode === "off" || mode === "adaptive" ? {} : { reasoning_effort: fixedEffort(mode) };
480
- return {
481
- chat_template_kwargs: { enable_thinking: on, ...templateEffort },
482
- reasoning_format: preserveGrammarSentence ? "none" : "auto",
483
- ...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
484
- };
485
- }
486
- case "think": return on ? { think: true } : {};
487
- case "include_reasoning": return on ? { include_reasoning: true } : {};
488
- case "effort": return mode === "off"
489
- ? this.#compatibleOffReasoning === undefined
490
- ? {}
491
- : { reasoning_effort: this.#compatibleOffReasoning }
492
- : mode === "adaptive"
493
- ? this.#compatibleAdaptiveReasoning === "provider-default"
494
- ? {}
495
- : { reasoning_effort: this.#compatibleAdaptiveReasoning }
496
- : { reasoning_effort: fixedEffort(mode) };
497
- // Graded reasoning is mandatory when the route advertises an effort
498
- // value. Cataloged routes supply the exact strongest legal value;
499
- // construction rejects an unsupported off or fixed policy.
500
- case "effort_required": {
501
- if (mode === "off") {
502
- if (this.#compatibleOffReasoning === undefined) {
503
- throw new TypeError(`${this.#source}: required reasoning effort has no off projection`);
504
- }
505
- return { reasoning_effort: this.#compatibleOffReasoning };
506
- }
507
- if (mode === "adaptive") {
508
- return this.#compatibleAdaptiveReasoning === "provider-default"
509
- ? {}
510
- : { reasoning_effort: this.#compatibleAdaptiveReasoning };
511
- }
512
- return { reasoning_effort: fixedEffort(mode) };
513
- }
514
- // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
515
- // reason-by-default model (DeepSeek V4: default 'high') reasoning.
516
- // ADAPTIVE omits the field UNLESS the catalog declares a toggle control:
517
- // toggle routes (nemotron-lightning) default reasoning OFF, so adaptive
518
- // sends the documented Fireworks Boolean enable (#457). The literal
519
- // "adaptive" is MiniMax-M3-only — Fireworks 400s it for every other
520
- // model (wire-verified; the 1.0.2 adaptive default refused to boot on
521
- // it). V4 gotcha: integer efforts 400.
522
- case "effort_explicit": return mode === "off"
523
- ? { reasoning_effort: "none" }
524
- : mode === "adaptive"
525
- ? this.#reasoningToggle ? { reasoning_effort: true } : {}
526
- : { reasoning_effort: fixedEffort(mode) };
527
- // {§deepseek-reasoning-request}
528
- case "thinking_effort": return mode === "off"
529
- ? { thinking: { type: "disabled" } }
530
- : mode === "adaptive" ? { thinking: { type: "enabled" } } : {
531
- thinking: { type: "enabled" },
532
- reasoning_effort: fixedEffort(mode),
533
- };
534
- // Anthropic-compatible native dynamic or manual budget mode.
535
- case "anthropic": return mode === "off"
536
- ? { thinking: { type: "disabled" } }
537
- : mode === "adaptive" ? { thinking: { type: "adaptive" } } : {
538
- thinking: {
539
- type: "enabled",
540
- budget_tokens: budget,
541
- },
542
- };
543
- case "none": return {};
544
- }
545
- }
546
- // Per-worker slot affinity: the consumer passes which worker this is; the
547
- // provider owns WHICH slot serves it. Sticky per workerId, round-robin across
548
- // new runs (distinct runs → distinct slots while slots last), LRU-bounded
549
- // bookkeeping so a long-lived daemon never grows the map unboundedly —
550
- // an evicted-and-returning run simply re-pins, worst case one cold prefill.
551
- #runSlots = new Map();
552
- #nextSlot = 0;
553
- #slotBody(workerId) {
554
- if (!this.#supportsSlotPinning || this.#slotCount === null || this.#slotCount < 1)
555
- return {};
556
- let slot = this.#runSlots.get(workerId);
557
- if (slot === undefined) {
558
- slot = this.#nextSlot++ % this.#slotCount;
559
- if (this.#runSlots.size >= this.#slotCount * 8) {
560
- this.#runSlots.delete(this.#runSlots.keys().next().value);
561
- }
562
- }
563
- else {
564
- this.#runSlots.delete(workerId); // re-insert to refresh LRU recency
565
- }
566
- this.#runSlots.set(workerId, slot);
567
- return { id_slot: slot };
568
- }
569
- // Optional local llama-server GBNF transport ({§gbnf-response-observation}). Unsupported
570
- // backends receive no grammar-related field.
571
- #grammarBody(grammar) {
572
- if (grammar === undefined)
573
- return {};
574
- switch (this.#grammarStyle) {
575
- // Grammar-constrained decoding can loop under the mask; a configured
576
- // per-alias repeat_penalty is the measured remedy ({§provider-sampling-passthrough}).
577
- case "llamacpp": return { grammar, ...(this.#repeatPenalty !== null ? { repeat_penalty: this.#repeatPenalty } : {}) };
578
- case "none": return {};
579
- }
580
- }
581
- // Anti-degeneration default on every request, keyed to the backend's wire
582
- // convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
583
- // alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
584
- // straight to the token cap on pure looped repetition (run52). Ships next to
585
- // temperature so caller `sampling` can tune it; the grammar path re-asserts it as a
586
- // managed FLOOR in #grammarBody. llama.cpp takes the repeat_penalty
587
- // MULTIPLIER; the plain cloud path ("none") can't, so it gets
588
- // frequency_penalty - OpenAI-standard, accepted by every OpenAI-compat backend (verified
589
- // live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
590
- #repetitionPenaltyBody() {
591
- switch (this.#grammarStyle) {
592
- // repeat_penalty + optional DRY (repeated-sequence penalty) + a wider
593
- // repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
594
- // Each rides only when its operator knob is set; absent = the box's default.
595
- case "llamacpp": return {
596
- ...(this.#repeatPenalty !== null ? { repeat_penalty: this.#repeatPenalty } : {}),
597
- ...(this.#repeatLastN !== undefined ? { repeat_last_n: this.#repeatLastN } : {}),
598
- ...(this.#dryMultiplier !== undefined && this.#dryMultiplier > 0 ? {
599
- dry_multiplier: this.#dryMultiplier,
600
- ...(this.#dryBase !== undefined ? { dry_base: this.#dryBase } : {}),
601
- ...(this.#dryAllowedLength !== undefined ? { dry_allowed_length: this.#dryAllowedLength } : {}),
602
- } : {}),
603
- };
604
- case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
605
- }
606
- }
607
- // First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
608
- // opted in (the plurnk endpoint). The gate is here, not at the call site, so
609
- // attributions/client/strikes can never reach a third-party backend even if
610
- // the consumer passes them to the wrong provider. Empty values emit no header
611
- // — EXCEPT strikes, where 0 is a real value (clean streak) distinct from
612
- // absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
613
- // ride HTTP headers only — the packet never carries them (the model must
614
- // never see strike state; engine accounting is not a metric to game).
615
- #metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind) {
616
- if (!this.#firstPartyMetadata)
617
- return {};
618
- const h = {};
619
- if (attributions !== undefined && attributions.length > 0)
620
- h["Plurnk-Attribution"] = JSON.stringify(attributions);
621
- if (client !== undefined && client.length > 0)
622
- h["Plurnk-Client"] = client;
623
- if (strikes !== undefined && Number.isInteger(strikes) && strikes >= 0)
624
- h["Plurnk-Strikes"] = String(strikes);
625
- // Worker identity: the opaque workerId
626
- // the consumer already supplies, forwarded so the endpoint can key
627
- // per-worker affinity/telemetry — same gate as every first-party signal.
628
- h["Plurnk-Worker-Id"] = workerId;
629
- // Root worker of the lineage ({§worker-primary}): the no-parent ancestor of this turn's
630
- // worker tree. The consumer classifies primary-vs-spawned by equality
631
- // (primaryWorkerId == workerId ⇒ the primary/root worker). The provider
632
- // EMITS what the consumer supplies and never invents a primary; the
633
- // consumer's contract is to stamp it EVERY turn (including the primary's
634
- // own, where it equals workerId). Absence is the consumer's violation for
635
- // the endpoint to surface, not a provider default.
636
- if (primaryWorkerId !== undefined && primaryWorkerId.length > 0)
637
- h["Plurnk-Worker-Primary"] = primaryWorkerId;
638
- // Turn coordinate ({§lifecycle-terms}): workspace/loop/turn, the
639
- // daemon-side sequence the endpoint can never scrape from the wire.
640
- // Coordinates are 1-based — 0 is not a real value, so no strikes-style
641
- // zero exception; absent/empty/0 emits no header.
642
- if (workspaceId !== undefined && workspaceId.length > 0)
643
- h["Plurnk-Workspace-Id"] = workspaceId;
644
- if (loop !== undefined && Number.isInteger(loop) && loop >= 1)
645
- h["Plurnk-Loop"] = String(loop);
646
- if (turn !== undefined && Number.isInteger(turn) && turn >= 1)
647
- h["Plurnk-Turn"] = String(turn);
648
- if (callKind !== undefined)
649
- h["Plurnk-Call-Kind"] = callKind;
650
- return h;
651
- }
652
- // PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
653
- // hard if it's malformed, BEFORE any wire call — and the grammar is NOT
654
- // transported, so the request runs unconstrained. A debug aid to catch invalid
655
- // grammars (e.g. while editing the plurnk grammar) without a model round-trip;
656
- // off in production. `validateGbnf(grammar, "")` parses the grammar + resolves
657
- // its root, throwing iff the grammar itself is invalid (the empty input's
658
- // verdict is irrelevant — we only care that parsing succeeded).
659
- #assertGrammarValid(grammar) {
660
- try {
661
- validateGbnf(grammar, "");
662
- }
663
- catch (cause) {
664
- throw new Error(`grammar validation (PLURNK_PROVIDERS_GBNF_DEBUG): invalid GBNF — ${cause.message}`, { cause });
665
- }
666
- }
667
437
  // Per-turn metadata bag: pass the backend's non-standard top-level fields
668
438
  // through verbatim. Providers do not reinterpret vendor currency or account
669
439
  // metadata; a monetary value carries its own amount and currency.
@@ -671,71 +441,6 @@ export default class AiSdkProvider {
671
441
  const meta = { ...chunkMetadata };
672
442
  return Object.keys(meta).length > 0 ? meta : undefined;
673
443
  }
674
- // Caller-supplied OpenAI-compat sampling params (temperature, top_p, top_k,
675
- // penalties, stop, seed, …) merged UNDER the managed body: model, messages,
676
- // reasoning, grammar (+ its repeat-penalty floor), max_tokens and slot always
677
- // win, and reserved transport/protocol keys are stripped so the passthrough
678
- // can't smuggle a grammar, a stream toggle, or a backend slot
679
- // ({§provider-request-authority}).
680
- #samplingBody(sampling) {
681
- if (sampling === undefined)
682
- return {};
683
- const out = {};
684
- for (const [k, v] of Object.entries(sampling))
685
- if (!RESERVED_BODY_KEYS.has(k))
686
- out[k] = v;
687
- return out;
688
- }
689
- #requestProviderOptions(workerId, nativeReasoningBudget) {
690
- const responseOptions = this.#reasoning.mode === "off"
691
- ? undefined
692
- : this.#reasoningResponseProviderOptions;
693
- const adaptiveOptions = this.#reasoning.mode === "adaptive"
694
- && nativeReasoningBudget === null
695
- ? this.#adaptiveReasoningProviderOptions
696
- : undefined;
697
- const nativeReasoning = nativeReasoningBudget !== null
698
- ? this.#additiveReasoningProvider === "anthropic"
699
- ? { anthropic: { thinking: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
700
- : this.#additiveReasoningProvider === "bedrock"
701
- ? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
702
- : undefined
703
- : undefined;
704
- const options = {};
705
- for (const part of [responseOptions, adaptiveOptions, nativeReasoning]) {
706
- for (const [provider, values] of Object.entries(part ?? {})) {
707
- options[provider] = mergeJsonObjects(options[provider] ?? {}, values);
708
- }
709
- }
710
- if (this.#cacheAffinity?.target === "provider-option") {
711
- const { provider, name } = this.#cacheAffinity;
712
- options[provider] = { ...options[provider], [name]: workerId };
713
- }
714
- return Object.keys(options).length === 0 ? undefined : options;
715
- }
716
- #nativeMaxOutputTokens(outputBudget, nativeReasoningBudget) {
717
- if (outputBudget === null)
718
- return undefined;
719
- return nativeReasoningBudget !== null
720
- ? outputBudget - nativeReasoningBudget
721
- : outputBudget;
722
- }
723
- #nativeReasoningBudget(outputBudget, configuredReasoningBudget) {
724
- if (this.#additiveReasoningProvider === undefined || this.#reasoning.mode === "off")
725
- return null;
726
- if (configuredReasoningBudget !== null)
727
- return configuredReasoningBudget;
728
- if (this.#adaptiveReasoningProviderOptions !== undefined)
729
- return null;
730
- if (outputBudget === null) {
731
- throw new TypeError(`${this.#source}: manual provider reasoning requires a resolved total output budget`);
732
- }
733
- if (outputBudget <= MANUAL_REASONING_MINIMUM) {
734
- throw new TypeError(`${this.#source}: total output budget must exceed the provider's ${MANUAL_REASONING_MINIMUM}-token minimum reasoning allowance`);
735
- }
736
- const fraction = MANUAL_REASONING_FRACTIONS[this.#reasoning.mode];
737
- return Math.min(outputBudget - 1, Math.max(MANUAL_REASONING_MINIMUM, Math.round(outputBudget * fraction)));
738
- }
739
444
  #accounting(outcome, usage, evidence, status) {
740
445
  const knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
741
446
  const direct = this.#normalizeCost?.(evidence);
@@ -762,7 +467,7 @@ export default class AiSdkProvider {
762
467
  // supplied grammar before the call but withholds it from the backend.
763
468
  const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
764
469
  if (wantGrammar && this.#gbnfDebug)
765
- this.#assertGrammarValid(grammar);
470
+ this.#requestBody.assertGrammarValid(grammar);
766
471
  const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
767
472
  const preserveGrammarSentence = wantGrammar
768
473
  && this.#reasoningStyle === "template";
@@ -776,7 +481,7 @@ export default class AiSdkProvider {
776
481
  // {§provider-flexed-allowance} (#482): the wire grants the flexed
777
482
  // allowance — the floor, or the exactly-measured slack above it.
778
483
  const effectiveMaxOutputTokens = capacity.responseMax ?? capacity.outputBudget ?? undefined;
779
- const nativeReasoningBudget = this.#nativeReasoningBudget(capacity.outputBudget, capacity.reasoningBudget);
484
+ const nativeReasoningBudget = this.#requestBody.nativeReasoningBudget(capacity.outputBudget, capacity.reasoningBudget);
780
485
  // Assembly order = precedence: the family's sampling DEFAULTS
781
486
  // (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
782
487
  // paths and the name promises every request) < the caller's `sampling`
@@ -784,24 +489,24 @@ export default class AiSdkProvider {
784
489
  const body = {
785
490
  // Floors are suppressed on router-owned-tuning providers (plurnk) —
786
491
  // the router's per-model tuning must not be overridden by client floors.
787
- ...(this.#tuningFloors ? { ...(this.#temperature !== null ? { temperature: this.#temperature } : {}), ...this.#repetitionPenaltyBody() } : {}),
788
- ...this.#samplingBody(sampling),
492
+ ...(this.#tuningFloors ? { ...(this.#temperature !== null ? { temperature: this.#temperature } : {}), ...this.#requestBody.repetitionPenaltyBody() } : {}),
493
+ ...this.#requestBody.samplingBody(sampling),
789
494
  ...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
790
495
  model: this.#model,
791
496
  messages,
792
- ...this.#reasoningBody(preserveGrammarSentence, capacity.reasoningBudget),
793
- ...this.#grammarBody(sendGrammar),
497
+ ...this.#requestBody.reasoningBody(preserveGrammarSentence, capacity.reasoningBudget),
498
+ ...this.#requestBody.grammarBody(sendGrammar),
794
499
  ...(effectiveMaxOutputTokens !== undefined ? { max_tokens: effectiveMaxOutputTokens } : {}),
795
500
  // Request per-token logprobs only when enabled (managed field —
796
501
  // reserved from caller sampling; the env flag is the single control).
797
502
  ...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
798
- ...this.#slotBody(workerId),
503
+ ...this.#requestBody.slotBody(workerId),
799
504
  ...(this.#cacheAffinity?.target === "body"
800
505
  ? { [this.#cacheAffinity.name]: workerId }
801
506
  : {}),
802
507
  };
803
508
  // Per-request headers = static auth/routing + any first-party telemetry.
804
- const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
509
+ const metaHeaders = this.#requestBody.metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
805
510
  const headers = new Headers(this.#headers);
806
511
  if (this.#cacheAffinity?.target === "header") {
807
512
  headers.set(this.#cacheAffinity.name, workerId);
@@ -910,7 +615,7 @@ export default class AiSdkProvider {
910
615
  : await executeAiSdkModel({
911
616
  languageModel: this.#languageModel,
912
617
  headers: requestHeaders,
913
- providerOptions: this.#requestProviderOptions(workerId, nativeReasoningBudget),
618
+ providerOptions: this.#requestBody.requestProviderOptions(workerId, nativeReasoningBudget),
914
619
  systemProviderOptions: this.#systemCacheProviderOptions,
915
620
  messages,
916
621
  signal: operationSignal,
@@ -935,7 +640,7 @@ export default class AiSdkProvider {
935
640
  ? sampling.stop
936
641
  : undefined,
937
642
  seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
938
- maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget, nativeReasoningBudget),
643
+ maxOutputTokens: this.#requestBody.nativeMaxOutputTokens(capacity.outputBudget, nativeReasoningBudget),
939
644
  reasoning: this.#reasoning.mode === "off"
940
645
  ? "none"
941
646
  : this.#reasoning.mode === "adaptive"
@@ -960,7 +665,13 @@ export default class AiSdkProvider {
960
665
  maxRetries: this.#retryAttempts,
961
666
  abortSignal: operationSignal,
962
667
  });
963
- raw = await retry(executeRequest);
668
+ // {§provider-connectivity} — the operation deadline and caller cancellation are backstopped by
669
+ // racing, not only by the advisory operationSignal, so a transport that ignores the abort still
670
+ // surfaces the deadline (→ the operationTimeout/caller branches below) rather than hanging the
671
+ // loop (#505). A well-behaved transport settles first and wins the race with its own evidence.
672
+ raw = operationSignal === undefined
673
+ ? await retry(executeRequest)
674
+ : await raceAgainstDeadline(retry(executeRequest), operationSignal);
964
675
  }
965
676
  catch (err) {
966
677
  if (err instanceof ProviderRequestObserverError