@plurnk/plurnk-providers 1.21.1 → 1.22.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. package/.env.defaults +101 -23
  2. package/README.md +10 -6
  3. package/SPEC.md +182 -69
  4. package/dist/AiSdkProvider.d.ts +15 -13
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +130 -52
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/AiSdkRequestBody.d.ts +7 -10
  9. package/dist/AiSdkRequestBody.d.ts.map +1 -1
  10. package/dist/AiSdkRequestBody.js +16 -78
  11. package/dist/AiSdkRequestBody.js.map +1 -1
  12. package/dist/InferenceAdmission.d.ts +7 -0
  13. package/dist/InferenceAdmission.d.ts.map +1 -0
  14. package/dist/InferenceAdmission.js +58 -0
  15. package/dist/InferenceAdmission.js.map +1 -0
  16. package/dist/Mock.d.ts +1 -1
  17. package/dist/Mock.d.ts.map +1 -1
  18. package/dist/Mock.js +2 -2
  19. package/dist/Mock.js.map +1 -1
  20. package/dist/Pool.d.ts +2 -2
  21. package/dist/Pool.d.ts.map +1 -1
  22. package/dist/Pool.js +2 -2
  23. package/dist/Pool.js.map +1 -1
  24. package/dist/RepeatedLine.d.ts +9 -0
  25. package/dist/RepeatedLine.d.ts.map +1 -0
  26. package/dist/RepeatedLine.js +27 -0
  27. package/dist/RepeatedLine.js.map +1 -0
  28. package/dist/RequestFields.d.ts +25 -0
  29. package/dist/RequestFields.d.ts.map +1 -0
  30. package/dist/RequestFields.js +240 -0
  31. package/dist/RequestFields.js.map +1 -0
  32. package/dist/aiSdkTransport.d.ts +7 -3
  33. package/dist/aiSdkTransport.d.ts.map +1 -1
  34. package/dist/aiSdkTransport.js +56 -15
  35. package/dist/aiSdkTransport.js.map +1 -1
  36. package/dist/catalogProvider.d.ts +4 -3
  37. package/dist/catalogProvider.d.ts.map +1 -1
  38. package/dist/catalogProvider.js +72 -168
  39. package/dist/catalogProvider.js.map +1 -1
  40. package/dist/compatibleProvider.d.ts.map +1 -1
  41. package/dist/compatibleProvider.js +10 -7
  42. package/dist/compatibleProvider.js.map +1 -1
  43. package/dist/discover.d.ts +0 -1
  44. package/dist/discover.d.ts.map +1 -1
  45. package/dist/discover.js +1 -10
  46. package/dist/discover.js.map +1 -1
  47. package/dist/env.d.ts +9 -6
  48. package/dist/env.d.ts.map +1 -1
  49. package/dist/env.js +97 -15
  50. package/dist/env.js.map +1 -1
  51. package/dist/errors.d.ts +1 -8
  52. package/dist/errors.d.ts.map +1 -1
  53. package/dist/errors.js +9 -13
  54. package/dist/errors.js.map +1 -1
  55. package/dist/index.d.ts +8 -7
  56. package/dist/index.d.ts.map +1 -1
  57. package/dist/index.js +6 -5
  58. package/dist/index.js.map +1 -1
  59. package/dist/model-options.d.ts +3 -0
  60. package/dist/model-options.d.ts.map +1 -0
  61. package/dist/model-options.js +26 -0
  62. package/dist/model-options.js.map +1 -0
  63. package/dist/ollama.d.ts.map +1 -1
  64. package/dist/ollama.js +1 -0
  65. package/dist/ollama.js.map +1 -1
  66. package/dist/provider-env.d.ts +3 -0
  67. package/dist/provider-env.d.ts.map +1 -0
  68. package/dist/provider-env.js +8 -0
  69. package/dist/provider-env.js.map +1 -0
  70. package/dist/providerError.d.ts +2 -2
  71. package/dist/providerError.d.ts.map +1 -1
  72. package/dist/providerError.js +19 -22
  73. package/dist/providerError.js.map +1 -1
  74. package/dist/reasoning-effort.d.ts +4 -3
  75. package/dist/reasoning-effort.d.ts.map +1 -1
  76. package/dist/reasoning-effort.js +18 -2
  77. package/dist/reasoning-effort.js.map +1 -1
  78. package/dist/sdkModels.d.ts +3 -2
  79. package/dist/sdkModels.d.ts.map +1 -1
  80. package/dist/sdkModels.js +39 -66
  81. package/dist/sdkModels.js.map +1 -1
  82. package/dist/types.d.ts +8 -8
  83. package/dist/types.d.ts.map +1 -1
  84. package/dist/types.js +5 -5
  85. package/dist/types.js.map +1 -1
  86. package/package.json +6 -7
  87. package/src/AiSdkProvider.test.ts +0 -2843
  88. package/src/AiSdkProvider.ts +0 -1064
  89. package/src/AiSdkRequestBody.ts +0 -353
  90. package/src/LeadingReasoning.test.ts +0 -34
  91. package/src/LeadingReasoning.ts +0 -87
  92. package/src/Mock.test.ts +0 -209
  93. package/src/Mock.ts +0 -197
  94. package/src/Pool.test.ts +0 -267
  95. package/src/Pool.ts +0 -261
  96. package/src/ProviderRegistry.test.ts +0 -483
  97. package/src/ProviderRegistry.ts +0 -128
  98. package/src/accounting.test.ts +0 -212
  99. package/src/accounting.ts +0 -202
  100. package/src/accountingPublic.ts +0 -9
  101. package/src/aiSdkTransport.test.ts +0 -573
  102. package/src/aiSdkTransport.ts +0 -790
  103. package/src/boundaries.test.ts +0 -73
  104. package/src/capacity.test.ts +0 -124
  105. package/src/capacity.ts +0 -174
  106. package/src/catalogProvider.test.ts +0 -928
  107. package/src/catalogProvider.ts +0 -475
  108. package/src/compatibleProvider.test.ts +0 -157
  109. package/src/compatibleProvider.ts +0 -216
  110. package/src/cost.test.ts +0 -114
  111. package/src/cost.ts +0 -140
  112. package/src/defaults.test.ts +0 -30
  113. package/src/defaults.ts +0 -9
  114. package/src/discover.test.ts +0 -200
  115. package/src/discover.ts +0 -136
  116. package/src/env.test.ts +0 -335
  117. package/src/env.ts +0 -402
  118. package/src/errors.test.ts +0 -278
  119. package/src/errors.ts +0 -234
  120. package/src/index.ts +0 -100
  121. package/src/inputModalities.test.ts +0 -54
  122. package/src/lexicon-guard.test.ts +0 -58
  123. package/src/notices.ts +0 -22
  124. package/src/ollama.test.ts +0 -66
  125. package/src/ollama.ts +0 -66
  126. package/src/openai.ts +0 -15
  127. package/src/promptTokens.ts +0 -45
  128. package/src/providerDefaults.test.ts +0 -51
  129. package/src/providerError.ts +0 -143
  130. package/src/reasoning-effort.ts +0 -15
  131. package/src/routerAccounting.test.ts +0 -137
  132. package/src/sampling.test.ts +0 -178
  133. package/src/sdkModels.test.ts +0 -293
  134. package/src/sdkModels.ts +0 -460
  135. package/src/types.ts +0 -337
  136. package/src/usage.test.ts +0 -223
  137. package/src/usage.ts +0 -312
  138. package/src/warnings.test.ts +0 -31
  139. package/src/warnings.ts +0 -0
package/.env.defaults CHANGED
@@ -11,11 +11,15 @@
11
11
  # PLURNK_MODEL_local=openai/model-name-from-endpoint
12
12
  # PLURNK_BASEURL_local=http://127.0.0.1:8080/v1
13
13
  # PLURNK_MODEL=cloud
14
- # PLURNK_MODEL=google/gemini-3-flash
14
+ # PLURNK_MODEL=deepseek/deepseek-v4-flash
15
15
 
16
16
  # --- Capacity ---
17
- # Total output envelope INCLUDING reasoning: positive tokens or context percentage in (0,100).
18
- # Known model limits cap it. Must leave input capacity; the context budget is derived, not set here.
17
+ # Simultaneous inference attempts per resolved endpoint in this process; -1 = unrestricted.
18
+ # Positive limits queue excess calls. Aliases sharing an endpoint must agree on its limit.
19
+ PLURNK_PROVIDERS_MAX_CONCURRENCY=-1
20
+ # PLURNK_PROVIDERS_MAX_CONCURRENCY_local=1
21
+ # Reserved output INCLUDING reasoning: positive tokens or context percentage in (0,100).
22
+ # Known model limits cap it; input capacity is derived. Exact-counting calls may grant unused context.
19
23
  PLURNK_PROVIDERS_OUTPUT_BUDGET=35%
20
24
  # Optional reasoning subset of output, using the same units; must be smaller than total output.
21
25
  # Unset/empty = provider-adaptive budget, not reasoning off.
@@ -26,16 +30,29 @@ PLURNK_PROVIDERS_OUTPUT_BUDGET=35%
26
30
 
27
31
  # --- Reasoning ---
28
32
  # off | adaptive | low | medium | high | xhigh | max. Seed for durable Worker reasoning.
29
- # adaptive uses native dynamic reasoning where supported, otherwise the supported high posture.
30
- PLURNK_PROVIDERS_REASONING=adaptive
33
+ # adaptive prefers native dynamic reasoning, then the supported fallback below.
34
+ PLURNK_PROVIDERS_EFFORT=adaptive
35
+ # Fixed effort when native adaptation is unavailable; empty/unsupported = enabled provider default.
36
+ PLURNK_PROVIDERS_EFFORT_FALLBACK=high
31
37
  # Response interpretation: verbatim = trust structured reasoning fields; think-tags = extract one
32
38
  # leading <think> envelope. Select only for an endpoint that emits it; independent of activation.
33
39
  PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE=verbatim
34
- # Reasoning wire for one route, over PLURNK_PROVIDERS_PROVIDER_<NAME>_REASONING_STYLE; usually
35
- # alias-scoped. thinking_config = Gemini through an OpenAI-compatible endpoint (readable thoughts,
36
- # levels low/medium/high, no off). Empty = the provider's declaration.
37
- # PLURNK_PROVIDERS_REASONING_STYLE=
38
- # PLURNK_PROVIDERS_REASONING_STYLE_<alias>=thinking_config
40
+ # Compatible wire declarations below supplement Models.dev; native SDKs own their protocols.
41
+ # Route knobs use the same suffix, e.g. PLURNK_PROVIDERS_REASONING_BUDGET_PATH_<alias>.
42
+ # PLURNK_PROVIDERS_OUTPUT_PATH_<alias>=/max_completion_tokens
43
+ # PLURNK_PROVIDERS_REASONING_EFFORT_PATH_<alias>=/reasoning_effort
44
+ # PLURNK_PROVIDERS_REASONING_CONTROLS_<alias>=exclusive
45
+ # PLURNK_PROVIDERS_REASONING_ON_BODY_<alias>='{"enable_thinking":true}'
46
+ # PLURNK_PROVIDERS_REASONING_OFF_BODY_<alias>='{"enable_thinking":false}'
47
+ # PLURNK_PROVIDERS_REASONING_EFFORTS_<alias>=low,medium,high
48
+ # PLURNK_PROVIDERS_REASONING_TRANSPORT_EFFORTS_<alias>=none,low,medium,high
49
+ # PLURNK_PROVIDERS_OPTIONS_NAMESPACE_<alias>=openrouter
50
+ # PLURNK_PROVIDERS_REASONING_ADAPTIVE_BODY_<alias>='{}'
51
+ # PLURNK_PROVIDERS_REASONING_TOGGLE_BODY_<alias>='{"reasoning":{"enabled":true}}'
52
+ # PLURNK_PROVIDERS_ADAPTIVE_OPTIONS_<alias>='[{"models":["*"],"options":{"<sdk-namespace>":{…}}}]'
53
+ # PLURNK_PROVIDERS_SYSTEM_CACHE_OPTIONS_<alias>=
54
+ # PLURNK_PROVIDERS_APP_URL_<alias>=https://example.test/app
55
+ # PLURNK_PROVIDERS_APP_NAME_<alias>=Example
39
56
 
40
57
  # --- Sampling ---
41
58
  # Temperature; empty = omit, retaining the provider's sampling default.
@@ -64,7 +81,15 @@ PLURNK_PROVIDERS_DRY_MULTIPLIER=0
64
81
  # --- Cache and billing ---
65
82
  # Stable Worker affinity on supported routes: 1 = on; 0 = off. Unknown routes get no guessed field.
66
83
  PLURNK_PROVIDERS_CACHE_AFFINITY=1
67
- # Explicit cache writes: stable-system = reusable system boundary on supported Claude routes;
84
+ # FIELD declares placement, independent of the enable switch; null clears it for an alias.
85
+ # target is header or body with name, or provider-option with provider and name.
86
+ # PLURNK_PROVIDERS_CACHE_AFFINITY_FIELD_<alias>='{"target":"header","name":"x-session-id"}'
87
+ PLURNK_PROVIDERS_PROVIDER_OPENAI_CACHE_AFFINITY_FIELD='{"target":"provider-option","provider":"openai","name":"promptCacheKey"}'
88
+ PLURNK_PROVIDERS_PROVIDER_DEEPINFRA_CACHE_AFFINITY_FIELD='{"target":"provider-option","provider":"deepinfra","name":"prompt_cache_key"}'
89
+ PLURNK_PROVIDERS_PROVIDER_OPENROUTER_CACHE_AFFINITY_FIELD='{"target":"header","name":"x-session-id"}'
90
+ PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_WORKERS_AI_CACHE_AFFINITY_FIELD='{"target":"header","name":"x-session-affinity"}'
91
+ PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_CACHE_AFFINITY_FIELD='{"target":"header","name":"x-session-affinity"}'
92
+ # Explicit cache writes: stable-system = reusable system boundary where SYSTEM_CACHE_OPTIONS declares it;
68
93
  # off = no explicit writes. This is distinct from affinity and can affect billing.
69
94
  PLURNK_PROVIDERS_CACHE_WRITE_POLICY=stable-system
70
95
  # Provider service tier; unset = provider default. Usually set per alias: routing can change cost.
@@ -75,14 +100,21 @@ PLURNK_PROVIDERS_CACHE_WRITE_POLICY=stable-system
75
100
  # PLURNK_PROVIDERS_COST=input=0.22,output=0.66,reasoning=0.66,cacheRead=0.007
76
101
 
77
102
  # --- Connectivity (milliseconds; each deadline accepts 0 = off) ---
78
- # Logical generation deadline, including physical attempts and retry delays.
103
+ # Logical generation deadline, including admission queues, physical attempts and retry delays.
79
104
  PLURNK_PROVIDERS_OPERATION_TIMEOUT=2700000
80
105
  # Deadline per physical generation attempt; a streamed attempt is bounded only until content flows.
81
106
  PLURNK_PROVIDERS_FETCH_TIMEOUT=600000
82
- # Wait from stream start to first semantic content; metadata/empty deltas do not count.
107
+ # Wait from dispatch (after admission, before response headers) to first semantic content;
108
+ # metadata/empty deltas do not count.
83
109
  PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT=180000
84
110
  # Maximum silence between semantic content chunks after content begins.
85
111
  PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT=120000
112
+ # A line of 16+ characters streamed this many times stops the response as a repetition; 0 = off.
113
+ PLURNK_PROVIDERS_REPEATED_LINE_LIMIT=32
114
+ # A response billed for this many more output tokens than it streamed characters (text and reasoning)
115
+ # never delivered its output; it fails as output_dropped and is re-issued (a tool-call finish always
116
+ # does). Reasoning billed but not streamed is exempt. 0 = off.
117
+ PLURNK_PROVIDERS_DROPPED_OUTPUT_TOKENS=32
86
118
  # Retries for 429, Retry-After, or X-Should-Retry:true; 0 = none. Other failures return immediately
87
119
  # to Core's separate recovery window. Retry-After takes precedence over backoff.
88
120
  PLURNK_PROVIDERS_RETRY_ATTEMPTS=3
@@ -112,14 +144,60 @@ PLURNK_PROVIDERS_PROBE_DELAY=250
112
144
  # PLURNK_PROVIDERS_PROVIDER_ACME_NPM=@ai-sdk/openai-compatible
113
145
  # PLURNK_PROVIDERS_PROVIDER_ACME_BASE_URL=https://api.acme.example/v1
114
146
  # PLURNK_PROVIDERS_PROVIDER_ACME_API_KEY_ENV=ACME_API_KEY
115
- # Reasoning wire adaptations that are not expressed by catalog metadata:
116
- # Fireworks requires an explicit off effort; Cloudflare graded models require an effort.
117
- PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_STYLE=effort_explicit
118
- PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_WORKERS_AI_REASONING_STYLE=effort_required
147
+ # Compatible request fields use RFC 6901 pointers; output defaults to the SDK's max_tokens.
148
+ # Effort vocabulary and budget bounds come from Models.dev; EFFORTS adds declared values.
149
+ # ON_BODY applies whenever reasoning is active. OFF_BODY disables it. ADAPTIVE_BODY replaces
150
+ # the graded fallback; {} leaves the enabled endpoint at its native default.
151
+ # TOGGLE_BODY enables a catalog toggle when no adaptive body or supported fallback applies.
152
+ # Bodies contain only reasoning controls, not model/messages, sampling, or token ceilings.
153
+ # With both paths declared, CONTROLS must say exclusive or combined. Fixed effort plus a numeric
154
+ # budget on an exclusive route fails before inference; adaptive + budget selects the numeric dial.
155
+ # PLURNK_PROVIDERS_PROVIDER_ACME_OUTPUT_PATH=/max_completion_tokens
156
+ # PLURNK_PROVIDERS_PROVIDER_ACME_REASONING_EFFORT_PATH=/reasoning/effort
157
+ # PLURNK_PROVIDERS_PROVIDER_ACME_REASONING_BUDGET_PATH=/reasoning/max_tokens
158
+ # PLURNK_PROVIDERS_PROVIDER_ACME_REASONING_CONTROLS=exclusive
159
+ # PLURNK_PROVIDERS_PROVIDER_ACME_REASONING_EFFORTS=low,medium,high
160
+ # PLURNK_PROVIDERS_PROVIDER_ACME_REASONING_ON_BODY='{"reasoning":{"enabled":true}}'
161
+ # PLURNK_PROVIDERS_PROVIDER_ACME_REASONING_OFF_BODY='{"reasoning":{"enabled":false}}'
162
+ # PLURNK_PROVIDERS_PROVIDER_ACME_REASONING_ADAPTIVE_BODY='{}'
163
+ PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_EFFORT_PATH=/reasoning_effort
164
+ PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_OFF_BODY='{"reasoning_effort":"none"}'
165
+ PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_TOGGLE_BODY='{"reasoning_effort":true}'
166
+ PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_WORKERS_AI_REASONING_EFFORT_PATH=/reasoning_effort
119
167
  # Additional supported Cloudflare effort values, comma-separated; empty = Models.dev only.
120
168
  PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_WORKERS_AI_REASONING_EFFORTS=
121
169
  # Direct DeepSeek thinking activation and effort.
122
- PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_STYLE=thinking_effort
170
+ PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_EFFORT_PATH=/reasoning_effort
171
+ PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_ON_BODY='{"thinking":{"type":"enabled"}}'
172
+ PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_OFF_BODY='{"thinking":{"type":"disabled"}}'
173
+ # OpenRouter's native SDK accepts these same fields in per-call provider options.
174
+ PLURNK_PROVIDERS_PROVIDER_OPENROUTER_OPTIONS_NAMESPACE=openrouter
175
+ PLURNK_PROVIDERS_PROVIDER_OPENROUTER_REASONING_EFFORT_PATH=/reasoning/effort
176
+ PLURNK_PROVIDERS_PROVIDER_OPENROUTER_REASONING_BUDGET_PATH=/reasoning/max_tokens
177
+ PLURNK_PROVIDERS_PROVIDER_OPENROUTER_REASONING_CONTROLS=exclusive
178
+ PLURNK_PROVIDERS_PROVIDER_OPENROUTER_REASONING_TRANSPORT_EFFORTS=none,minimal,low,medium,high,xhigh,max
179
+ PLURNK_PROVIDERS_PROVIDER_OPENROUTER_REASONING_ON_BODY='{"reasoning":{"enabled":true}}'
180
+ PLURNK_PROVIDERS_PROVIDER_OPENROUTER_REASONING_OFF_BODY='{"reasoning":{"effort":"none"}}'
181
+ # DeepInfra's and Together's native SDKs extend openai-compatible, which overwrites a raw `reasoning_effort`
182
+ # option; the effort rides the SDK's own `reasoningEffort`, so every catalog effort, max included, reaches the wire.
183
+ PLURNK_PROVIDERS_PROVIDER_DEEPINFRA_OPTIONS_NAMESPACE=deepinfra
184
+ PLURNK_PROVIDERS_PROVIDER_DEEPINFRA_REASONING_EFFORT_PATH=/reasoningEffort
185
+ PLURNK_PROVIDERS_PROVIDER_DEEPINFRA_REASONING_OFF_BODY='{"reasoningEffort":"none"}'
186
+ PLURNK_PROVIDERS_PROVIDER_TOGETHERAI_OPTIONS_NAMESPACE=togetherai
187
+ PLURNK_PROVIDERS_PROVIDER_TOGETHERAI_REASONING_EFFORT_PATH=/reasoningEffort
188
+ PLURNK_PROVIDERS_PROVIDER_TOGETHERAI_REASONING_OFF_BODY='{"reasoningEffort":"none"}'
189
+ # Model-family provider options: a JSON array of {"models":["<glob>",…],"options":{…}} rules matched
190
+ # against the route's model id, first match wins ({§provider-model-options}). ADAPTIVE_OPTIONS replaces the
191
+ # adaptive policy's projection; SYSTEM_CACHE_OPTIONS marks the stable system prompt for cache writes.
192
+ # None ship: the shipped defaults serve open models only; declare a closed family's options yourself.
193
+ # Alibaba DashScope (OpenAI-compatible): max_tokens is answer-only, max_completion_tokens is inclusive;
194
+ # effort and thinking budget are alternatives; enable_thinking switches reasoning.
195
+ PLURNK_PROVIDERS_PROVIDER_ALIBABA_OUTPUT_PATH=/max_completion_tokens
196
+ PLURNK_PROVIDERS_PROVIDER_ALIBABA_REASONING_EFFORT_PATH=/reasoning_effort
197
+ PLURNK_PROVIDERS_PROVIDER_ALIBABA_REASONING_BUDGET_PATH=/thinking_budget
198
+ PLURNK_PROVIDERS_PROVIDER_ALIBABA_REASONING_CONTROLS=exclusive
199
+ PLURNK_PROVIDERS_PROVIDER_ALIBABA_REASONING_ON_BODY='{"enable_thinking":true}'
200
+ PLURNK_PROVIDERS_PROVIDER_ALIBABA_REASONING_OFF_BODY='{"enable_thinking":false}'
123
201
  # Uncataloged compatible providers: SDK, endpoint, and credential-variable name.
124
202
  PLURNK_PROVIDERS_PROVIDER_BAICHUAN_NPM=@ai-sdk/openai-compatible
125
203
  PLURNK_PROVIDERS_PROVIDER_BAICHUAN_BASE_URL=https://api.baichuan-ai.com/v1
@@ -132,8 +210,8 @@ PLURNK_PROVIDERS_PROVIDER_QIANFAN_API_KEY_ENV=QIANFAN_API_KEY
132
210
  # OpenAI-compatible fallback endpoint; PLURNK_BASEURL_<alias> overrides it for one declared alias.
133
211
  OPENAI_BASE_URL=https://api.openai.com/v1
134
212
 
135
- # --- OpenRouter application attribution ---
136
- # Public application URL; empty suppresses attribution, including the title.
137
- OPENROUTER_HTTP_REFERER=https://github.com/plurnk/plurnk-service
138
- # Application display name, sent only with a nonempty URL.
139
- OPENROUTER_APP_TITLE=Plurnk
213
+ # --- Application attribution (routes on the OpenRouter SDK) ---
214
+ # Public application URL the SDK sends as HTTP-Referer; empty suppresses attribution, including the name.
215
+ PLURNK_PROVIDERS_PROVIDER_OPENROUTER_APP_URL=https://github.com/plurnk/plurnk-service
216
+ # Application display name the SDK sends as X-OpenRouter-Title, only with a nonempty URL.
217
+ PLURNK_PROVIDERS_PROVIDER_OPENROUTER_APP_NAME=Plurnk
package/README.md CHANGED
@@ -47,16 +47,16 @@ discovery and environment-file defaults.
47
47
  Select a catalog route directly:
48
48
 
49
49
  ```dotenv
50
- PLURNK_MODEL=google/gemini-3-flash
51
- GEMINI_API_KEY=...
50
+ PLURNK_MODEL=deepseek/deepseek-v4-flash
51
+ DEEPSEEK_API_KEY=...
52
52
  ```
53
53
 
54
54
  Declare an alias when the route needs a reusable name or scoped tuning:
55
55
 
56
56
  ```dotenv
57
- PLURNK_MODEL_fast=openai/gpt-5-mini
57
+ PLURNK_MODEL_fast=deepinfra/zai-org/GLM-5.3-Flash
58
58
  PLURNK_MODEL=fast
59
- OPENAI_API_KEY=...
59
+ DEEPINFRA_API_KEY=...
60
60
  ```
61
61
 
62
62
  Cataloged providers need no endpoint declaration. To add an
@@ -108,8 +108,9 @@ flowchart TD
108
108
  ```
109
109
 
110
110
  - **Effort** and **budget** are independent. `adaptive` requests native dynamic
111
- reasoning where supported, otherwise the supported high posture. Fixed effort
112
- names must be supported by the route; an unsupported request is not silently downgraded.
111
+ reasoning where supported, otherwise `PLURNK_PROVIDERS_EFFORT_FALLBACK`
112
+ (default `high`). An unavailable fallback retains the reasoning-enabled provider default.
113
+ Fixed effort names must be supported by the route; unsupported requests are not silently downgraded.
113
114
  - **Output** accepts positive tokens or a percentage of context; known model
114
115
  limits cap it. An explicit reasoning budget must be smaller than total output.
115
116
  Leaving the reasoning budget unset does not disable reasoning.
@@ -166,6 +167,8 @@ PLURNK_MODEL=local
166
167
  # Optional explicit generation allowances, in tokens:
167
168
  # PLURNK_PROVIDERS_OUTPUT_BUDGET_local=8192
168
169
  # PLURNK_PROVIDERS_REASONING_BUDGET_local=4096
170
+ # Optional: serialize inference on a single-slot endpoint without serializing workers:
171
+ # PLURNK_PROVIDERS_MAX_CONCURRENCY_local=1
169
172
  ```
170
173
 
171
174
  Endpoint probing supplies served-model capacity and llama-server capabilities.
@@ -184,6 +187,7 @@ opt-in for endpoints that emit a leading `<think>` envelope, not a reasoning swi
184
187
 
185
188
  | Concern | Boundary |
186
189
  | --- | --- |
190
+ | Inference concurrency | Optional per-endpoint, process-local admission; queued calls remain cancellable. The default is unrestricted ({§provider-inference-admission}). |
187
191
  | Attempt, first-content, idle deadlines | Provider transport; a timeout surfaces a failure, not a fabricated response. |
188
192
  | Provider-directed waits | Bounded retries honor `Retry-After`; other recoverable failures return to Core. |
189
193
  | Loop recovery | Core reissues within its recovery window, then parks for a prompt or wake. |
package/SPEC.md CHANGED
@@ -261,8 +261,8 @@ PLURNK maps its generic settings to AI SDK call settings:
261
261
  - presence and frequency penalties;
262
262
  - stop sequences and seed;
263
263
  - output-token ceiling;
264
- - `off`, `adaptive`, or fixed `low`, `medium`, or `high` reasoning policy, with
265
- an independent optional operator budget.
264
+ - the supported efforts and optional numeric control under
265
+ {§provider-effort}.
266
266
 
267
267
  Provider-specific options are permitted only where they preserve a documented
268
268
  PLURNK product contract the generic SDK surface cannot express.
@@ -277,28 +277,31 @@ catalog figure. Without catalog rates the override must declare `input` and
277
277
  response cost still outranks any estimate. Unknown keys, repeats, and negative
278
278
  or non-numeric rates refuse at construction.
279
279
 
280
- §provider-reasoning-policy The portable vocabulary comes from
281
- {§reasoning-policy-wire}. `adaptive` requests the provider's
282
- documented dynamic mechanism where one exists. Otherwise, a cataloged graded
283
- route receives its strongest positive Models.dev effort that the installed
284
- transport can represent; a route that declares a toggle control receives an
285
- explicit enable on transports that document one (OpenRouter `reasoning.enabled`,
286
- Fireworks Boolean `reasoning_effort`) — an unconfigured reasoning-capable route
287
- must not run reasoning-off, and `off` stays the explicit opt-out; otherwise the
288
- provider default holds. A fixed policy retains its exact name and is rejected before
289
- provider I/O when either the route or transport cannot represent it. Every
290
- provider exposes that exact intersection. A numeric reasoning budget constrains
280
+ §provider-effort The portable vocabulary comes from
281
+ {§effort-wire}. `adaptive` uses the first applicable projection:
282
+
283
+ | Condition | Projection |
284
+ | --- | --- |
285
+ | Explicit adaptive declaration: `REASONING_ADAPTIVE_BODY`, or a native model family's `ADAPTIVE_OPTIONS` ({§provider-model-options}) | Preserve that mechanism; an explicit `{}` retains the enabled endpoint's default. |
286
+ | The configured `PLURNK_PROVIDERS_EFFORT_FALLBACK` is supported by both model and transport | Send that exact effort. The shipped fallback is `high`, not the strongest available level. |
287
+ | No supported fallback, including an empty fallback setting | Retain reasoning activation and the provider's default effort; never invent a level or escalate to another one. |
288
+
289
+ Activation is distinct from effort: declared enable fields accompany active
290
+ reasoning, and a toggle-only route uses its documented enable mechanism.
291
+ Non-reasoning models receive no reasoning controls; `off` is the explicit opt-out.
292
+ A fixed policy retains its exact name and is rejected before provider I/O when
293
+ either the route or transport cannot represent it. Every provider exposes that
294
+ exact intersection. A numeric reasoning budget constrains
291
295
  the generation envelope independently and never selects or changes policy. On
292
- the OpenRouter transport a resolved budget under `adaptive` reaches the wire as
293
- the `reasoning.max_tokens` form — the only dial a budget_tokens-only route
294
- understands; a fixed policy keeps the `effort` form, and `off` suppresses the
295
- budget.
296
-
297
- `catalogReasoningPolicies` projects that same admission calculation from catalog
298
- facts and provider-wide environment declarations without constructing a model,
299
- requiring credentials, or performing provider I/O.
296
+ routes whose controls are exclusive, fixed effort plus a numeric budget is
297
+ rejected before I/O. Under `adaptive`, an explicit budget selects the numeric
298
+ control; `off` suppresses it.
299
+
300
+ `catalogEfforts` projects that same admission calculation from catalog
301
+ facts and provider-wide environment declarations over the installed defaults,
302
+ without constructing a model, requiring credentials, or performing provider I/O.
300
303
  Admission respects the installed projection: native SDK fixed efforts exclude
301
- `max`; the OpenRouter model-settings projection also excludes `xhigh`.
304
+ `max`; declared transport vocabularies constrain per-call option projections.
302
305
 
303
306
  Models.dev's route-specific `reasoning_options` is the capability authority for
304
307
  what the daemon offers on its own; the daemon never adds vendor behaviour absent
@@ -315,7 +318,7 @@ explicit compatible adapter owns the wire projection:
315
318
  | --- | --- | --- |
316
319
  | `reasoning: false` | No reasoning request | `off` |
317
320
  | `reasoning_options: []` | Provider default | None |
318
- | `effort.values` | Native dynamic mechanism, otherwise strongest transportable positive value | Transportable fixed members of {§reasoning-policy-wire}; `off` only when `none` is transportable |
321
+ | `effort.values` | Native dynamic mechanism, otherwise the supported configured fallback or provider default | Transportable fixed members of {§effort-wire}; `off` only when `none` is transportable |
319
322
  | `toggle` | Native or explicitly declared activation, otherwise provider default | `off` only when that transport owns the toggle wire |
320
323
  | `budget_tokens` | Does not select policy | None; an adapter may use its bounds when projecting the independent budget |
321
324
  | No catalog entry | Explicit adapter declaration | Only the declaration's exact subset |
@@ -345,34 +348,36 @@ provider's documented header, body field, or native SDK option. The common
345
348
  transport neither guesses from protocol resemblance nor sends a generic cache
346
349
  field to an unknown provider. The operator may disable affinity globally or per
347
350
  alias; automatic provider caching without an affinity control remains untouched.
351
+ The environment's `CACHE_AFFINITY_FIELD` declares that placement as
352
+ `{"target":"header"|"body","name":"…"}` or
353
+ `{"target":"provider-option","provider":"…","name":"…"}`. It follows
354
+ the provider/route/alias precedence of {§provider-wire-declaration}; `null`
355
+ clears the declaration. Placement is independent of the enable switch, cannot
356
+ replace transport-owned fields, and does not imply support from a provider's name.
357
+
358
+ §provider-model-options **A model family's native options are a provider declaration.** Models.dev
359
+ does not say which models on a native SDK take adaptive reasoning or explicit cache writes, and the SDKs
360
+ do not export their tables, so a provider declares them as data: `ADAPTIVE_OPTIONS` and
361
+ `SYSTEM_CACHE_OPTIONS` are JSON arrays of `{"models":["<glob>",…],"options":{…}}` rules. The first rule
362
+ with a glob matching the route's model id (`path.matchesGlob`) supplies its per-call provider options;
363
+ without a match none apply. They follow the provider/route/alias precedence of
364
+ {§provider-wire-declaration}, an empty value declares none, and a malformed value fails construction.
365
+ None ships: Plurnk's shipped defaults serve open models only, and an operator who routes a closed model
366
+ family declares its options in their own configuration.
348
367
 
349
368
  §provider-cache-write-policy **Cache-write policy is separate from affinity.**
350
369
  `PLURNK_PROVIDERS_CACHE_WRITE_POLICY` is `off` or `stable-system`. The latter
351
370
  marks only the final leading system instruction as an explicit reusable cache
352
- boundary, and only on routes whose native SDK documents that control. It does
371
+ boundary, and only on routes whose provider declares the control in
372
+ `SYSTEM_CACHE_OPTIONS` ({§provider-model-options}). It does
353
373
  not mark the changing user packet or enable an API-wide automatic cache mode.
354
374
  Unsupported routes receive no invented option. The default five-minute
355
375
  provider lifetime is used; a longer, differently priced lifetime is not an
356
376
  implicit transport choice.
357
377
 
358
- §deepseek-reasoning-request The direct DeepSeek catalog path maps the common
359
- reasoning intent to its OpenAI-compatible controls:
360
-
361
- | PLURNK posture | `thinking` | `reasoning_effort` |
362
- | --------------- | --------------------- | -------------------- |
363
- | `off` | `{ type: disabled }` | omitted |
364
- | `adaptive` | `{ type: enabled }` | omitted |
365
- | `high` | `{ type: enabled }` | `high` |
366
-
367
- The direct API does not distinguish portable `low` or `medium` intent and
368
- therefore advertises only `off`, `adaptive`, and `high`.
369
-
370
- §provider-reasoning-style A reasoning style names the wire a compatible route's
371
- reasoning controls take. `PLURNK_PROVIDERS_PROVIDER_<PREFIX>_REASONING_STYLE`
372
- declares it for every route of a provider; `PLURNK_PROVIDERS_REASONING_STYLE`,
373
- alias-scopable like every bare knob, declares it for one route and wins. One
374
- provider can serve models whose controls differ, so a route states its own style. The daemon never infers a
375
- style from a model name.
378
+ §deepseek-reasoning-request Direct DeepSeek uses the panel's request-field
379
+ declarations under {§provider-wire-declaration}. Models.dev supplies each
380
+ model's effort vocabulary; it is not a second fixed list in this adapter.
376
381
 
377
382
  The compatible transport is deliberately retained for:
378
383
 
@@ -383,6 +388,49 @@ The compatible transport is deliberately retained for:
383
388
  It carries PLURNK-only fields and raw wire evidence without reimplementing the
384
389
  SDK's ordinary transport.
385
390
 
391
+ ### §provider-wire-declaration Request-field declarations
392
+
393
+ Models.dev owns model capabilities, effort vocabulary, and numeric budget bounds.
394
+ For endpoints whose field names it does not describe, the provider's
395
+ environment panel supplies a bounded projection, independent of provider identity:
396
+
397
+ | Declaration suffix | Meaning |
398
+ | --- | --- |
399
+ | `OPTIONS_NAMESPACE` | Native SDK's per-call `providerOptions` namespace. Without one, declared fields target the compatible request body. |
400
+ | `OUTPUT_PATH` | RFC 6901 object-member pointer for the **inclusive** output ceiling; absent uses the compatible SDK's `max_tokens` field. |
401
+ | `REASONING_EFFORT_PATH` / `REASONING_BUDGET_PATH` | Pointers for exact effort or numeric reasoning subset. No pointer means no such control. |
402
+ | `REASONING_EFFORTS` | Additional declared effort values, unioned with Models.dev. |
403
+ | `REASONING_TRANSPORT_EFFORTS` | Optional transport vocabulary, intersected with catalog/declaration efforts. It adds no model capability. |
404
+ | `REASONING_CONTROLS` | Required when both pointers exist: `exclusive` refuses fixed effort plus budget; `combined` sends both. |
405
+ | `REASONING_ON_BODY` | Static reasoning activation fields, merged with the selected control. |
406
+ | `REASONING_OFF_BODY` | Explicit disable fields; otherwise a declared `none` effort can disable. |
407
+ | `REASONING_ADAPTIVE_BODY` | Explicit adaptive fields, ahead of graded fallback; `{}` retains the enabled endpoint's default. |
408
+ | `REASONING_TOGGLE_BODY` | Reasoning enable fields when Models.dev declares a toggle and neither an explicit adaptive body nor a supported fallback effort applies. |
409
+
410
+ The existing provider declaration prefix is
411
+ `PLURNK_PROVIDERS_PROVIDER_<NAME>_`; `PLURNK_PROVIDERS_<suffix>` overrides
412
+ it for the route and accepts the ordinary alias suffix. These are data, not
413
+ executable transformations. Static bodies cannot replace transport, sampling,
414
+ or managed numeric controls; pointers cannot overlap. Configuration fails at
415
+ construction before inference when a requested policy or numeric control is
416
+ unrepresentable. A tighter per-call envelope reprojects the same declaration.
417
+ Discovery and generation share the resolved policy set. A cataloged
418
+ non-reasoning model receives no reasoning fields.
419
+ Native SDKs retain their own output field; `OUTPUT_PATH` is incompatible with an
420
+ options namespace. Request-local options are projected after the effective
421
+ envelope is known, never frozen into SDK model construction. Streaming and
422
+ non-streaming calls use the same projection.
423
+ Retired `REASONING_STYLE` selectors fail at the selected provider/alias boundary;
424
+ they neither select a preset nor silently coexist with these declarations.
425
+
426
+ A native SDK's portable reasoning setting has no `max`, so a native route without
427
+ a namespace cannot send an effort the catalog documents beyond it; construction
428
+ refuses it and names the declaration that would. Native SDKs built on
429
+ openai-compatible (DeepInfra, Together) write `reasoning_effort` from their own
430
+ `reasoningEffort` option after spreading the others, overwriting a raw
431
+ `reasoning_effort`; their shipped declarations therefore point at
432
+ `/reasoningEffort`, with `{"reasoningEffort":"none"}` as the off body.
433
+
386
434
  ## §4 Operator configuration
387
435
 
388
436
  §provider-configuration Every operational value is an environment knob
@@ -458,8 +506,8 @@ Provider and model facts resolve independently:
458
506
  | Context window | Catalog metadata or local endpoint probe. | `PLURNK_PROVIDERS_CONTEXT_WINDOW`. | Minimum when both exist; sole value otherwise. Cataloged cloud miss fails construction; compatible probe miss remains `null` with one warning. |
459
507
  | Maximum input | Catalog `limit.input`; no generic live probe. | None. | Catalog value or `null`; never reconstructed from context and output. |
460
508
  | Maximum output | Catalog `limit.output`; no generic live probe. | None. | Minimum of catalog value and effective context, or `null`. |
461
- | Total output budget | None. | `PLURNK_PROVIDERS_OUTPUT_BUDGET`. | Percentage of effective context or absolute count, capped by known context/output limits; a call may only tighten it. |
462
- | Reasoning policy | Catalog `reasoning_options` intersected with the installed adapter; explicit adapter declaration for uncataloged routes. | `PLURNK_PROVIDERS_REASONING`, initially; durable worker selection thereafter. | One supported member of `off`, `adaptive`, `low`, `medium`, or `high`; `adaptive` is the default. |
509
+ | Total output budget | None. | `PLURNK_PROVIDERS_OUTPUT_BUDGET`. | Curation reservation: percentage of effective context or absolute count, capped by known context/output limits; a call may only tighten it. The response grant may expand under {§provider-flexed-allowance}. |
510
+ | Effort | Catalog `reasoning_options` intersected with the installed adapter; explicit adapter declaration for uncataloged routes. | `PLURNK_PROVIDERS_EFFORT`, initially; durable worker selection thereafter. | A supported member of {§effort-wire}, projected under {§provider-effort}. The shipped selection is `adaptive`. |
463
511
  | Reasoning budget | None. | Optional `PLURNK_PROVIDERS_REASONING_BUDGET`. | Percentage of effective context or absolute count; valid only as a strict subset of total output and effective unless reasoning is `off`. |
464
512
  | Cost override | None. | Optional `PLURNK_PROVIDERS_COST`. | {§operator-cost-override} — comma-separated `key=value` per-1M-token USD rates over `input, output, reasoning, cacheRead, cacheWrite`; merges over the Models.dev catalog block (the catalog is the starting point), alias-scoped like every knob. Without catalog rates the override must declare `input` and `output`. The cost estimate's `source` names the override; a provider-reported response cost still outranks any estimate. |
465
513
  | Reasoning capability | Catalog `reasoning` and route-specific `reasoning_options`. | Adapter wire style only where the catalog cannot name the native field. | Catalog controls determine admissible policy; the adapter determines its wire projection. |
@@ -554,8 +602,7 @@ PLURNK `Provider`, read PLURNK tuning knobs, or reproduce transport policy.
554
602
  Discovery is scope-agnostic and memoized per process. Duplicate names fail hard.
555
603
  The common plugin trust gate applies before import ({§plugin-trust-boundary}). A plugin absent from
556
604
  Models.dev requires an explicit context-window pin because PLURNK will not guess
557
- model physics. `Discovery.packageAttributions` carries the canonical package map;
558
- the published name-keyed `Discovery.attributions` remains its 1.x projection.
605
+ model physics. `Discovery.packageAttributions` carries the canonical package map.
559
606
 
560
607
  ## §7 Local capabilities
561
608
 
@@ -599,6 +646,16 @@ reasoning enclosure only after preserving grammar evidence. Process-wide
599
646
  llama-server flags are fallback server configuration, not part of the PLURNK
600
647
  contract and need not be synchronized with an alias.
601
648
 
649
+ §repetition-stop **A response that repeats itself is stopped.** A degenerate model can stream one line
650
+ without end; content keeps arriving, so no silence deadline fires, and the call runs to its output
651
+ allowance. The transport counts the complete lines of each streamed channel, text and reasoning, and
652
+ when one line of 16 or more characters has appeared `PLURNK_PROVIDERS_REPEATED_LINE_LIMIT` times it stops
653
+ reading: the stream is cancelled, the physical request settles with whatever usage the chunks carried,
654
+ and the call fails as `repetition` (422, not retryable) carrying the partial attempt and the sentence
655
+ "The response repeated one line N times and was stopped: `<line>`". The consumer reads it as an invalid
656
+ emission, never as a provider failure to recover. Measured: a granite-4.2-8b run repeated one line 358
657
+ times across 45,875 tokens and 9 min 42 s; a limit of 32 stops it 14.6% of the way in (#876).
658
+
602
659
  §llama-reasoning-request The allowance is cumulative across the complete response. Opening a second or
603
660
  later reasoning block does not replenish it. Template parsing, the reasoning
604
661
  sampler, normalized usage, and the returned reasoning channel MUST agree on that
@@ -622,14 +679,14 @@ Generic AI SDK calls accept only settings represented by the SDK's portable
622
679
  surface. Compatible endpoints may carry additional sampling keys after reserved
623
680
  keys are removed.
624
681
 
625
- §openrouter-app-attribution **The cataloged OpenRouter route identifies the
626
- calling application through OpenRouter's current app-attribution headers.**
627
- `HTTP-Referer` is the absolute HTTP(S) application URL and
628
- `X-OpenRouter-Title` is its optional display title. The shipped floor identifies
629
- the public Plurnk repository and may be replaced by operator configuration; an
630
- explicitly empty `OPENROUTER_HTTP_REFERER` suppresses both headers. Attribution
631
- applies only to the cataloged `openrouter` route and never leaks to another
632
- provider merely because it uses the same SDK package.
682
+ §openrouter-app-attribution **A route on the OpenRouter SDK identifies the calling
683
+ application only as its provider declares.** `APP_URL`, an absolute HTTP(S) URL, and
684
+ the optional `APP_NAME` are provider declarations (`PLURNK_PROVIDERS_PROVIDER_<NAME>_`,
685
+ overridable per route or alias); the SDK sends them as `HTTP-Referer` and
686
+ `X-OpenRouter-Title`. The shipped floor declares them for `openrouter` only, so
687
+ another provider on the same SDK package sends none; an empty `APP_URL` suppresses
688
+ both. The retired `OPENROUTER_HTTP_REFERER`, `OPENROUTER_APP_TITLE` and
689
+ `OPENROUTER_X_TITLE` fail construction.
633
690
 
634
691
  ## §9 Failures, retries, and cancellation
635
692
 
@@ -642,8 +699,15 @@ into an empty model turn or reduced to a message plus a generic status.
642
699
  Upstream diagnostic text is bounded by
643
700
  `PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT`; the committed `.env.defaults` owns its
644
701
  normal value. Retry exhaustion is preserved as `attempts` and
645
- `retryExhausted`, and the resulting Problem is not marked retryable after the
646
- provider has consumed its automatic retry budget.
702
+ `retryExhausted`; it does not change the Problem's `retryable`.
703
+
704
+ §provider-retryable-truth **`retryable` states what happens next.** A Problem's
705
+ `retryable` is exactly membership of its kind in the exported
706
+ `RETRYABLE_PROVIDER_KINDS` — `rate_limit`, `network_failure`, `deadline_exceeded`,
707
+ `resource_interrupted`, `output_dropped` — and Core's provider recovery re-issues
708
+ exactly those kinds by reading the same set. It never copies the transport's own
709
+ retry policy: a 502 or 524 that the transport will not replay is still re-issued by
710
+ the consumer, so it reports `retryable: true`. Every other kind reports `false`.
647
711
 
648
712
  §provider-failure-cause **The wrapper's cause is evidence, bounded.** The SDK
649
713
  reports every processing failure of a 2xx body with one message ("Failed to
@@ -716,20 +780,29 @@ deadline:
716
780
 
717
781
  | Layer | Operator knob | Boundary | Expiry |
718
782
  | --- | --- | --- | --- |
719
- | Operation | `PLURNK_PROVIDERS_OPERATION_TIMEOUT` | Complete logical call, including every attempt and retry delay. | Final `deadline_exceeded` Problem at 504 with `timeoutPhase=operation`; never retried. Enforced as a race, not only the advisory signal, so a wedged transport that never observes the abort cannot hang the loop past the deadline (#505); a well-behaved transport unwinds within a short grace and settles its own attempt evidence first. |
720
- | Attempt | `PLURNK_PROVIDERS_FETCH_TIMEOUT` | One physical generation request. A non-streamed request is bounded through response consumption; a streamed one only until its first semantic content, after which stream-idle catches a stall and the operation deadline bounds the whole, so a stream still producing (long reasoning) is never cut off. | Surfaced `network_failure` with `timeoutPhase=attempt`; never transport-retried (#479) — the consumer's recovery owns re-issue. |
721
- | First content | `PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT` | Response-stream start through first semantic model content; metadata, empty deltas, and transport activity do not satisfy it. | Surfaced `network_failure` with `timeoutPhase=first_content`; never transport-retried (#479). |
783
+ | Operation | `PLURNK_PROVIDERS_OPERATION_TIMEOUT` | Complete logical call, including admission queueing, every attempt and retry delay. | Final `deadline_exceeded` Problem at 504 with `timeoutPhase=operation`; not retried inside this operation. Consumer recovery is separate. Enforced as a race, not only the advisory signal, so a wedged transport that never observes the abort cannot hang the loop past the deadline (#505); a well-behaved transport unwinds within a short grace and settles its own attempt evidence first. |
784
+ | Attempt | `PLURNK_PROVIDERS_FETCH_TIMEOUT` | One physical generation request. A non-streamed request is bounded through response consumption; a streamed one only until its first semantic content. After content begins, stream-idle bounds silence and the operation deadline still bounds the whole call. | Surfaced `network_failure` with `timeoutPhase=attempt`; never transport-retried (#479) — the consumer's recovery owns re-issue. |
785
+ | First content | `PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT` | Dispatch through first semantic model content ({§provider-first-content-at-dispatch}); metadata, empty deltas, and transport activity do not satisfy it. | Surfaced `network_failure` with `timeoutPhase=first_content`; never transport-retried (#479). |
722
786
  | Stream idle | `PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT` | Silence between semantic content chunks after content begins. | Surfaced `network_failure` with `timeoutPhase=stream_idle`; never transport-retried (#479). |
787
+ | Repeated line | `PLURNK_PROVIDERS_REPEATED_LINE_LIMIT` | A streamed line of 16 or more characters repeated this many times; `0` disables. | Stopped as `repetition` ({§repetition-stop}); never transport-retried. |
788
+
789
+ §provider-first-content-at-dispatch The first-content deadline is armed when the
790
+ admitted attempt is dispatched, before response headers, so an endpoint that accepts
791
+ the request and never answers fails at `timeoutPhase=first_content` instead of holding
792
+ the attempt for the whole attempt deadline. It still begins only after admission
793
+ ({§provider-inference-admission}).
794
+
795
+ Settled calls remove their deadline timers and cancellation subscriptions.
723
796
 
724
797
  Caller cancellation spans the operation and preserves the caller's reason.
725
798
  A 2xx exchange whose body cannot be processed (a provider invalid-response)
726
799
  classifies as the non-retryable 502 on the first failure unless an explicit
727
- `x-should-retry` directive says otherwise (#479 supersedes #446's budgeted
728
- promotion). Inner deadline failures surface on the first failure; when a
800
+ `x-should-retry` directive says otherwise (#479). Inner deadline failures surface on the first failure; when a
729
801
  directive-driven retry sequence exhausts, `attempts` and `retryExhausted`
730
- are added and the classification is final. Every scheduler iteration opens and settles exactly
731
- one ordered {§provider-request-accounting} record, including response-less
732
- network failures and timed-out attempts.
802
+ are added and the classification is final. Every admitted physical attempt opens
803
+ and settles exactly one ordered {§provider-request-accounting} record, including
804
+ response-less network failures and timed-out attempts. A cancelled or expired
805
+ admission wait opens no physical request.
733
806
 
734
807
  Each streamed physical request assembles its own response. Failed partial answer
735
808
  bytes never enter a later request's completed `ProviderResponse`; recovery is a
@@ -748,6 +821,21 @@ ordinary 5xx surface on the first failure as consumer-recoverable kinds; one
748
821
  retry authority — the consumer's own provider-recovery machinery — owns
749
822
  re-issue, backoff, and park above the transport.
750
823
 
824
+ §provider-output-dropped **A completed exchange whose own evidence proves its
825
+ text never arrived is `output_dropped`**, a 502 carrying the complete response as
826
+ `error.attempt` and `stage: "provider-response"`; it is never admitted as an empty or
827
+ truncated turn, and the consumer re-issues it ({§provider-retryable-truth}):
828
+
829
+ | Evidence | Detail | Facts |
830
+ | --- | --- | --- |
831
+ | `finish_reason` `tool_calls` (Plurnk never declares tools) | `The provider ended the response with tool calls although no tools were declared (N native tool call(s)); the response text was not delivered as text.` | `toolCallCount` |
832
+ | Billed output tokens exceed the code points streamed across every channel — text and reasoning, before projection — by at least `PLURNK_PROVIDERS_DROPPED_OUTPUT_TOKENS` (`0` disables; an output token decodes to at least one character) | `The provider billed T output tokens but streamed C characters of text and reasoning; at least T−C tokens of output never arrived.` | `billedOutputTokens`, `streamedCharacters`, `droppedOutputTokens` |
833
+
834
+ The token rule needs a billed output count, and it does not apply when the response bills
835
+ reasoning tokens but streamed no reasoning in any channel: that is hidden or summarized
836
+ reasoning, which the characters cannot account for. Text-only comparisons are not made —
837
+ a route may bill reasoning inside its text count.
838
+
751
839
  §provider-request-rejection An HTTP 4xx rejection not classified as authorization,
752
840
  quota, capacity, grammar, rate limit, or transient transport failure is
753
841
  `request_rejected`: preserve the upstream status and detail, with `retryable: false`.
@@ -766,7 +854,7 @@ completed exchange.
766
854
  | Normalized finish reason | Exact `insufficient_system_resource` becomes `resource_interrupted`; the raw value remains in `assistantRaw`. |
767
855
  | `generate` outcome | Throw `ProviderError(kind="resource_interrupted")` at local status 503 with the normalized attempt on `error.attempt` and the same accounting on `error.accounting`. |
768
856
  | Partial response | Preserve content, reasoning, usage, model, metadata, optional raw body, and other evidence without admitting it as success. |
769
- | Automatic replay | None; the Problem has `retryable: false`, and AI SDK retry scheduling has already completed at the successful transport. |
857
+ | Automatic replay | None inside the provider; AI SDK retry scheduling has already completed. The Problem has `retryable: true`: the consumer re-issues it ({§provider-retryable-truth}). |
770
858
  | Capacity-pool overflow | None under the existing routing policy; when other overflow-eligible failures do reach a sibling, the pool concatenates their request accounting. |
771
859
  | Consumer admission | Persist the evidence as an unaccepted attempt; never parse it into executable work, even when its frame looks complete. |
772
860
 
@@ -852,8 +940,9 @@ budget is a strict subset of that total, never an additive reserve. The
852
940
  configured total is a percentage of effective context or an absolute count;
853
941
  percentages resolve to the nearest whole token with a one-token minimum. It is
854
942
  capped by known context and model-output limits; `generate.maxOutputTokens` may
855
- only tighten it for one call. The effective reasoning subset tightens with that
856
- total and remains strictly smaller.
943
+ only tighten this reservation for one call. The effective reasoning subset tightens
944
+ with that total and remains strictly smaller. Exact prompt measurements may expand
945
+ the response grant beyond the reservation under {§provider-flexed-allowance}.
857
946
 
858
947
  The adapter owns native projection. A backend whose generic SDK maximum already
859
948
  includes reasoning receives the total directly. When a native SDK instead adds
@@ -867,7 +956,7 @@ and provider minimum; an envelope too small to represent the minimum fails
867
956
  before provider I/O.
868
957
 
869
958
  §provider-output-budget-conformance When a completed response reports
870
- normalized output-token usage greater than its effective total output budget,
959
+ normalized output-token usage greater than its response grant ({§provider-flexed-allowance}),
871
960
  the exchange is an `invalid_response` at 502 rather than an admitted result or
872
961
  a prompt-capacity 413. Its complete failed-attempt evidence and settled charged
873
962
  request remain available. The violation is final and is never automatically
@@ -880,7 +969,29 @@ limit advertises `requiresOutputBudget` and fails construction when no total can
880
969
  be resolved. The retired additive reserve knobs fail hard rather than creating
881
970
  a second envelope contract.
882
971
 
883
- ## §13 Capacity pool
972
+ ## §13 Inference capacity
973
+
974
+ ### Admission
975
+
976
+ §provider-inference-admission `PLURNK_PROVIDERS_MAX_CONCURRENCY` controls physical
977
+ generation attempts per resolved endpoint within one process: `-1` is unrestricted;
978
+ a positive safe integer admits that many concurrent attempts. It follows ordinary
979
+ alias scoping. Zero, other negative values and non-integers are invalid.
980
+
981
+ | Boundary | Contract |
982
+ | --- | --- |
983
+ | Identity | Provider instances and aliases sharing the resolved API base URL share one allowance. SDK/plugin-owned endpoints without a resolved URL share their provider identity. Conflicting limits for one identity fail construction, naming the identity and both values; they never create independent queues. The daemon reads its environment once at boot, so an identity's limit cannot change within a process: reconstructing a provider from that environment reuses its allowance, and in-flight leases keep counting. |
984
+ | Admission | FIFO among live waiters. The lease begins before the physical request observer and ends after the complete response or transport failure settles, including streamed bodies. |
985
+ | Cancellation | A queued abort removes that waiter and preserves the caller's reason. It opens no physical request or accounting row. In-flight cancellation signals the transport; capacity is released when that attempt unwinds. A transport still running despite abort does not authorize exceeding the limit. |
986
+ | Retries | Backoff holds no lease. Each retry rejoins admission as a new physical attempt. |
987
+ | Deadlines | The existing operation deadline includes queueing. Attempt, first-content and stream-idle deadlines begin only after admission; queued work is not a stalled stream. |
988
+ | Scope | Workers, tools, messages, waits, token measurement and endpoint discovery are not serialized by inference admission. Independent endpoints progress independently. No cross-process or machine-wide capacity guarantee is implied. |
989
+
990
+ Admission neither changes worker lifecycle nor selects another model or endpoint.
991
+ Core consumes the same asynchronous Provider contract. A waiting worker holds no
992
+ inference lease; BARE and ordinary generation use the same physical boundary.
993
+
994
+ ### Pool
884
995
 
885
996
  §provider-capacity-pool `Pool` fronts interchangeable `Provider` instances. It
886
997
  keeps workers sticky for
@@ -908,6 +1019,8 @@ Coverage MUST prove:
908
1019
  - native SDK request mapping and normalized responses;
909
1020
  - compatible extension preservation;
910
1021
  - timeout, retry, cancellation, interrupted-attempt, and final-error behavior;
1022
+ - shared endpoint inference admission, FIFO queueing, queued/in-flight cancellation,
1023
+ full-stream lease lifetime, retry release, and independent endpoint progress;
911
1024
  - local capability probes and pins;
912
1025
  - exact, bounded, estimated, and unavailable complete-request measurements;
913
1026
  - independent input/context/output limits, asymmetric admission, and normalized