@plurnk/plurnk-providers 1.3.11 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/.env.defaults +40 -23
  2. package/README.md +65 -4
  3. package/SPEC.md +222 -56
  4. package/dist/AiSdkProvider.d.ts +18 -5
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +240 -153
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +14 -15
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +26 -10
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -2
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +41 -8
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/ProviderRegistry.d.ts +4 -1
  17. package/dist/ProviderRegistry.d.ts.map +1 -1
  18. package/dist/ProviderRegistry.js +7 -3
  19. package/dist/ProviderRegistry.js.map +1 -1
  20. package/dist/aiSdkTransport.d.ts +4 -2
  21. package/dist/aiSdkTransport.d.ts.map +1 -1
  22. package/dist/aiSdkTransport.js +18 -3
  23. package/dist/aiSdkTransport.js.map +1 -1
  24. package/dist/catalogProvider.d.ts +3 -1
  25. package/dist/catalogProvider.d.ts.map +1 -1
  26. package/dist/catalogProvider.js +20 -7
  27. package/dist/catalogProvider.js.map +1 -1
  28. package/dist/compatibleProvider.d.ts.map +1 -1
  29. package/dist/compatibleProvider.js +11 -4
  30. package/dist/compatibleProvider.js.map +1 -1
  31. package/dist/cost.d.ts +11 -0
  32. package/dist/cost.d.ts.map +1 -0
  33. package/dist/cost.js +64 -0
  34. package/dist/cost.js.map +1 -0
  35. package/dist/discover.d.ts +2 -0
  36. package/dist/discover.d.ts.map +1 -1
  37. package/dist/discover.js +15 -9
  38. package/dist/discover.js.map +1 -1
  39. package/dist/env.d.ts +4 -0
  40. package/dist/env.d.ts.map +1 -1
  41. package/dist/env.js +29 -9
  42. package/dist/env.js.map +1 -1
  43. package/dist/errors.d.ts +27 -0
  44. package/dist/errors.d.ts.map +1 -0
  45. package/dist/errors.js +150 -0
  46. package/dist/errors.js.map +1 -0
  47. package/dist/index.d.ts +11 -5
  48. package/dist/index.d.ts.map +1 -1
  49. package/dist/index.js +7 -4
  50. package/dist/index.js.map +1 -1
  51. package/dist/notices.d.ts +10 -0
  52. package/dist/notices.d.ts.map +1 -0
  53. package/dist/notices.js +11 -0
  54. package/dist/notices.js.map +1 -0
  55. package/dist/ollama.d.ts.map +1 -1
  56. package/dist/ollama.js +3 -3
  57. package/dist/ollama.js.map +1 -1
  58. package/dist/openai.d.ts +1 -1
  59. package/dist/openai.d.ts.map +1 -1
  60. package/dist/promptTokens.d.ts +4 -0
  61. package/dist/promptTokens.d.ts.map +1 -0
  62. package/dist/promptTokens.js +32 -0
  63. package/dist/promptTokens.js.map +1 -0
  64. package/dist/sdkModels.d.ts.map +1 -1
  65. package/dist/sdkModels.js +4 -3
  66. package/dist/sdkModels.js.map +1 -1
  67. package/dist/types.d.ts +43 -16
  68. package/dist/types.d.ts.map +1 -1
  69. package/dist/types.js +1 -1
  70. package/dist/types.js.map +1 -1
  71. package/dist/usage.d.ts +3 -0
  72. package/dist/usage.d.ts.map +1 -1
  73. package/dist/usage.js +26 -14
  74. package/dist/usage.js.map +1 -1
  75. package/dist/warnings.js +0 -0
  76. package/dist/warnings.js.map +1 -1
  77. package/package.json +13 -9
  78. package/src/AiSdkProvider.test.ts +480 -159
  79. package/src/AiSdkProvider.ts +320 -196
  80. package/src/Mock.test.ts +29 -14
  81. package/src/Mock.ts +33 -15
  82. package/src/Pool.test.ts +43 -6
  83. package/src/Pool.ts +56 -10
  84. package/src/ProviderRegistry.test.ts +158 -9
  85. package/src/ProviderRegistry.ts +19 -6
  86. package/src/aiSdkTransport.ts +25 -6
  87. package/src/boundaries.test.ts +8 -3
  88. package/src/catalogProvider.test.ts +17 -0
  89. package/src/catalogProvider.ts +25 -10
  90. package/src/compatibleProvider.test.ts +96 -0
  91. package/src/compatibleProvider.ts +15 -6
  92. package/src/cost.test.ts +63 -0
  93. package/src/cost.ts +83 -0
  94. package/src/defaults.test.ts +1 -0
  95. package/src/discover.test.ts +48 -7
  96. package/src/discover.ts +31 -21
  97. package/src/env.test.ts +38 -23
  98. package/src/env.ts +45 -18
  99. package/src/errors.test.ts +148 -0
  100. package/src/errors.ts +207 -0
  101. package/src/index.ts +29 -7
  102. package/src/lexicon-guard.test.ts +6 -6
  103. package/src/notices.ts +22 -0
  104. package/src/ollama.test.ts +64 -0
  105. package/src/ollama.ts +6 -3
  106. package/src/openai.ts +3 -0
  107. package/src/promptTokens.ts +41 -0
  108. package/src/sdkModels.test.ts +7 -0
  109. package/src/sdkModels.ts +4 -8
  110. package/src/types.ts +106 -64
  111. package/src/usage.test.ts +15 -4
  112. package/src/usage.ts +32 -14
  113. package/src/warnings.test.ts +10 -10
  114. package/src/warnings.ts +0 -0
  115. package/dist/OpenAICompat.d.ts +0 -76
  116. package/dist/OpenAICompat.d.ts.map +0 -1
  117. package/dist/OpenAICompat.js +0 -555
  118. package/dist/OpenAICompat.js.map +0 -1
  119. package/dist/openaiStream.d.ts +0 -47
  120. package/dist/openaiStream.d.ts.map +0 -1
  121. package/dist/openaiStream.js +0 -280
  122. package/dist/openaiStream.js.map +0 -1
  123. package/dist/standardProviders.d.ts +0 -31
  124. package/dist/standardProviders.d.ts.map +0 -1
  125. package/dist/standardProviders.js +0 -518
  126. package/dist/standardProviders.js.map +0 -1
  127. package/dist/telemetry.d.ts +0 -24
  128. package/dist/telemetry.d.ts.map +0 -1
  129. package/dist/telemetry.js +0 -85
  130. package/dist/telemetry.js.map +0 -1
  131. package/src/telemetry.test.ts +0 -69
  132. package/src/telemetry.ts +0 -116
package/.env.defaults CHANGED
@@ -1,6 +1,6 @@
1
1
  # REFERENCE - @plurnk/plurnk-providers' shipped .env.defaults: the operative floor for
2
- # PLURNK_PROVIDERS_* knobs and provider declarations (providers#44: every package owns
3
- # what it reads; the file IS the configuration reference).
2
+ # PLURNK_PROVIDERS_* knobs and provider declarations ({§operator-config-env-defaults});
3
+ # every package owns what it reads, and this file IS the configuration reference.
4
4
  # The daemon assembles every installed member's file into one floor (set-if-unset under the
5
5
  # operator's env) - do NOT edit this file; put YOUR config in ~/.plurnk/.env or ./.env. A key
6
6
  # claimed by two packages crashes boot naming both.
@@ -10,8 +10,8 @@
10
10
  # whose UNSET state is meaningful (derive/auto/off, or a required secret) - uncomment to pin.
11
11
  #
12
12
  # Models.dev supplies cataloged providers' SDK package, endpoint, credential names,
13
- # model metadata, and prices. Operator declarations below cover only facts absent
14
- # from or deliberately overridden over that catalog. Secret VALUES never belong here.
13
+ # and model facts. {§model-fact-resolution} defines precedence per fact; there is
14
+ # no live price fetch. Secret VALUES never belong here.
15
15
 
16
16
  # --- Side-channel reasoning (SPEC §4, #32/#33/#399) ---
17
17
  # ACTIVATION and BUDGET are separate so a numeric can never silently flip wire flags.
@@ -19,14 +19,22 @@
19
19
  # (reasoning_effort, enable_thinking, think, ...). Default ADAPTIVE (owner ruling,
20
20
  # #399): reasoning ACTIVE on a fresh install, each backend's own adaptive depth,
21
21
  # no shipped magnitude - a reasoning model that ships un-reasoning blind-edits and
22
- # declares done (svc#396 A/B: 25 EXECs -> 0, build broken, confident SEND[200]).
22
+ # declares done.
23
23
  PLURNK_PROVIDERS_REASONING=adaptive
24
- # Positive int, REQUIRED iff REASONING=on - the magnitude for tier/budget mapping. On
25
- # llama.cpp the ENFORCEMENT is the box's --reasoning-budget LAUNCH flag (per-request
26
- # numerics are ignored): env budget and launch flag are the same number, changed together.
24
+ # Positive int, REQUIRED iff REASONING=on - the magnitude for tier/budget mapping.
25
+ # On llama-server it is a request-scoped reasoning allowance and may tighten, but
26
+ # cannot exceed, the resolved PLURNK_PROVIDERS_REASONING_RESERVE.
27
27
  # PLURNK_PROVIDERS_REASONING_BUDGET=4096
28
28
 
29
- # --- Decode tuning (measured canonical values; bench/TUNING-EPIC.md F1/F6; #9/#30) ---
29
+ # Response-content interpretation ({§provider-tagged-reasoning}) is independent
30
+ # from request-side reasoning activation. The portable floor trusts only
31
+ # structured provider/SDK reasoning fields and leaves visible content verbatim.
32
+ # Set think-tags only on an alias whose exact model/endpoint contract emits one
33
+ # leading <think> envelope; it is never inferred from generic reasoning capability.
34
+ PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE=verbatim
35
+ # PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE_cfds1=think-tags
36
+
37
+ # --- Decode tuning (measured canonical values; #9/#30) ---
30
38
  # TEMPERATURE is the default for EVERY request (under caller sampling); REPEAT_PENALTY is
31
39
  # the floor the provider manages wherever a grammar rides (greedy-under-mask loops without it).
32
40
  PLURNK_PROVIDERS_TEMPERATURE=0.2
@@ -44,19 +52,20 @@ PLURNK_PROVIDERS_FREQUENCY_PENALTY=0
44
52
  # stable worker id for replica-local prefix affinity. Official native SDKs own
45
53
  # their cache mechanisms and do not receive this compatible extension.
46
54
  PLURNK_PROVIDERS_PROMPT_CACHE_KEY=1
47
- # #567: DRY loop-breaker - llama.cpp-only (grammarStyle "none" paths skip it). MULTIPLIER=0
48
- # disables DRY at the general floor: the #567 sweep showed L=2 is WORSE than DRY-off on both
49
- # runaways AND corruption; L=32 is the measured plurnk-safe threshold (0 corruption fails,
50
- # ~6% runaways vs 19% off). Enable per-alias: DRY_MULTIPLIER_<alias>=0.8, ALLOWED_LENGTH_<alias>=32.
55
+ # #567: DRY is a llama.cpp-only repeated-sequence penalty. It can reduce
56
+ # degenerate loops, but it can also corrupt exact source, identifiers, quoted
57
+ # evidence, and other repetition required by PLURNK operations. The portable
58
+ # server-wide default is off. A nonzero alias override is an explicit fidelity
59
+ # tradeoff; DRY_BASE_<alias> and DRY_ALLOWED_LENGTH_<alias> tune that opt-in.
51
60
  PLURNK_PROVIDERS_DRY_MULTIPLIER=0
52
- PLURNK_PROVIDERS_DRY_BASE=1.75
53
- PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH=32
54
- # REPEAT_LAST_N widens the older repeat_penalty window past the box's 64; optional (DRY's
55
- # whole-context window makes it secondary). Set per-alias if a model wants it.
61
+ # PLURNK_PROVIDERS_DRY_BASE_<alias>=
62
+ # PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH_<alias>=
63
+ # REPEAT_LAST_N widens the older repeat_penalty window past the box's default.
64
+ # Set it per alias only from measured model behavior.
56
65
  # PLURNK_PROVIDERS_REPEAT_LAST_N_<alias>=512
57
66
 
58
67
  # --- Transport budgets (§4, #18) ---
59
- # Per-attempt fetch timeout (ms); the caller's abort signal spans retries.
68
+ # Total generation-operation timeout (ms), including retries; caller cancellation also spans the operation.
60
69
  PLURNK_PROVIDERS_FETCH_TIMEOUT=600000
61
70
  # Maximum silence (ms) between streamed response-body chunks after response
62
71
  # streaming begins. Disabled at the portable floor: slow local inference may
@@ -66,6 +75,9 @@ PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT=0
66
75
  # Transient-failure retries: 0 = surface the first failure; N = retries on 429/5xx/timeout
67
76
  # with exponential backoff (Retry-After wins). RETRY_DELAY is the backoff base (ms).
68
77
  PLURNK_PROVIDERS_RETRY_ATTEMPTS=3
78
+ # Maximum characters retained from an upstream provider diagnostic in the
79
+ # public Problem detail. Structured failure facts are not truncated.
80
+ PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT=512
69
81
 
70
82
  # --- Endpoint probe (#34) ---
71
83
  # GET /v1/models at construction (context window + llama-server fingerprint). One failure
@@ -82,13 +94,14 @@ PLURNK_PROVIDERS_PROBE_DELAY=250
82
94
  # unconstrained output's divergence. Development aid; leave unset in production.
83
95
  # PLURNK_PROVIDERS_GBNF_DEBUG=0
84
96
 
85
- # --- Window (SPEC §11) ---
86
- # Unset = derive: env -> live endpoint probe (n_ctx) -> models.dev catalog -> null (surfaced
87
- # once via PLURNK_CONTEXT_UNKNOWN naming the model). Set to pin the window deliberately -
88
- # per-alias (_<alias>) for a box whose model the catalog doesn't know.
97
+ # --- Window ({§model-fact-resolution}) ---
98
+ # Physical provider context only. Unset derives from a live endpoint probe (n_ctx) or
99
+ # models.dev, else null (surfaced once via PLURNK_CONTEXT_UNKNOWN). A configured value
100
+ # caps detected physics or declares it when unknown. Prompt packing and model-facing
101
+ # pressure are consumer policy, not provider configuration.
89
102
  # PLURNK_PROVIDERS_CONTEXT_WINDOW=200000
90
103
 
91
- # --- Model pricing override (USD per million tokens; SPEC §4) ---
104
+ # --- Model pricing override (USD per million tokens; {§model-fact-resolution}) ---
92
105
  # Models.dev owns catalog rates. Set these together for a model absent from the
93
106
  # snapshot or to deliberately override its rates. Cached input defaults to the
94
107
  # input rate when omitted. All three knobs are per-alias scopable.
@@ -122,6 +135,8 @@ PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_API_KEY_ENV=CLOUDFLARE_API_TOKEN,CLOUDFLARE
122
135
  # Fireworks reasoners default on when reasoning_effort is omitted; explicit
123
136
  # "none" is its off switch. This is a declared wire exception, not a registry.
124
137
  PLURNK_PROVIDERS_PROVIDER_FIREWORKS_REASONING_STYLE=effort_explicit
138
+ # Direct DeepSeek reasoning follows {§deepseek-reasoning-request}.
139
+ PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_STYLE=thinking_effort
125
140
  PLURNK_PROVIDERS_PROVIDER_VOLCENGINE_NPM=@ai-sdk/openai-compatible
126
141
  PLURNK_PROVIDERS_PROVIDER_VOLCENGINE_BASE_URL=https://ark.ap-southeast.bytepluses.com/api/v3
127
142
  PLURNK_PROVIDERS_PROVIDER_VOLCENGINE_API_KEY_ENV=ARK_API_KEY
@@ -162,5 +177,7 @@ PLURNK_BASE_URL=https://plurnk.ai/v1
162
177
  # tuning. Each accepts a percentage of the window ("10%") or an absolute token count
163
178
  # ("4096"; absolutes win outright, alias-scopable for measured envelopes). The prompt
164
179
  # budget is window - reasoning - completion - the consumer's own safety margin.
180
+ # On llama-server, the resolved reasoning reserve is also the adaptive per-response
181
+ # reasoning ceiling. It is one cumulative allowance across every reasoning block.
165
182
  PLURNK_PROVIDERS_REASONING_RESERVE=10%
166
183
  PLURNK_PROVIDERS_COMPLETION_RESERVE=25%
package/README.md CHANGED
@@ -5,13 +5,21 @@ PLURNK's stable model-provider contract and its adapter to the
5
5
 
6
6
  Ordinary provider behavior is intentionally not reimplemented here:
7
7
 
8
- - Models.dev supplies provider package, endpoint, credential, context-window,
9
- output-limit, and pricing metadata at build time.
8
+ - A release-time Models.dev snapshot supplies provider package, endpoint, and
9
+ credential names, plus context-window, output-limit, reasoning-capability,
10
+ and pricing metadata.
10
11
  - Official AI SDK providers own vendor request and response protocols.
11
12
  - PLURNK owns aliases, generation envelopes, normalized usage and errors,
12
13
  evidence capture, first-party metadata, and local endpoint capabilities.
13
14
 
14
15
  See `SPEC.md` for the contract and `.env.defaults` for every operational knob.
16
+ The package-owned `PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT` bounds upstream
17
+ diagnostic text in public provider Problems.
18
+
19
+ Model facts do not share one fallback chain. Context windows, output envelopes,
20
+ reasoning activation, and estimated prices resolve independently
21
+ ({§model-fact-resolution}). PLURNK does not fetch live per-token prices, and the
22
+ local estimate is not an authoritative relay-settled charge.
15
23
 
16
24
  ## Configure a model
17
25
 
@@ -60,10 +68,15 @@ It may use any npm scope. Its package manifest declares the PLURNK name:
60
68
  The default export is an AI SDK provider with
61
69
  `languageModel(modelId)`. PLURNK adapts that language model into its own
62
70
  contract, so plugins do not reproduce retries, usage normalization, envelopes,
63
- or telemetry.
71
+ notices, or RFC 9457 failure normalization.
72
+
73
+ The manifest may declare always-on `plurnk.attribution`. The default export may
74
+ also implement synchronous `attributions(context)` and decide per provider
75
+ attempt whether to return no, one, or many additional opaque tags
76
+ ({§plugin-attribution}).
64
77
 
65
78
  Discovery is scope-agnostic and rejects duplicate names.
66
- `PLURNK_PLUGINS_TRUSTED_ONLY` restricts third-party discovery.
79
+ Third-party discovery uses the shared pre-import trust contract ({§plugin-trust-boundary}).
67
80
 
68
81
  ## Local endpoints
69
82
 
@@ -75,6 +88,54 @@ The `plurnk` provider retains the compatible transport because it carries
75
88
  first-party attribution and loop metadata and leaves model tuning to the
76
89
  endpoint.
77
90
 
91
+ ## Configured-provider packet conformance matrix
92
+
93
+ Every configured model alias is exercised through a real PLURNK loop — the
94
+ production packet, a model-selected operation, its materialized result, and
95
+ completion — never a transport-only completion. Provider-exposed reasoning must
96
+ survive in the durable assistant packet and digest; a provider with no private
97
+ reasoning is valid when the observable operation cycle succeeds.
98
+
99
+ One specimen at a time, deterministically:
100
+
101
+ ```sh
102
+ cd plurnk-core
103
+ npm run test:live:specimen -- "<test-name-pattern>"
104
+ ```
105
+
106
+ The selector inserts `--test-name-pattern` before the expanded live file list in
107
+ the exact standard `test:live` invocation; trailing npm arguments alone cannot
108
+ narrow the suite. This procedure is `plurnk-core`'s own; the ledger below is
109
+ maintained with the evidence for every alias it names.
110
+
111
+ ### Classifications
112
+
113
+ | Class | Meaning |
114
+ |---|---|
115
+ | pass | Full packet cycle completed; durable packet and digest verified |
116
+ | auth/credential | The route is blocked before the model by authorization or credential handling |
117
+ | transport | The route fails at a transport/capability boundary, not the model |
118
+ | op:stable-fail | An operation-level failure repeated on replay; assertion unweakened |
119
+ | op:stochastic | An operation-level failure did not repeat on a later roll |
120
+ | unreachable | The endpoint cannot be reached from this machine |
121
+
122
+ Authorization and credential failures are reported as their own class, never as
123
+ model failures. Repeated stochastic and stable operation failures are reported
124
+ separately in the ledger's specimens.
125
+
126
+ ### Ledger
127
+
128
+ | Alias | Route (snapshot) | Class | Evidence |
129
+ |---|---|---|---|
130
+ | 38 configured aliases — 2026-07 sweep | — | 20 × pass; 14 × auth/credential/transport; 4 × operation-level | Initial READ line-slice sweep (`#7`) |
131
+ | `cfgpt120b`, `grok` | — | op:stochastic → pass on replay | `#7` |
132
+ | `cfkimi27`, `kimi` | — | op:stable-fail (unresolved, unweakened assertion) | `#7` |
133
+ | `cfds1` | `cloudflare/@cf/deepseek-ai/deepseek-r1-distill-qwen-32b` | op:stable-fail — READ repeated 13 turns to the 508 strike threshold; retrieval materialization verified (`2:beta` present, exact 409 recovery given) | `/home/hyzen/benchmarks/live-contract-read-L-TIgart/digest/` (`#7`) |
134
+
135
+ The full current classifications of every configured alias are refreshed by the
136
+ frozen-candidate live drill's honest reporting; this ledger is the maintained
137
+ record of that procedure, never a substitute for it.
138
+
78
139
  ## Development
79
140
 
80
141
  ```sh
package/SPEC.md CHANGED
@@ -8,8 +8,9 @@ ordinary provider protocols.
8
8
 
9
9
  The provider stack has four owners:
10
10
 
11
- 1. Models.dev supplies a build-time snapshot of provider package, API endpoint,
12
- credential names, models, context windows, output limits, and USD prices.
11
+ 1. Models.dev supplies a release-time snapshot of provider package, API
12
+ endpoint, credential names, models, context windows, output limits,
13
+ reasoning capability, and USD prices.
13
14
  2. Official AI SDK providers own vendor request and response protocols.
14
15
  3. This package owns the PLURNK contract: aliases, envelopes, normalized usage
15
16
  and errors, evidence, local capabilities, and first-party metadata.
@@ -23,7 +24,8 @@ provider declaration, or left explicitly unknown.
23
24
 
24
25
  ## §2 Provider interface
25
26
 
26
- `Provider` exposes immutable model facts and one generation operation:
27
+ §provider-interface `Provider` exposes immutable model facts and one generation
28
+ operation:
27
29
 
28
30
  ```ts
29
31
  interface Provider {
@@ -35,26 +37,49 @@ interface Provider {
35
37
  readonly reasoningReserve?: number | null;
36
38
  readonly completionReserve?: number | null;
37
39
 
38
- countTokens(text: string): number;
40
+ countPromptTokens(
41
+ messages: readonly ChatMessage[],
42
+ signal?: AbortSignal,
43
+ ): Promise<PromptTokenMeasurement>;
39
44
  tokenize?(text: string): Promise<number[]>;
40
45
  calculateCost(usage: ProviderUsage): number;
46
+ calculateCharge?(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }>;
41
47
  generate(args: GenerateArgs): Promise<ProviderResponse>;
42
48
  }
43
49
  ```
44
50
 
45
- `contextWindow: null` means genuinely unknown. A consumer MUST NOT invent a
46
- stand-in. A cataloged cloud model without a context window fails construction
47
- unless the operator pins `PLURNK_PROVIDERS_CONTEXT_WINDOW`. A local probe
48
- failure degrades to `null` and emits one warning because a transient probe
49
- failure must not make a usable local endpoint unbootable.
50
-
51
- `countTokens` is synchronous and non-negative. The common fallback is a
52
- conservative chars/2 ruler and is announced once. `tokenize` exists only when
53
- the endpoint exposes its real vocabulary.
54
-
55
- `calculateCost` returns estimated USD. Models.dev rates are converted at the
56
- provider boundary. Unknown pricing returns `0`; it is not represented as a
57
- fabricated rate.
51
+ `contextWindow` is provider physics and resolves under
52
+ {§model-fact-resolution}. `null` means genuinely unknown; a consumer MUST NOT
53
+ invent a stand-in. The context-window knob never carries model-facing prompt
54
+ policy or grinder pressure.
55
+
56
+ `PromptTokenMeasurement` is a discriminated request-level result:
57
+
58
+ | `kind` | Meaning | May authorize physical admission |
59
+ | ------------- | ------------------------------------------------------ | -------------------------------- |
60
+ | `exact` | Exact count for the complete provider request. | Yes. |
61
+ | `upper_bound` | Proven upper bound for the complete provider request. | Yes. |
62
+ | `estimate` | Empirical prediction with required causal `detail`. | No. |
63
+
64
+ Every result carries a non-negative integer `tokens` and a non-empty `source`.
65
+ `countPromptTokens` receives the same messages supplied to `generate` and may
66
+ perform cancellable provider I/O. The common fallback is chars/2 over message
67
+ content; it is announced once and reported honestly as an estimate because it
68
+ knows neither the serving vocabulary nor provider-owned request framing.
69
+ An unavailable optional counting endpoint likewise returns an estimate naming
70
+ the cause. A malformed measurement is a provider contract violation and fails
71
+ hard; consumers do not reinterpret it as ordinary unavailability.
72
+
73
+ `tokenize` is the separate content-token capability and exists only when the
74
+ endpoint exposes its real vocabulary. Content tokenization does not substitute
75
+ for complete-request measurement.
76
+
77
+ §provider-monetary-evidence A provider response may carry an authoritative
78
+ settled `charge`; it wins over every local calculation. Otherwise
79
+ `calculateCharge` returns estimated, explicitly free, or unknown evidence under
80
+ {§provider-cost}. The numeric `calculateCost` method remains only as a frozen
81
+ 1.x compatibility surface: a positive value adapts to an estimate, while zero
82
+ adapts to unknown because it cannot prove free.
58
83
 
59
84
  ### Generation
60
85
 
@@ -64,13 +89,13 @@ fabricated rate.
64
89
  - caller cancellation through `signal`;
65
90
  - optional `grammar` and `maxTokens`;
66
91
  - standard `sampling` intent;
67
- - first-party attribution, client, strike, workspace, loop, and turn metadata.
92
+ - opaque attribution tags plus client, strike, workspace, loop, and turn metadata.
68
93
 
69
- It returns the model's raw content and reasoning, normalized usage, normalized
70
- finish reason, model identity, opaque evidence, optional metadata, and optional
71
- telemetry. The provider transports and observes model output; it never retries,
72
- discards, or repairs an otherwise completed exchange because PLURNK grammar did
73
- not accept it.
94
+ A successful return carries the model's raw content and reasoning, normalized
95
+ usage, normalized finish reason, model identity, opaque evidence, optional
96
+ metadata, and optional notices. The provider transports and observes model
97
+ output; it never retries, discards, or repairs an otherwise completed exchange
98
+ because PLURNK grammar did not accept it.
74
99
 
75
100
  Usage obeys:
76
101
 
@@ -79,13 +104,38 @@ total = prompt + completion + reasoning
79
104
  cached ⊆ prompt
80
105
  ```
81
106
 
82
- `completion` excludes reasoning. Known vendor finish reasons normalize to
107
+ `completion` excludes reasoning. Ordinary vendor finish reasons normalize to
83
108
  `stop`, `length`, `tool_calls`, or `content_filter`; an unknown value becomes
84
- `null` and emits a warning.
109
+ `null` and emits a warning. `resource_interrupted` is the distinct failed-attempt
110
+ disposition defined by {§provider-interrupted-attempt}.
111
+
112
+ ### Tagged reasoning responses
113
+
114
+ §provider-tagged-reasoning Structured provider or SDK reasoning fields are
115
+ authoritative. Visible content is interpreted as tagged reasoning only under an
116
+ explicit alias-scoped response style:
117
+
118
+ | Effective style | Leading content | Normalized result |
119
+ | --------------- | ---------------------------------------------------------- | --------------------------------------------------------------------------------------------------------- |
120
+ | `verbatim` | Any bytes. | Content remains exact; only structured reasoning fields populate `reasoning`. |
121
+ | `think-tags` | No exact leading `<think>`. | Content remains exact. |
122
+ | `think-tags` | `<think>reasoning</think>visible`. | The first envelope body becomes reasoning; the exact suffix becomes content. Later tags remain literal. |
123
+ | `think-tags` | `<think>reasoning` with no close, including a capped turn. | The complete post-open tail becomes reasoning; content is empty. |
124
+
125
+ Tag projection never runs when readable structured reasoning is already
126
+ present. Streamed and buffered transports converge on this response boundary.
127
+ When the upstream reports only combined output usage, the existing
128
+ sum-preserving text-proportion estimate reclassifies completion versus reasoning
129
+ without changing prompt, cached input, total output, or billed output.
130
+
131
+ Grammar evidence retains the exact pre-projection sentence and its Unicode
132
+ content offset. Response classification cannot rewrite what a transported GBNF
133
+ rail observed.
85
134
 
86
135
  ## §3 AI SDK boundary
87
136
 
88
- Cataloged providers instantiate their Models.dev-declared AI SDK package.
137
+ §provider-sdk-boundary Cataloged providers instantiate their
138
+ Models.dev-declared AI SDK package.
89
139
  Standard request shaping, streaming, retries, cancellation, timeouts, usage,
90
140
  and vendor error parsing belong to the SDK.
91
141
 
@@ -100,6 +150,15 @@ PLURNK maps its generic settings to AI SDK call settings:
100
150
  Provider-specific options are permitted only where they preserve a documented
101
151
  PLURNK product contract the generic SDK surface cannot express.
102
152
 
153
+ §deepseek-reasoning-request The direct DeepSeek catalog path maps the common
154
+ reasoning intent to its OpenAI-compatible controls:
155
+
156
+ | PLURNK mode | `thinking` | `reasoning_effort` |
157
+ | ----------- | --------------------- | -------------------- |
158
+ | `off` | `{ type: disabled }` | omitted |
159
+ | `adaptive` | omitted | omitted |
160
+ | `on` | `{ type: enabled }` | budget-derived tier |
161
+
103
162
  The compatible transport is deliberately retained for:
104
163
 
105
164
  - `openai` local endpoints, including llama-server and vLLM;
@@ -112,7 +171,8 @@ SDK's ordinary transport.
112
171
 
113
172
  ## §4 Operator configuration
114
173
 
115
- Every operational value is an environment knob documented in `.env.defaults`.
174
+ §provider-configuration Every operational value is an environment knob
175
+ documented in `.env.defaults`.
116
176
  There are no hidden tuning constants. Every `PLURNK_PROVIDERS_*` knob may be
117
177
  scoped to an alias by appending `_<alias>`; the scoped value wins.
118
178
 
@@ -125,6 +185,7 @@ knob's owning contract.
125
185
  The universal groups are:
126
186
 
127
187
  - reasoning activation and optional explicit budget;
188
+ - explicit reasoning response-content style;
128
189
  - decode tuning;
129
190
  - request, stream-idle, retry, and probe budgets;
130
191
  - local GBNF and llama-server capability pins;
@@ -138,10 +199,27 @@ defaults.
138
199
 
139
200
  ## §5 Resolution
140
201
 
141
- `PLURNK_MODEL_<alias>=<provider>/<model-id>` declares an alias.
202
+ §provider-resolution `PLURNK_MODEL_<alias>=<provider>/<model-id>` declares an
203
+ alias.
142
204
  `PLURNK_MODEL=<alias>` selects the boot alias. Model IDs may contain `/`.
143
205
  `PLURNK_BASEURL_<alias>` is a per-alias endpoint override.
144
206
 
207
+ ### §model-fact-resolution Model fact precedence
208
+
209
+ Provider and model facts resolve independently:
210
+
211
+ | Fact | Natural source | Operator source | Effective value |
212
+ | -------------------- | ------------------------------------------------------ | ---------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------ |
213
+ | Context window | Catalog metadata or a local endpoint probe. | `PLURNK_PROVIDERS_CONTEXT_WINDOW`. | Minimum when both exist; the sole value otherwise. A cataloged cloud miss fails construction; a compatible probe miss remains `null` with one warning. |
214
+ | Completion envelope | Catalog `maxOutput`; there is no live limit probe. | `PLURNK_PROVIDERS_COMPLETION_RESERVE`. | An absolute reserve wins. A percentage derives from the effective window and is capped by catalog `maxOutput` when present. |
215
+ | Reasoning capability | Catalog `reasoning: true`, exposed by snapshot lookup. | Runtime activation, reserve, and adapter wire style. | The catalog bit is informational; provider construction neither activates nor blocks reasoning from it. |
216
+ | Estimated USD rates | Catalog input, output, and optional cache-read rates. | Complete input/output rate override; cache optional. | Operator rates win; otherwise catalog rates apply. Missing rates produce unknown evidence; explicit all-zero rates produce free evidence. No live price fetch exists. |
217
+
218
+ The cached-input override defaults to the explicit input rate when omitted.
219
+ Catalog cache-read cost likewise defaults to catalog input cost. Catalog
220
+ cache-write cost remains snapshot information; the current usage estimator has
221
+ no cache-write quantity to price.
222
+
145
223
  `instantiateProvider` resolves in this order:
146
224
 
147
225
  1. A Models.dev provider and model, using its declared AI SDK package.
@@ -181,29 +259,35 @@ belongs in MCP, schemes, executors, or mimetypes instead.
181
259
 
182
260
  A provider plugin:
183
261
 
184
- 1. declares `plurnk: { kind: "provider", name }` in `package.json`;
185
- 2. may use any npm scope;
186
- 3. default-exports an AI SDK provider with `languageModel(modelId)`;
187
- 4. peers on compatible `ai` and `@plurnk/plurnk-providers` majors.
262
+ 1. declares the exact string `plurnk: { kind: "provider", name }` in `package.json` ({§plugin-family-kind});
263
+ 2. may declare always-on package-level `plurnk.attribution` and/or implement the
264
+ synchronous runtime `attributions(context)` hook under {§plugin-attribution};
265
+ 3. may use any npm scope;
266
+ 4. default-exports an AI SDK provider with `languageModel(modelId)`;
267
+ 5. peers on compatible `ai` and `@plurnk/plurnk-providers` majors.
188
268
 
189
269
  PLURNK adapts the returned language model. The plugin does not implement the
190
270
  PLURNK `Provider`, read PLURNK tuning knobs, or reproduce transport policy.
191
271
 
192
272
  Discovery is scope-agnostic and memoized per process. Duplicate names fail hard.
193
- The common plugin trust gate applies before import. A plugin absent from
273
+ The common plugin trust gate applies before import ({§plugin-trust-boundary}). A plugin absent from
194
274
  Models.dev requires an explicit context-window pin because PLURNK will not guess
195
- model physics.
275
+ model physics. `Discovery.packageAttributions` carries the canonical package map;
276
+ the published name-keyed `Discovery.attributions` remains its 1.x projection.
196
277
 
197
278
  ## §7 Local capabilities
198
279
 
199
- The `openai` local adapter probes `/v1/models`. A llama-server fingerprint may
280
+ §provider-local-capabilities The `openai` local adapter probes `/v1/models`. A
281
+ llama-server fingerprint may
200
282
  also expose:
201
283
 
202
284
  - the actual served model and per-slot context window;
285
+ - request-scoped reasoning parsing and a cumulative response-wide allowance;
203
286
  - GBNF constrained sampling;
204
287
  - slot count and worker-sticky slot affinity;
205
288
  - EOS marker removal;
206
- - exact `/tokenize`;
289
+ - exact complete-request counting through `/v1/chat/completions/input_tokens`;
290
+ - exact content token IDs through `/tokenize`;
207
291
  - the requirement that the caller provide `maxTokens`.
208
292
 
209
293
  `PLURNK_PROVIDERS_LLAMA_SERVER` may force or disable detection. Probe attempts
@@ -212,9 +296,31 @@ and delay are knobs. A failed probe does not silently assert capabilities.
212
296
  Ollama probes `/api/show` for its model context and uses its documented
213
297
  OpenAI-compatible generation endpoint through the SDK adapter.
214
298
 
299
+ ### llama-server reasoning
300
+
301
+ For a detected llama-server, PLURNK sends the complete reasoning contract on
302
+ every request:
303
+
304
+ | Mode | Template activation | `thinking_budget_tokens` |
305
+ |---|---:|---:|
306
+ | `off` | false | `0` |
307
+ | `adaptive` | true | resolved reasoning reserve |
308
+ | `on` | true | explicit reasoning budget |
309
+
310
+ The explicit budget may tighten but MUST NOT exceed the resolved reasoning
311
+ reserve. `reasoning_format: "auto"` requests a separate readable reasoning
312
+ channel. Process-wide llama-server flags are fallback server configuration, not
313
+ part of the PLURNK contract and need not be synchronized with an alias.
314
+
315
+ §llama-reasoning-request The allowance is cumulative across the complete response. Opening a second or
316
+ later reasoning block does not replenish it. Template parsing, the reasoning
317
+ sampler, normalized usage, and the returned reasoning channel MUST agree on that
318
+ response boundary.
319
+
215
320
  ## §8 Request authority
216
321
 
217
- The caller's `sampling` bag expresses sampling intent. It cannot override:
322
+ §provider-request-authority The caller's `sampling` bag expresses sampling
323
+ intent. It cannot override:
218
324
 
219
325
  - model or messages;
220
326
  - stream mode;
@@ -235,9 +341,17 @@ backend.
235
341
 
236
342
  ## §9 Failures, retries, and cancellation
237
343
 
238
- Provider failures normalize to `ProviderError` with source, kind, status where
239
- available, and the original cause. A caught failure is surfaced or deliberately
240
- preserved; it is never converted into an empty model turn.
344
+ §provider-failure-normalization Provider failures normalize to `ProviderError`.
345
+ Its public contract is an RFC
346
+ 9457 Problem Details object with an exact status, stable type, occurrence
347
+ detail, and provider-kind extension; the original error remains its cause.
348
+ A caught failure is surfaced or deliberately preserved; it is never converted
349
+ into an empty model turn or reduced to a message plus a generic status.
350
+ Upstream diagnostic text is bounded by
351
+ `PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT`; the committed `.env.defaults` owns its
352
+ normal value. Retry exhaustion is preserved as `attempts` and
353
+ `retryExhausted`, and the resulting Problem is not marked retryable after the
354
+ provider has consumed its automatic retry budget.
241
355
 
242
356
  The AI SDK owns attempt scheduling. `PLURNK_PROVIDERS_RETRY_ATTEMPTS` is the
243
357
  maximum retry count. Caller cancellation spans the operation. Total and
@@ -247,20 +361,51 @@ HTTP 408, 409, 429, and ordinary 5xx responses are retryable unless the endpoint
247
361
  explicitly says otherwise. Endpoint control responses 520–527 are final so a
248
362
  router can prevent multiplicative retries behind its own retry policy.
249
363
 
364
+ ### §provider-interrupted-attempt Provider-declared interruption
365
+
366
+ A successful transport response can still declare that inference did not
367
+ complete. That response is evidence for one failed provider attempt, never a
368
+ completed exchange.
369
+
370
+ | Concern | Contract |
371
+ | -------------------------- | ---------------------------------------------------------------------------------------------------------------------------- |
372
+ | Normalized finish reason | Exact `insufficient_system_resource` becomes `resource_interrupted`; the raw value remains in `assistantRaw`. |
373
+ | `generate` outcome | Throw `ProviderError(kind="resource_interrupted")` at local status 503 with the normalized attempt on `error.attempt`. |
374
+ | Partial response | Preserve content, reasoning, usage, model, metadata, optional raw body, and other evidence without admitting it as success. |
375
+ | Automatic replay | None; the Problem has `retryable: false`, and AI SDK retry scheduling has already completed at the successful transport. |
376
+ | Capacity-pool overflow | None; a sibling success would erase the known billed failed attempt from the current `ProviderResponse` surface. |
377
+ | Consumer admission | Persist the evidence as an unaccepted attempt; never parse it into executable work, even when its frame looks complete. |
378
+
250
379
  ## §10 Grammar
251
380
 
252
381
  GBNF is a local llama-server capability, not a generic provider expectation.
253
382
  The consumer chooses whether to supply a grammar. The provider never creates or
254
383
  rewrites one.
255
384
 
256
- When transported, output is validated locally after completion. Divergence
257
- attaches `grammar_unenforced` telemetry with its position; the bytes still
258
- return. `PLURNK_PROVIDERS_GBNF_DEBUG` validates but withholds the grammar and
259
- compares the unconstrained result for diagnostics.
385
+ GBNF defines the accepted raw sampled text; it runs before response reasoning
386
+ is separated from regular content. The shipped PLURNK sentence is owned by
387
+ `plurnk-contracts` {§gbnf-turn-shape} and {§gbnf-reasoning-boundary}; this package does
388
+ not restate or rewrite it.
389
+
390
+ When a grammar-capable adapter receives a grammar, `ProviderResponse` carries
391
+ `grammarEvidence: { input, contentStart, transported }`. `input` is the exact
392
+ pre-projection sentence represented by the response, `contentStart` is its
393
+ Unicode-code-point offset to `assistant.content`, and `transported` says whether
394
+ the grammar was actually sent. For active llama-server template reasoning, the
395
+ adapter reconstructs the Harmony enclosure only when the response demonstrates
396
+ that `reasoning_format: "auto"` projected it; otherwise it cannot claim evidence.
397
+ For an unsplit response, `input` is `content` and `contentStart` is zero.
398
+
399
+ §gbnf-response-observation The provider transports and represents; it does not grade its own enforcement.
400
+ The consumer validates `grammarEvidence.input` outside the enforcer's failure
401
+ domain. `PLURNK_PROVIDERS_GBNF_DEBUG` still validates grammar syntax before the
402
+ call and sets `transported: false` for the unconstrained comparison.
403
+
260
404
 
261
405
  ## §11 Evidence and metadata
262
406
 
263
- `assistantRaw` is an opaque normalized transport record. Provider top-level
407
+ §provider-evidence `assistantRaw` is an opaque normalized transport record.
408
+ Provider top-level
264
409
  metadata is forwarded as an open bag without reinterpreting currencies or
265
410
  vendor fields.
266
411
 
@@ -270,25 +415,42 @@ When enabled, raw per-token model logprob is canonical; alternatives are
270
415
  preserved when returned. Raw body/chunks preserve wire evidence the normalized
271
416
  record omits.
272
417
 
273
- Readable reasoning and encrypted reasoning are separate. Encrypted reasoning is
274
- preserved verbatim and never decoded or synthesized.
418
+ §provider-encrypted-reasoning **Readable reasoning and encrypted reasoning are
419
+ separate.** Encrypted payload bytes remain opaque and are never decoded. The
420
+ provider boundary distinguishes preserved detail evidence from derived entity
421
+ classification:
422
+
423
+ | Provider fact | Meaning |
424
+ | -------------------------------- | ------- |
425
+ | `id` | The provider's reasoning-detail identity, or `null`; never an AG-UI message or tool-call identity. |
426
+ | `subtype` | A provider-normalized classification supported by wire structure. OpenAI-compatible `message.reasoning_details` is `message`; no PLURNK operation is reclassified as a native tool call. |
427
+ | `encrypted[*].data` / `format` | Ordered provider evidence, retained without decoding or concatenation across distinct details. |
428
+
429
+ Unrecognized detail shapes are omitted at this normalization boundary. Core
430
+ may preserve normalized items as forensic evidence, but a client protocol must
431
+ correlate them to an entity it actually created rather than reusing `id`.
275
432
 
276
433
  ## §12 Generation envelopes
277
434
 
278
- Reasoning and completion reserves are percentages of the resolved context
435
+ §provider-generation-envelope Reasoning and completion reserves are percentages
436
+ of the resolved context
279
437
  window or absolute token counts. Absolute pins win. The provider reports the
280
- resolved reserves; the consumer owns prompt packing and the per-call output cap.
281
-
282
- Models.dev `maxOutput` constrains a percentage-derived completion reserve but
283
- does not override an explicit absolute operator choice. Unknown model physics
284
- remain unknown.
438
+ resolved reserves; the consumer owns prompt packing and the per-call output
439
+ cap. Model-limit precedence is defined once in {§model-fact-resolution}.
285
440
 
286
441
  ## §13 Capacity pool
287
442
 
288
- `Pool` fronts interchangeable `Provider` instances. It keeps workers sticky for
443
+ §provider-capacity-pool `Pool` fronts interchangeable `Provider` instances. It
444
+ keeps workers sticky for
289
445
  cache locality, selects a healthy sibling for overflow, and preserves the same
290
446
  Provider contract. Whether endpoints are interchangeable is a consumer
291
- decision, not inferred from provider names.
447
+ decision, not inferred from provider names. Overflow is limited to transport
448
+ availability and rate-limit failures that carry no normalized response attempt;
449
+ {§provider-interrupted-attempt} propagates without overflow.
450
+
451
+ Prompt measurement covers every backend that could receive the request. The
452
+ pool takes the largest result; differing exact counts or any proven bound yield
453
+ an `upper_bound`, while any estimate makes the aggregate an estimate.
292
454
 
293
455
  ## §14 Conformance
294
456
 
@@ -298,8 +460,12 @@ Coverage MUST prove:
298
460
  - exact and unique-suffix model lookup;
299
461
  - native SDK request mapping and normalized responses;
300
462
  - compatible extension preservation;
301
- - timeout, retry, cancellation, and final-error behavior;
463
+ - timeout, retry, cancellation, interrupted-attempt, and final-error behavior;
302
464
  - local capability probes and pins;
465
+ - exact, bounded, and estimated complete-request token measurements;
466
+ - local reasoning activation, response-wide allowance, and GBNF coexistence;
467
+ - explicit tagged-reasoning projection across streamed, buffered, capped, and
468
+ literal-tag responses;
303
469
  - usage, costs, evidence, metadata isolation, and grammar observation;
304
470
  - alias scoping and fail-hard invalid configuration.
305
471