@plurnk/plurnk-providers 1.5.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +36 -22
- package/SPEC.md +133 -59
- package/dist/AiSdkProvider.d.ts +19 -26
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +318 -106
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +4 -9
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +36 -9
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -21
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +19 -14
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts +5 -2
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +100 -16
- package/dist/accounting.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +9 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +160 -62
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +7 -3
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +30 -24
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +18 -7
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +10 -10
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +90 -42
- package/dist/cost.js.map +1 -1
- package/dist/env.d.ts +5 -1
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +30 -10
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +14 -2
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +58 -2
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -2
- package/dist/index.js.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/sdkModels.d.ts +6 -2
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +38 -5
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +33 -31
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +21 -5
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +164 -83
- package/dist/usage.js.map +1 -1
- package/package.json +7 -6
- package/src/AiSdkProvider.test.ts +788 -191
- package/src/AiSdkProvider.ts +381 -124
- package/src/Mock.test.ts +37 -12
- package/src/Mock.ts +45 -14
- package/src/Pool.test.ts +19 -6
- package/src/Pool.ts +20 -16
- package/src/ProviderRegistry.test.ts +16 -11
- package/src/accounting.test.ts +58 -22
- package/src/accounting.ts +120 -18
- package/src/aiSdkTransport.test.ts +42 -49
- package/src/aiSdkTransport.ts +174 -62
- package/src/boundaries.test.ts +1 -0
- package/src/catalogProvider.test.ts +258 -22
- package/src/catalogProvider.ts +42 -27
- package/src/compatibleProvider.test.ts +6 -3
- package/src/compatibleProvider.ts +20 -7
- package/src/cost.test.ts +55 -36
- package/src/cost.ts +111 -50
- package/src/defaults.test.ts +13 -3
- package/src/env.test.ts +54 -5
- package/src/env.ts +43 -18
- package/src/errors.test.ts +47 -2
- package/src/errors.ts +67 -3
- package/src/index.ts +21 -5
- package/src/ollama.test.ts +4 -1
- package/src/ollama.ts +3 -3
- package/src/sdkModels.test.ts +76 -4
- package/src/sdkModels.ts +45 -7
- package/src/types.ts +77 -38
- package/src/usage.test.ts +112 -116
- package/src/usage.ts +209 -93
package/.env.defaults
CHANGED
|
@@ -17,13 +17,12 @@
|
|
|
17
17
|
# ACTIVATION and BUDGET are separate so a numeric can never silently flip wire flags.
|
|
18
18
|
# off | adaptive | on. The provider maps intent to each backend's native mechanism
|
|
19
19
|
# (reasoning_effort, enable_thinking, think, ...). Default ADAPTIVE (#399):
|
|
20
|
-
#
|
|
21
|
-
#
|
|
22
|
-
# declares done.
|
|
20
|
+
# defer activation and depth to the backend's documented default. Use an alias-scoped
|
|
21
|
+
# ON when a reasoning-capable model defaults off and the operator wants it enabled.
|
|
23
22
|
PLURNK_PROVIDERS_REASONING=adaptive
|
|
24
|
-
#
|
|
25
|
-
#
|
|
26
|
-
# cannot exceed, the resolved
|
|
23
|
+
# Optional positive int when REASONING=on. Without one, ON activates reasoning at
|
|
24
|
+
# the adapter's ordinary enabled posture. On llama-server an explicit value is a
|
|
25
|
+
# request-scoped allowance and may tighten, but cannot exceed, the resolved reserve.
|
|
27
26
|
# PLURNK_PROVIDERS_REASONING_BUDGET=4096
|
|
28
27
|
|
|
29
28
|
# Response-content interpretation ({§provider-tagged-reasoning}) is independent
|
|
@@ -48,10 +47,14 @@ PLURNK_PROVIDERS_FREQUENCY_PENALTY=0
|
|
|
48
47
|
# Fixed provider service tier. Fireworks accepts auto|default|flex|priority;
|
|
49
48
|
# normally set per alias so a paid routing choice is explicit.
|
|
50
49
|
# PLURNK_PROVIDERS_SERVICE_TIER_myfireworks=priority
|
|
51
|
-
#
|
|
52
|
-
#
|
|
53
|
-
#
|
|
54
|
-
|
|
50
|
+
# Route adapters project the stable worker identity through only the provider's
|
|
51
|
+
# documented session/cache-affinity control. Unknown compatible routes receive
|
|
52
|
+
# no guessed field. Disable globally or per alias only for operational diagnosis.
|
|
53
|
+
PLURNK_PROVIDERS_CACHE_AFFINITY=1
|
|
54
|
+
# Explicit cache writes are distinct from affinity and may affect billing.
|
|
55
|
+
# stable-system marks only the reusable system boundary on supported Claude
|
|
56
|
+
# routes; off requests no explicit cache write. Provider-default lifetime is 5m.
|
|
57
|
+
PLURNK_PROVIDERS_CACHE_WRITE_POLICY=stable-system
|
|
55
58
|
# #567: DRY is a llama.cpp-only repeated-sequence penalty. It can reduce
|
|
56
59
|
# degenerate loops, but it can also corrupt exact source, identifiers, quoted
|
|
57
60
|
# evidence, and other repetition required by PLURNK operations. The portable
|
|
@@ -64,16 +67,27 @@ PLURNK_PROVIDERS_DRY_MULTIPLIER=0
|
|
|
64
67
|
# Set it per alias only from measured model behavior.
|
|
65
68
|
# PLURNK_PROVIDERS_REPEAT_LAST_N_<alias>=512
|
|
66
69
|
|
|
67
|
-
# ---
|
|
68
|
-
#
|
|
70
|
+
# --- Connectivity budgets ({§provider-connectivity}, #240) ---
|
|
71
|
+
# Complete logical generation deadline (ms), spanning every physical attempt
|
|
72
|
+
# and retry delay. Zero disables this outer deadline; caller cancellation still
|
|
73
|
+
# spans the operation. The floor leaves all four ten-minute attempts plus normal
|
|
74
|
+
# backoff available under the three-retry floor.
|
|
75
|
+
PLURNK_PROVIDERS_OPERATION_TIMEOUT=2700000
|
|
76
|
+
# Maximum duration (ms) of one physical generation attempt. Zero disables the
|
|
77
|
+
# per-attempt deadline without changing the operation deadline.
|
|
69
78
|
PLURNK_PROVIDERS_FETCH_TIMEOUT=600000
|
|
70
|
-
# Maximum
|
|
71
|
-
#
|
|
72
|
-
#
|
|
73
|
-
#
|
|
79
|
+
# Maximum wait (ms) from response-stream start to first semantic model content.
|
|
80
|
+
# Transport metadata and empty deltas do not satisfy it. The portable floor
|
|
81
|
+
# allows endpoints that intentionally buffer a full attempt; tighten per alias
|
|
82
|
+
# from measured time-to-first-content. Zero disables it.
|
|
83
|
+
PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT=600000
|
|
84
|
+
# Maximum silence (ms) between semantic streamed-content chunks after content
|
|
85
|
+
# begins. Zero disables it. Both content deadlines consume the ordinary retry
|
|
86
|
+
# budget when they expire.
|
|
74
87
|
PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT=120000
|
|
75
88
|
# Transient-failure retries: 0 = surface the first failure; N = retries on
|
|
76
|
-
# 429/5xx
|
|
89
|
+
# network failure, 408/409/429/ordinary 5xx, or an inner deadline, with the AI
|
|
90
|
+
# SDK's backoff (Retry-After wins).
|
|
77
91
|
PLURNK_PROVIDERS_RETRY_ATTEMPTS=3
|
|
78
92
|
# Maximum characters retained from an upstream provider diagnostic in the
|
|
79
93
|
# public Problem detail. Structured failure facts are not truncated.
|
|
@@ -89,16 +103,16 @@ PLURNK_PROVIDERS_PROBE_DELAY=250
|
|
|
89
103
|
# Unset by default. Set per alias only for a local llama-server whose GBNF
|
|
90
104
|
# transport is detected or pinned. Cloud and endpoint-managed model settings do
|
|
91
105
|
# not use this knob.
|
|
92
|
-
# PLURNK_PROVIDERS_GBNF=plurnk.gbnf
|
|
106
|
+
# PLURNK_PROVIDERS_GBNF=plurnk.qwen.gbnf
|
|
93
107
|
# Debug toggle: validate but withhold a configured local GBNF, then report the
|
|
94
108
|
# unconstrained output's divergence. Development aid; leave unset in production.
|
|
95
109
|
# PLURNK_PROVIDERS_GBNF_DEBUG=0
|
|
96
110
|
|
|
97
111
|
# --- Window ({§model-fact-resolution}) ---
|
|
98
|
-
#
|
|
99
|
-
# models.dev, else null (surfaced once via PLURNK_CONTEXT_UNKNOWN). A configured
|
|
100
|
-
#
|
|
101
|
-
# pressure
|
|
112
|
+
# Effective total context envelope. Unset derives natural capacity from a live endpoint probe
|
|
113
|
+
# (n_ctx) or models.dev, else null (surfaced once via PLURNK_CONTEXT_UNKNOWN). A configured
|
|
114
|
+
# value is a final hard cap on known capacity or declares the envelope when unknown. Model-facing
|
|
115
|
+
# prompt pressure is separate consumer policy.
|
|
102
116
|
# PLURNK_PROVIDERS_CONTEXT_WINDOW=200000
|
|
103
117
|
|
|
104
118
|
# --- llama-server detection pin (#34) ---
|
package/SPEC.md
CHANGED
|
@@ -42,24 +42,23 @@ interface Provider {
|
|
|
42
42
|
signal?: AbortSignal,
|
|
43
43
|
): Promise<PromptTokenMeasurement>;
|
|
44
44
|
tokenize?(text: string): Promise<number[]>;
|
|
45
|
-
calculateCost(usage: ProviderUsage): number;
|
|
46
|
-
calculateCharge?(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }>;
|
|
47
45
|
generate(args: GenerateArgs): Promise<ProviderResponse>;
|
|
48
46
|
}
|
|
49
47
|
```
|
|
50
48
|
|
|
51
|
-
`contextWindow` is
|
|
52
|
-
{§model-fact-resolution}
|
|
53
|
-
|
|
54
|
-
|
|
49
|
+
`contextWindow` is the effective total context envelope resolved under
|
|
50
|
+
{§model-fact-resolution}: the minimum of known model capacity and any stricter
|
|
51
|
+
operator cap. `null` means genuinely unknown; a consumer MUST NOT invent a
|
|
52
|
+
stand-in. The context-window knob is a hard cap, never model-facing grinder
|
|
53
|
+
pressure.
|
|
55
54
|
|
|
56
55
|
`PromptTokenMeasurement` is a discriminated request-level result:
|
|
57
56
|
|
|
58
|
-
| `kind` | Meaning | May authorize
|
|
59
|
-
| ------------- | ------------------------------------------------------ |
|
|
60
|
-
| `exact` | Exact count for the complete provider request. | Yes.
|
|
61
|
-
| `upper_bound` | Proven upper bound for the complete provider request. | Yes.
|
|
62
|
-
| `estimate` | Empirical prediction with required causal `detail`. | No.
|
|
57
|
+
| `kind` | Meaning | May authorize hard context-envelope admission |
|
|
58
|
+
| ------------- | ------------------------------------------------------ | --------------------------------------------- |
|
|
59
|
+
| `exact` | Exact count for the complete provider request. | Yes. |
|
|
60
|
+
| `upper_bound` | Proven upper bound for the complete provider request. | Yes. |
|
|
61
|
+
| `estimate` | Empirical prediction with required causal `detail`. | No. |
|
|
63
62
|
|
|
64
63
|
Every result carries a non-negative integer `tokens` and a non-empty `source`.
|
|
65
64
|
`countPromptTokens` receives the same messages supplied to `generate` and may
|
|
@@ -74,45 +73,72 @@ hard; consumers do not reinterpret it as ordinary unavailability.
|
|
|
74
73
|
endpoint exposes its real vocabulary. Content tokenization does not substitute
|
|
75
74
|
for complete-request measurement.
|
|
76
75
|
|
|
77
|
-
§provider-monetary-evidence One precedence path converts each provider
|
|
78
|
-
into {§provider-cost} evidence
|
|
76
|
+
§provider-monetary-evidence One precedence path converts each physical provider
|
|
77
|
+
request into {§provider-cost} evidence before the request leaves the provider
|
|
78
|
+
boundary:
|
|
79
79
|
|
|
80
80
|
| Precedence | Evidence | Result |
|
|
81
81
|
| --- | --- | --- |
|
|
82
|
-
| 1 | A documented monetary field on that response | The adapter
|
|
83
|
-
| 2 |
|
|
84
|
-
| 3 | Neither |
|
|
82
|
+
| 1 | A documented monetary field on that response or error | The adapter validates and preserves its documented `charged` or `estimated` character; its exact amount wins. |
|
|
83
|
+
| 2 | Known response usage and the exact model's Models.dev rates | The adapter returns an exact decimal USD `estimated` amount only when every differently-priced applicable category is known. |
|
|
84
|
+
| 3 | Neither | `unknown` with a concrete reason. |
|
|
85
85
|
|
|
86
|
-
Models.dev is the sole supported fallback rate table. Missing usage
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
86
|
+
Models.dev is the sole supported fallback rate table. Missing usage, a missing
|
|
87
|
+
applicable category, or missing rates never proves zero. An exact zero rate
|
|
88
|
+
produces an ordinary estimated amount of USD `0`. Rate calculation is internal
|
|
89
|
+
to the provider request; Core, digest, ping, and clients never call a parallel
|
|
90
|
+
pricing method.
|
|
91
91
|
|
|
92
92
|
### Generation
|
|
93
93
|
|
|
94
|
-
`generate` requires a non-empty, stable, opaque
|
|
94
|
+
§provider-cache-identity `generate` requires a non-empty, stable, opaque
|
|
95
|
+
`workerId`. A durable worker uses one globally unique value for its lifetime;
|
|
96
|
+
independent databases and processes cannot mint the same local sequence. A
|
|
97
|
+
`bare` call instead uses a fresh per-call value, preventing unrelated prompts
|
|
98
|
+
from acquiring affinity with either the parent worker or another BARE call.
|
|
99
|
+
Providers MUST NOT interpret either value.
|
|
100
|
+
|
|
101
|
+
`generate` accepts:
|
|
95
102
|
|
|
96
103
|
- `messages`: system, user, and assistant text messages;
|
|
97
104
|
- caller cancellation through `signal`;
|
|
98
105
|
- optional `grammar` and `maxTokens`;
|
|
99
106
|
- standard `sampling` intent;
|
|
107
|
+
- the caller-owned `callKind` output contract when one applies;
|
|
100
108
|
- opaque attribution tags plus client, strike, workspace, loop, and turn metadata.
|
|
101
109
|
|
|
110
|
+
§provider-call-kind `callKind` is either `emission` (the response is a PLURNK
|
|
111
|
+
turn emission) or `bare` (the response is unconstrained answer text). The
|
|
112
|
+
consumer states this semantic fact explicitly; providers MUST NOT infer it from
|
|
113
|
+
message count, grammar presence, worker identity, or another incidental request
|
|
114
|
+
shape. The first-party adapter transports a supplied value as
|
|
115
|
+
`Plurnk-Call-Kind`; the metadata gate drops it for every third-party backend.
|
|
116
|
+
The signal is request metadata and never enters model-facing messages. Generic
|
|
117
|
+
provider callers MAY omit it; Core supplies it for every model call.
|
|
118
|
+
|
|
102
119
|
A successful return carries the model's raw content and reasoning, normalized
|
|
103
|
-
|
|
104
|
-
metadata, and optional notices.
|
|
120
|
+
finish reason, model identity, its ordered {§provider-request-accounting},
|
|
121
|
+
opaque evidence, optional metadata, and optional notices. A `ProviderError`
|
|
122
|
+
carries the same accounting array. The provider transports and observes model
|
|
105
123
|
output; it never retries, discards, or repairs an otherwise completed exchange
|
|
106
124
|
because PLURNK grammar did not accept it.
|
|
107
125
|
|
|
108
|
-
|
|
126
|
+
§provider-request-observer When a consumer supplies the request observer, the
|
|
127
|
+
provider opens one durable identity through it immediately before each physical
|
|
128
|
+
I/O and settles that identity with the resulting
|
|
129
|
+
`ProviderRequestAccounting`. This applies to every automatic retry and capacity
|
|
130
|
+
failover request. The observer is a durability sink, not an alternate evidence
|
|
131
|
+
representation; the same ordered records remain on the final response or error.
|
|
132
|
+
|
|
133
|
+
Usage obeys {§provider-usage}:
|
|
109
134
|
|
|
110
135
|
```text
|
|
111
|
-
|
|
112
|
-
|
|
136
|
+
totalTokens = inputTokens + outputTokens
|
|
137
|
+
cacheReadTokens, cacheWriteTokens ⊆ inputTokens
|
|
138
|
+
reasoningTokens ⊆ outputTokens
|
|
113
139
|
```
|
|
114
140
|
|
|
115
|
-
|
|
141
|
+
Unknown fields remain absent. Ordinary vendor finish reasons normalize to
|
|
116
142
|
`stop`, `length`, `tool_calls`, or `content_filter`; an unknown value becomes
|
|
117
143
|
`null` and emits a warning. `resource_interrupted` is the distinct failed-attempt
|
|
118
144
|
disposition defined by {§provider-interrupted-attempt}.
|
|
@@ -132,9 +158,9 @@ explicit alias-scoped response style:
|
|
|
132
158
|
|
|
133
159
|
Tag projection never runs when readable structured reasoning is already
|
|
134
160
|
present. Streamed and buffered transports converge on this response boundary.
|
|
135
|
-
When the upstream reports only combined output usage,
|
|
136
|
-
|
|
137
|
-
|
|
161
|
+
When the upstream reports only combined output usage, that value remains
|
|
162
|
+
`outputTokens` and its unavailable text/reasoning detail stays absent. The
|
|
163
|
+
adapter never apportions tokens from character lengths.
|
|
138
164
|
|
|
139
165
|
Grammar evidence retains the exact pre-projection sentence and its Unicode
|
|
140
166
|
content offset. Response classification cannot rewrite what a transported GBNF
|
|
@@ -144,8 +170,10 @@ rail observed.
|
|
|
144
170
|
|
|
145
171
|
§provider-sdk-boundary Cataloged providers instantiate their
|
|
146
172
|
Models.dev-declared AI SDK package.
|
|
147
|
-
Standard request shaping, streaming,
|
|
148
|
-
and
|
|
173
|
+
Standard request shaping, streaming, usage, and vendor error parsing belong to
|
|
174
|
+
the SDK. PLURNK supplies cancellation and deadline signals and owns the sole
|
|
175
|
+
cross-attempt scheduler so every physical request remains observable and
|
|
176
|
+
accountable.
|
|
149
177
|
|
|
150
178
|
PLURNK maps its generic settings to AI SDK call settings:
|
|
151
179
|
|
|
@@ -153,19 +181,48 @@ PLURNK maps its generic settings to AI SDK call settings:
|
|
|
153
181
|
- presence and frequency penalties;
|
|
154
182
|
- stop sequences and seed;
|
|
155
183
|
- output-token ceiling;
|
|
156
|
-
- `off`, `adaptive`, or
|
|
184
|
+
- `off`, provider-default `adaptive`, or explicit `on` reasoning intent, with
|
|
185
|
+
an optional operator budget.
|
|
157
186
|
|
|
158
187
|
Provider-specific options are permitted only where they preserve a documented
|
|
159
188
|
PLURNK product contract the generic SDK surface cannot express.
|
|
160
189
|
|
|
190
|
+
§provider-readable-reasoning When the effective reasoning posture is not
|
|
191
|
+
`off`, a native adapter MUST request readable reasoning summaries if its
|
|
192
|
+
provider requires a separate response-visibility option. That option neither
|
|
193
|
+
activates reasoning nor selects its depth. The exact wire projection belongs to
|
|
194
|
+
the provider adapter; Models.dev's reasoning bit remains capability metadata.
|
|
195
|
+
|
|
196
|
+
The portable SDK surface has no boolean-enabled reasoning value. An unqualified
|
|
197
|
+
`on` therefore projects to its conventional `medium` enabled posture. This is a
|
|
198
|
+
wire activation value, not a reasoning reserve or output-token ceiling.
|
|
199
|
+
|
|
200
|
+
§provider-cache-affinity **Cache affinity is route-owned request projection.**
|
|
201
|
+
When a provider documents a semantics-preserving conversation, session, or
|
|
202
|
+
prompt-cache routing key, its catalog adapter projects `workerId` through that
|
|
203
|
+
provider's documented header, body field, or native SDK option. The common
|
|
204
|
+
transport neither guesses from protocol resemblance nor sends a generic cache
|
|
205
|
+
field to an unknown provider. The operator may disable affinity globally or per
|
|
206
|
+
alias; automatic provider caching without an affinity control remains untouched.
|
|
207
|
+
|
|
208
|
+
§provider-cache-write-policy **Cache-write policy is separate from affinity.**
|
|
209
|
+
`PLURNK_PROVIDERS_CACHE_WRITE_POLICY` is `off` or `stable-system`. The latter
|
|
210
|
+
marks only the final leading system instruction as an explicit reusable cache
|
|
211
|
+
boundary, and only on routes whose native SDK documents that control. It does
|
|
212
|
+
not mark the changing user packet or enable an API-wide automatic cache mode.
|
|
213
|
+
Unsupported routes receive no invented option. The default five-minute
|
|
214
|
+
provider lifetime is used; a longer, differently priced lifetime is not an
|
|
215
|
+
implicit transport choice.
|
|
216
|
+
|
|
161
217
|
§deepseek-reasoning-request The direct DeepSeek catalog path maps the common
|
|
162
218
|
reasoning intent to its OpenAI-compatible controls:
|
|
163
219
|
|
|
164
|
-
| PLURNK
|
|
165
|
-
|
|
|
166
|
-
| `off`
|
|
167
|
-
| `adaptive`
|
|
168
|
-
| `on`
|
|
220
|
+
| PLURNK posture | `thinking` | `reasoning_effort` |
|
|
221
|
+
| --------------- | --------------------- | -------------------- |
|
|
222
|
+
| `off` | `{ type: disabled }` | omitted |
|
|
223
|
+
| `adaptive` | omitted | omitted |
|
|
224
|
+
| `on` | `{ type: enabled }` | omitted |
|
|
225
|
+
| `on` + budget | `{ type: enabled }` | budget-derived tier |
|
|
169
226
|
|
|
170
227
|
The compatible transport is deliberately retained for:
|
|
171
228
|
|
|
@@ -195,9 +252,10 @@ The universal groups are:
|
|
|
195
252
|
- reasoning activation and optional explicit budget;
|
|
196
253
|
- explicit reasoning response-content style;
|
|
197
254
|
- decode tuning;
|
|
198
|
-
-
|
|
255
|
+
- operation, physical-attempt, first-content, stream-idle, retry, and probe budgets;
|
|
199
256
|
- local GBNF and llama-server capability pins;
|
|
200
257
|
- context-window and generation-envelope overrides;
|
|
258
|
+
- provider-documented cache affinity and explicit cache-write policy;
|
|
201
259
|
- opt-in logprob and raw-body capture.
|
|
202
260
|
|
|
203
261
|
Operator secrets and machine-specific values never belong in committed
|
|
@@ -219,11 +277,11 @@ Provider and model facts resolve independently:
|
|
|
219
277
|
| Context window | Catalog metadata or a local endpoint probe. | `PLURNK_PROVIDERS_CONTEXT_WINDOW`. | Minimum when both exist; the sole value otherwise. A cataloged cloud miss fails construction; a compatible probe miss remains `null` with one warning. |
|
|
220
278
|
| Completion envelope | Catalog `maxOutput`; there is no live limit probe. | `PLURNK_PROVIDERS_COMPLETION_RESERVE`. | An absolute reserve wins. A percentage derives from the effective window and is capped by catalog `maxOutput` when present. |
|
|
221
279
|
| Reasoning capability | Catalog `reasoning: true`, exposed by snapshot lookup. | Runtime activation, reserve, and adapter wire style. | The catalog bit is informational; provider construction neither activates nor blocks reasoning from it. |
|
|
222
|
-
| Estimated USD rates | Models.dev input, output, and optional cache-read rates. | None. | Missing rates produce unknown evidence;
|
|
280
|
+
| Estimated USD rates | Models.dev input, output, and optional cache-read/cache-write rates. | None. | Missing differently-priced categories or rates produce unknown evidence; exact all-zero rates produce an estimated USD zero. No live price fetch exists. |
|
|
223
281
|
|
|
224
|
-
Models.dev cache-read cost
|
|
225
|
-
|
|
226
|
-
|
|
282
|
+
Models.dev cache-read and cache-write cost default to input cost when omitted.
|
|
283
|
+
When either differs from input, the corresponding usage detail must be known or
|
|
284
|
+
the request estimate is unknown.
|
|
227
285
|
|
|
228
286
|
`instantiateProvider` resolves in this order:
|
|
229
287
|
|
|
@@ -311,11 +369,12 @@ OpenAI-compatible generation endpoint through the SDK adapter.
|
|
|
311
369
|
For a detected llama-server, PLURNK sends the complete reasoning contract on
|
|
312
370
|
every request:
|
|
313
371
|
|
|
314
|
-
|
|
|
372
|
+
| Posture | Template activation | `thinking_budget_tokens` |
|
|
315
373
|
|---|---:|---:|
|
|
316
374
|
| `off` | false | `0` |
|
|
317
375
|
| `adaptive` | true | resolved reasoning reserve |
|
|
318
|
-
| `on` | true |
|
|
376
|
+
| `on` | true | resolved reasoning reserve |
|
|
377
|
+
| `on` + budget | true | explicit reasoning budget |
|
|
319
378
|
|
|
320
379
|
The explicit budget may tighten but MUST NOT exceed the resolved reasoning
|
|
321
380
|
reserve. Template calls normally use `reasoning_format: "auto"` for a separate
|
|
@@ -342,7 +401,7 @@ intent. It cannot override:
|
|
|
342
401
|
- data-capture settings;
|
|
343
402
|
- tool, modality, or multi-choice behavior;
|
|
344
403
|
- the consumer-owned output envelope;
|
|
345
|
-
-
|
|
404
|
+
- cache affinity identity or cache-write policy.
|
|
346
405
|
|
|
347
406
|
Generic AI SDK calls accept only settings represented by the SDK's portable
|
|
348
407
|
surface. Compatible endpoints may carry additional sampling keys after reserved
|
|
@@ -366,16 +425,30 @@ normal value. Retry exhaustion is preserved as `attempts` and
|
|
|
366
425
|
`retryExhausted`, and the resulting Problem is not marked retryable after the
|
|
367
426
|
provider has consumed its automatic retry budget.
|
|
368
427
|
|
|
369
|
-
The
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
The
|
|
373
|
-
|
|
374
|
-
|
|
428
|
+
§provider-connectivity The provider adapter owns one attempt scheduler around
|
|
429
|
+
the complete generation exchange; SDK-internal retries are disabled.
|
|
430
|
+
`PLURNK_PROVIDERS_RETRY_ATTEMPTS=N` permits at most `N + 1` physical requests.
|
|
431
|
+
The layers are independent and a configured value of zero disables only that
|
|
432
|
+
deadline:
|
|
433
|
+
|
|
434
|
+
| Layer | Operator knob | Boundary | Expiry |
|
|
435
|
+
| --- | --- | --- | --- |
|
|
436
|
+
| Operation | `PLURNK_PROVIDERS_OPERATION_TIMEOUT` | Complete logical call, including every attempt and retry delay. | Final `deadline_exceeded` Problem at 504 with `timeoutPhase=operation`; never retried. |
|
|
437
|
+
| Attempt | `PLURNK_PROVIDERS_FETCH_TIMEOUT` | One physical generation request, including response consumption. | Retryable `network_failure` with `timeoutPhase=attempt`. |
|
|
438
|
+
| First content | `PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT` | Response-stream start through first semantic model content; metadata, empty deltas, and transport activity do not satisfy it. | Retryable `network_failure` with `timeoutPhase=first_content`. |
|
|
439
|
+
| Stream idle | `PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT` | Silence between semantic content chunks after content begins. | Retryable `network_failure` with `timeoutPhase=stream_idle`. |
|
|
440
|
+
|
|
441
|
+
Caller cancellation spans the operation and preserves the caller's reason.
|
|
442
|
+
Inner deadline failures consume the ordinary retry budget; retry exhaustion
|
|
443
|
+
adds `attempts` and `retryExhausted`, retains the inner `timeoutPhase` and
|
|
444
|
+
`timeoutMs`, and is final. Every scheduler iteration opens and settles exactly
|
|
445
|
+
one ordered {§provider-request-accounting} record, including response-less
|
|
446
|
+
network failures and timed-out attempts.
|
|
375
447
|
|
|
376
448
|
HTTP 408, 409, 429, and ordinary 5xx responses are retryable unless the endpoint
|
|
377
|
-
explicitly says otherwise
|
|
378
|
-
|
|
449
|
+
explicitly says otherwise through `X-Should-Retry`. That header is authoritative;
|
|
450
|
+
without an explicit directive, endpoint control responses 520–527 are final so
|
|
451
|
+
a router can prevent multiplicative retries behind its own policy.
|
|
379
452
|
|
|
380
453
|
### §provider-interrupted-attempt Provider-declared interruption
|
|
381
454
|
|
|
@@ -386,10 +459,10 @@ completed exchange.
|
|
|
386
459
|
| Concern | Contract |
|
|
387
460
|
| -------------------------- | ---------------------------------------------------------------------------------------------------------------------------- |
|
|
388
461
|
| Normalized finish reason | Exact `insufficient_system_resource` becomes `resource_interrupted`; the raw value remains in `assistantRaw`. |
|
|
389
|
-
| `generate` outcome | Throw `ProviderError(kind="resource_interrupted")` at local status 503 with the normalized attempt on `error.attempt`.
|
|
462
|
+
| `generate` outcome | Throw `ProviderError(kind="resource_interrupted")` at local status 503 with the normalized attempt on `error.attempt` and the same accounting on `error.accounting`. |
|
|
390
463
|
| Partial response | Preserve content, reasoning, usage, model, metadata, optional raw body, and other evidence without admitting it as success. |
|
|
391
464
|
| Automatic replay | None; the Problem has `retryable: false`, and AI SDK retry scheduling has already completed at the successful transport. |
|
|
392
|
-
| Capacity-pool overflow | None
|
|
465
|
+
| Capacity-pool overflow | None under the existing routing policy; when other overflow-eligible failures do reach a sibling, the pool concatenates their request accounting. |
|
|
393
466
|
| Consumer admission | Persist the evidence as an unaccepted attempt; never parse it into executable work, even when its frame looks complete. |
|
|
394
467
|
|
|
395
468
|
## §10 Grammar
|
|
@@ -398,8 +471,9 @@ GBNF is a local llama-server capability, not a generic provider expectation.
|
|
|
398
471
|
The consumer chooses whether to supply a grammar. The provider never creates or
|
|
399
472
|
rewrites one.
|
|
400
473
|
|
|
401
|
-
GBNF defines the accepted
|
|
402
|
-
|
|
474
|
+
GBNF defines the accepted sampled text; it runs before response reasoning is
|
|
475
|
+
separated from regular content. A generated rail may declare a response root
|
|
476
|
+
that composes a template-provided prefix for independent evidence grading. The shipped PLURNK sentence is owned by
|
|
403
477
|
`plurnk-contracts` {§gbnf-turn-shape} and {§gbnf-reasoning-boundary}; this package does
|
|
404
478
|
not restate or rewrite it.
|
|
405
479
|
|
package/dist/AiSdkProvider.d.ts
CHANGED
|
@@ -1,17 +1,29 @@
|
|
|
1
|
-
import type {
|
|
1
|
+
import type { ChatMessage, PromptTokenMeasurement, Provider, ProviderCostNormalizer, ProviderGenerateArgs, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
2
2
|
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
3
|
+
import type { JSONValue } from "ai";
|
|
3
4
|
import type { Reasoning, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
|
|
4
5
|
import type { LanguageModel } from "ai";
|
|
5
6
|
import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
6
7
|
export type ProviderFetch = typeof globalThis.fetch;
|
|
7
8
|
export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "thinking_effort" | "template" | "anthropic";
|
|
8
9
|
export type GrammarStyle = "none" | "llamacpp";
|
|
10
|
+
export type CacheAffinity = {
|
|
11
|
+
readonly target: "header" | "body";
|
|
12
|
+
readonly name: string;
|
|
13
|
+
} | {
|
|
14
|
+
readonly target: "provider-option";
|
|
15
|
+
readonly provider: string;
|
|
16
|
+
readonly name: string;
|
|
17
|
+
};
|
|
18
|
+
export type AiSdkProviderOptions = Record<string, Record<string, JSONValue | undefined>>;
|
|
9
19
|
export type AiSdkProviderConfig = {
|
|
10
20
|
model: string;
|
|
11
21
|
url?: string;
|
|
12
22
|
languageModel?: LanguageModel;
|
|
13
23
|
attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
14
24
|
fetchTimeoutMs: number;
|
|
25
|
+
operationTimeoutMs: number;
|
|
26
|
+
firstContentTimeoutMs: number;
|
|
15
27
|
streamIdleTimeoutMs?: number;
|
|
16
28
|
headers?: Record<string, string>;
|
|
17
29
|
fetch?: ProviderFetch;
|
|
@@ -19,14 +31,13 @@ export type AiSdkProviderConfig = {
|
|
|
19
31
|
reasoningStyle?: ReasoningStyle;
|
|
20
32
|
reasoningResponseStyle?: ReasoningResponseStyle;
|
|
21
33
|
countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
kind: "authoritative";
|
|
25
|
-
}>;
|
|
26
|
-
normalizeCharge?: AuthoritativeChargeNormalizer;
|
|
34
|
+
estimateCost?: (usage: ProviderUsage | undefined) => ProviderCost;
|
|
35
|
+
normalizeCost?: ProviderCostNormalizer;
|
|
27
36
|
source?: string;
|
|
28
37
|
grammarStyle?: GrammarStyle;
|
|
29
|
-
|
|
38
|
+
cacheAffinity?: CacheAffinity;
|
|
39
|
+
systemCacheProviderOptions?: AiSdkProviderOptions;
|
|
40
|
+
reasoningResponseProviderOptions?: AiSdkProviderOptions;
|
|
30
41
|
serviceTier?: string;
|
|
31
42
|
gbnfDebug?: boolean;
|
|
32
43
|
streaming?: boolean;
|
|
@@ -69,24 +80,6 @@ export default class AiSdkProvider implements Provider {
|
|
|
69
80
|
get requiresMaxTokens(): boolean | undefined;
|
|
70
81
|
get constrainsOutput(): boolean;
|
|
71
82
|
countPromptTokens(messages: readonly ChatMessage[], signal?: AbortSignal): Promise<PromptTokenMeasurement>;
|
|
72
|
-
|
|
73
|
-
calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, {
|
|
74
|
-
kind: "authoritative";
|
|
75
|
-
}>;
|
|
76
|
-
generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }: {
|
|
77
|
-
messages: ChatMessage[];
|
|
78
|
-
workerId: string;
|
|
79
|
-
primaryWorkerId?: string;
|
|
80
|
-
signal?: AbortSignal;
|
|
81
|
-
grammar?: string;
|
|
82
|
-
maxTokens?: number;
|
|
83
|
-
attributions?: string[];
|
|
84
|
-
client?: string;
|
|
85
|
-
strikes?: number;
|
|
86
|
-
workspaceId?: string;
|
|
87
|
-
loop?: number;
|
|
88
|
-
turn?: number;
|
|
89
|
-
sampling?: Record<string, unknown>;
|
|
90
|
-
}): Promise<ProviderResponse>;
|
|
83
|
+
generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }: ProviderGenerateArgs): Promise<ProviderResponse>;
|
|
91
84
|
}
|
|
92
85
|
//# sourceMappingURL=AiSdkProvider.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"AiSdkProvider.d.ts","sourceRoot":"","sources":["../src/AiSdkProvider.ts"],"names":[],"mappings":"AAQA,OAAO,KAAK,
|
|
1
|
+
{"version":3,"file":"AiSdkProvider.d.ts","sourceRoot":"","sources":["../src/AiSdkProvider.ts"],"names":[],"mappings":"AAQA,OAAO,KAAK,EACR,WAAW,EAEX,sBAAsB,EACtB,QAAQ,EACR,sBAAsB,EAEtB,oBAAoB,EAIpB,gBAAgB,EAChB,aAAa,EAChB,MAAM,YAAY,CAAC;AACpB,OAAO,KAAK,EAAE,YAAY,EAAE,MAAM,0BAA0B,CAAC;AAC7D,OAAO,KAAK,EAAE,SAAS,EAAE,MAAM,IAAI,CAAC;AAEpC,OAAO,KAAK,EAAE,SAAS,EAAE,sBAAsB,EAAE,WAAW,EAAE,MAAM,UAAU,CAAC;AAM/E,OAAO,KAAK,EAAE,aAAa,EAAE,MAAM,IAAI,CAAC;AAOxC,OAAO,KAAK,EAAE,iBAAiB,EAAE,wBAAwB,EAAE,MAAM,qBAAqB,CAAC;AAKvF,MAAM,MAAM,aAAa,GAAG,OAAO,UAAU,CAAC,KAAK,CAAC;AAIpD,MAAM,MAAM,cAAc,GAAG,MAAM,GAAG,OAAO,GAAG,mBAAmB,GAAG,QAAQ,GAAG,iBAAiB,GAAG,iBAAiB,GAAG,UAAU,GAAG,WAAW,CAAC;AAIlJ,MAAM,MAAM,YAAY,GAAG,MAAM,GAAG,UAAU,CAAC;AAE/C,MAAM,MAAM,aAAa,GACnB;IAAE,QAAQ,CAAC,MAAM,EAAE,QAAQ,GAAG,MAAM,CAAC;IAAC,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GAC7D;IAAE,QAAQ,CAAC,MAAM,EAAE,iBAAiB,CAAC;IAAC,QAAQ,CAAC,QAAQ,EAAE,MAAM,CAAC;IAAC,QAAQ,CAAC,IAAI,EAAE,MAAM,CAAA;CAAE,CAAC;AAE/F,MAAM,MAAM,oBAAoB,GAAG,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,MAAM,EAAE,SAAS,GAAG,SAAS,CAAC,CAAC,CAAC;AAEzF,MAAM,MAAM,mBAAmB,GAAG;IAC9B,KAAK,EAAE,MAAM,CAAC;IACd,GAAG,CAAC,EAAE,MAAM,CAAC;IACb,aAAa,CAAC,EAAE,aAAa,CAAC;IAC9B,YAAY,CAAC,EAAE,CAAC,OAAO,EAAE,wBAAwB,KAAK,iBAAiB,CAAC;IACxE,cAAc,EAAE,MAAM,CAAC;IACvB,kBAAkB,EAAE,MAAM,CAAC;IAC3B,qBAAqB,EAAE,MAAM,CAAC;IAC9B,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACjC,KAAK,CAAC,EAAE,aAAa,CAAC;IACtB,aAAa,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC9B,cAAc,CAAC,EAAE,cAAc,CAAC;IAChC,sBAAsB,CAAC,EAAE,sBAAsB,CAAC;IAChD,iBAAiB,CAAC,EAAE,CAAC,QAAQ,EAAE,SAAS,WAAW,EAAE,EAAE,MAAM,CAAC,EAAE,WAAW,KAAK,sBAAsB,GAAG,OAAO,CAAC,sBAAsB,CAAC,CAAC;IACzI,YAAY,CAAC,EAAE,CAAC,KAAK,EAAE,aAAa,GAAG,SAAS,KAAK,YAAY,CAAC;IAClE,aAAa,CAAC,EAAE,sBAAsB,CAAC;IACvC,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,YAAY,CAAC,EAAE,YAAY,CAAC;IAG5B,aAAa,CAAC,EAAE,aAAa,CAAC;IAG9B,0BAA0B,CAAC,EAAE,oBAAoB,CAAC;IAGlD,gCAAgC,CAAC,EAAE,oBAAoB,CAAC;IAGxD,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,SAAS,CAAC,EAAE,OAAO,CAAC;IACpB,SAAS,CAAC,EAAE,OAAO,CAAC;IACpB,kBAAkB,CAAC,EAAE,OAAO,CAAC;IAC7B,qBAAqB,CAAC,EAAE,MAAM,CAAC;IAC/B,OAAO,CAAC,EAAE,MAAM,CAAC;IAEjB,mBAAmB,CAAC,EAAE,OAAO,CAAC;IAC9B,SAAS,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAI1B,WAAW,CAAC,EAAE,MAAM,CAAC;IAGrB,eAAe,CAAC,EAAE,MAAM,CAAC;IAKzB,WAAW,CAAC,EAAE,MAAM,CAAC;IAIrB,iBAAiB,CAAC,EAAE,OAAO,CAAC;IAM5B,SAAS,EAAE,SAAS,CAAC;IAOrB,WAAW,EAAE,MAAM,CAAC;IACpB,aAAa,EAAE,MAAM,CAAC;IAKtB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAK1B,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,WAAW,CAAC,EAAE,MAAM,CAAC;IAKrB,aAAa,EAAE,MAAM,CAAC;IAItB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAQ1B,WAAW,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC5B,OAAO,CAAC,EAAE,OAAO,CAAC;IAOlB,gBAAgB,CAAC,EAAE,WAAW,CAAC;IAC/B,iBAAiB,CAAC,EAAE,WAAW,CAAC;IAIhC,YAAY,CAAC,EAAE,OAAO,CAAC;CAC1B,CAAC;AAwFF,eAAO,MAAM,gBAAgB,WAAY,MAAM,KAAG,KAAK,GAAG,QAAQ,GAAG,MAIpE,CAAC;AA+BF,MAAM,CAAC,OAAO,OAAO,aAAc,YAAW,QAAQ;;IAgDlD,QAAQ,CAAC,YAAY,CAAC,EAAE,CAAC,OAAO,EAAE,wBAAwB,KAAK,iBAAiB,CAAC;IAMjF,QAAQ,CAAC,EAAE,CAAC,IAAI,EAAE,MAAM,KAAK,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;IAC/C,YAAY,MAAM,EAAE,mBAAmB,EA+HtC;IAED,IAAI,aAAa,IAAI,MAAM,GAAG,IAAI,CAAgC;IAQlE,IAAI,gBAAgB,IAAI,MAAM,GAAG,IAAI,CAAyD;IAC9F,IAAI,iBAAiB,IAAI,MAAM,GAAG,IAAI,CAA0D;IAChG,IAAI,KAAK,IAAI,MAAM,CAAwB;IAE3C,IAAI,WAAW,IAAI,MAAM,GAAG,SAAS,CAA8B;IAEnE,IAAI,iBAAiB,IAAI,OAAO,GAAG,SAAS,CAAoC;IAIhF,IAAI,gBAAgB,IAAI,OAAO,CAA0C;IAEnE,iBAAiB,CACnB,QAAQ,EAAE,SAAS,WAAW,EAAE,EAChC,MAAM,CAAC,EAAE,WAAW,GACrB,OAAO,CAAC,sBAAsB,CAAC,CAqDjC;IA+NK,QAAQ,CAAC,EAAE,QAAQ,EAAE,QAAQ,EAAE,eAAe,EAAE,MAAM,EAAE,OAAO,EAAE,SAAS,EAAE,YAAY,EAAE,MAAM,EAAE,OAAO,EAAE,WAAW,EAAE,IAAI,EAAE,IAAI,EAAE,QAAQ,EAAE,cAAc,EAAE,QAAQ,EAAE,EAAE,oBAAoB,GAAG,OAAO,CAAC,gBAAgB,CAAC,CA2V/N;CAEJ"}
|