@plurnk/plurnk-providers 1.3.3 → 1.3.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +20 -21
- package/SPEC.md +65 -44
- package/dist/Mock.d.ts +1 -1
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +1 -1
- package/dist/Mock.js.map +1 -1
- package/dist/OpenAICompat.d.ts +5 -4
- package/dist/OpenAICompat.d.ts.map +1 -1
- package/dist/OpenAICompat.js +38 -38
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/Pool.d.ts +1 -1
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +1 -1
- package/dist/Pool.js.map +1 -1
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +2 -0
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +2 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -2
- package/dist/index.js.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -1
- package/dist/openai.js.map +1 -1
- package/dist/openaiStream.d.ts +6 -1
- package/dist/openaiStream.d.ts.map +1 -1
- package/dist/openaiStream.js +30 -2
- package/dist/openaiStream.js.map +1 -1
- package/dist/standardProviders.d.ts +3 -3
- package/dist/standardProviders.d.ts.map +1 -1
- package/dist/standardProviders.js +36 -17
- package/dist/standardProviders.js.map +1 -1
- package/dist/types.d.ts +1 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +1 -1
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +4 -2
- package/dist/usage.js.map +1 -1
- package/package.json +10 -7
- package/src/Mock.test.ts +2 -2
- package/src/Mock.ts +1 -1
- package/src/OpenAICompat.test.ts +86 -86
- package/src/OpenAICompat.ts +46 -45
- package/src/Pool.test.ts +3 -3
- package/src/Pool.ts +1 -1
- package/src/ProviderRegistry.test.ts +30 -1
- package/src/aiSdkAdapter.spike.test.ts +242 -0
- package/src/env.test.ts +8 -0
- package/src/env.ts +2 -0
- package/src/index.ts +2 -2
- package/src/openai.ts +1 -1
- package/src/openaiStream.ts +30 -2
- package/src/standardProviders.test.ts +45 -31
- package/src/standardProviders.ts +42 -29
- package/src/types.ts +5 -5
- package/src/usage.test.ts +8 -10
- package/src/usage.ts +7 -3
package/.env.defaults
CHANGED
|
@@ -32,13 +32,15 @@ PLURNK_PROVIDERS_REASONING=adaptive
|
|
|
32
32
|
# the floor the provider manages wherever a grammar rides (greedy-under-mask loops without it).
|
|
33
33
|
PLURNK_PROVIDERS_TEMPERATURE=0.2
|
|
34
34
|
PLURNK_PROVIDERS_REPEAT_PENALTY=1.15
|
|
35
|
-
# FREQUENCY_PENALTY (#426):
|
|
36
|
-
#
|
|
37
|
-
#
|
|
38
|
-
#
|
|
39
|
-
|
|
40
|
-
#
|
|
41
|
-
|
|
35
|
+
# FREQUENCY_PENALTY (#426): optional cloud anti-degeneration tuning. API acceptance
|
|
36
|
+
# permits the field to ride but does not establish its semantic effect, and provider
|
|
37
|
+
# implementations differ. The portable floor is off; enable per alias only from
|
|
38
|
+
# provider documentation or a controlled behavioral experiment.
|
|
39
|
+
PLURNK_PROVIDERS_FREQUENCY_PENALTY=0
|
|
40
|
+
# PLURNK_PROVIDERS_FREQUENCY_PENALTY_myendpoint=0.4
|
|
41
|
+
# Fixed provider service tier. Fireworks accepts auto|default|flex|priority;
|
|
42
|
+
# normally set per alias so a paid routing choice is explicit.
|
|
43
|
+
# PLURNK_PROVIDERS_SERVICE_TIER_myfireworks=priority
|
|
42
44
|
# #567: DRY loop-breaker - llama.cpp-only (grammarStyle "none" paths skip it). MULTIPLIER=0
|
|
43
45
|
# disables DRY at the general floor: the #567 sweep showed L=2 is WORSE than DRY-off on both
|
|
44
46
|
# runaways AND corruption; L=32 is the measured plurnk-safe threshold (0 corruption fails,
|
|
@@ -53,6 +55,11 @@ PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH=32
|
|
|
53
55
|
# --- Transport budgets (§4, #18) ---
|
|
54
56
|
# Per-attempt fetch timeout (ms); the caller's abort signal spans retries.
|
|
55
57
|
PLURNK_PROVIDERS_FETCH_TIMEOUT=600000
|
|
58
|
+
# Maximum silence (ms) between streamed response-body chunks after response
|
|
59
|
+
# streaming begins. Disabled at the portable floor: slow local inference may
|
|
60
|
+
# pause legitimately. Enable per alias only from measured endpoint behavior.
|
|
61
|
+
PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT=0
|
|
62
|
+
# PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT_firefast=120000
|
|
56
63
|
# Transient-failure retries: 0 = surface the first failure; N = retries on 429/5xx/timeout
|
|
57
64
|
# with exponential backoff (Retry-After wins). RETRY_DELAY is the backoff base (ms).
|
|
58
65
|
PLURNK_PROVIDERS_RETRY_ATTEMPTS=3
|
|
@@ -64,21 +71,13 @@ PLURNK_PROVIDERS_RETRY_DELAY=2000
|
|
|
64
71
|
PLURNK_PROVIDERS_PROBE_ATTEMPTS=3
|
|
65
72
|
PLURNK_PROVIDERS_PROBE_DELAY=250
|
|
66
73
|
|
|
67
|
-
# ---
|
|
68
|
-
#
|
|
69
|
-
#
|
|
70
|
-
#
|
|
71
|
-
# (Engine §353). A GLOBAL rails default is unsafe - providers that ACCEPT a grammar quietly
|
|
72
|
-
# disable reasoning as a side effect, invisible to the constrainsOutput gate (it verifies rails
|
|
73
|
-
# are ENFORCED, not that reasoning SURVIVED). The reasoning model then blind-edits and looks
|
|
74
|
-
# like plurnk's fault on their platform (svc#396). Off degrades gracefully (free output, still
|
|
75
|
-
# verified; divergence -> grammar_unenforced telemetry, #24); rails-on-where-unsafe degrades
|
|
76
|
-
# silently. Turn ON per alias where rails coexist (local llama-server) or the reasoning
|
|
77
|
-
# tradeoff is deliberate.
|
|
74
|
+
# --- Optional local constrained sampling (GBNF, SPEC §13) ---
|
|
75
|
+
# Unset by default. Set per alias only for a local llama-server whose GBNF
|
|
76
|
+
# transport is detected or pinned. Cloud and endpoint-managed model settings do
|
|
77
|
+
# not use this knob.
|
|
78
78
|
# PLURNK_PROVIDERS_GBNF=plurnk.gbnf
|
|
79
|
-
# Debug toggle
|
|
80
|
-
#
|
|
81
|
-
# divergence surfaces as grammar_unenforced telemetry. Dev aid; leave off in production.
|
|
79
|
+
# Debug toggle: validate but withhold a configured local GBNF, then report the
|
|
80
|
+
# unconstrained output's divergence. Development aid; leave unset in production.
|
|
82
81
|
# PLURNK_PROVIDERS_GBNF_DEBUG=0
|
|
83
82
|
|
|
84
83
|
# --- Window (SPEC §11) ---
|
package/SPEC.md
CHANGED
|
@@ -36,7 +36,7 @@ interface Provider {
|
|
|
36
36
|
|
|
37
37
|
// Tokenomic primitives (synchronous, pure)
|
|
38
38
|
countTokens(text: string): number;
|
|
39
|
-
|
|
39
|
+
calculateCost(usage: ProviderUsage): number; // estimated USD
|
|
40
40
|
|
|
41
41
|
// OPTIONAL capability: exact tokenization served by the backend's own vocab
|
|
42
42
|
// (llama-server /tokenize). Probe-gated — undefined means the backend can't.
|
|
@@ -114,7 +114,7 @@ interface ProviderResponse {
|
|
|
114
114
|
model: string; // wire-reported (may differ from requested for relay providers)
|
|
115
115
|
};
|
|
116
116
|
assistantRaw: unknown; // verbatim wire response for forensics
|
|
117
|
-
meta?: Record<string, unknown>; // per-turn provider
|
|
117
|
+
meta?: Record<string, unknown>; // verbatim per-turn provider metadata; absent when empty (#23)
|
|
118
118
|
}
|
|
119
119
|
|
|
120
120
|
interface ProviderUsage {
|
|
@@ -159,7 +159,7 @@ The package root remains the Node daemon integration surface.
|
|
|
159
159
|
- `assistant.usage` is authoritative and follows the invariant above. Fill `0`s when the wire response omits a breakdown.
|
|
160
160
|
- `countTokens` is **synchronous**, returns a non-negative integer, deterministic for the same input. Without an exact tokenizer family configured it is the **chars/2 UPPER BOUND** — deliberately conservative (real agentic text measures ~2.9–3.2 chars/token on gemma/deepseek, so the former chars/4 silently UNDERcounted 20–27%; a fallback may overcount, never under) — and it is **surfaced at construction** (`process.emitWarning`, code `PLURNK_TOKENIZER_HEURISTIC`), never silent. Exact counting is the tokenizer seam's job (mimetypes family), fed by `tokenize()` where available.
|
|
161
161
|
- `tokenize?` is an **optional async capability**: token ids in the model's real vocabulary, served by the backend itself (llama-server's native root `/tokenize`, surfaced when the §11 probe fingerprints a llama-server and `detectLlamaServer` isn't false). `tokenize === undefined` is the honest "backend can't" signal. Exact-counting consumers prefer it over any client-side tokenizer data — the local model's own vocab needs no bundled `tokenizer.json` at all.
|
|
162
|
-
- `
|
|
162
|
+
- `calculateCost` is **pure**, returns USD non-negative integer. Returns `0` for siblings with no known rates (local Ollama, generic OpenAI-compat shims).
|
|
163
163
|
- `contextWindow` resolves to `null` when a PROBING provider (openai/llama-server) can't determine the window (consumer treats null as "no budget info"); a CLOUD provider with no window source FAILS HARD instead (#419, §11).
|
|
164
164
|
- `generate` rejects on signal abort — does NOT resolve with partial content.
|
|
165
165
|
- `generate` transports `grammar` verbatim when the backend supports grammar-constrained sampling, and silently ignores it otherwise (§13). The provider never chooses or modifies the grammar.
|
|
@@ -191,17 +191,36 @@ The consumer's instantiation path calls `mod.default.fromEnv(env, alias.model, o
|
|
|
191
191
|
|
|
192
192
|
## §4 Universal operator knobs
|
|
193
193
|
|
|
194
|
+
Defaults use four evidence states:
|
|
195
|
+
|
|
196
|
+
- **Measured** — a controlled behavioral experiment demonstrates the parameter's
|
|
197
|
+
effect on the relevant endpoint and model family.
|
|
198
|
+
- **Documented** — the provider's primary documentation defines the parameter and
|
|
199
|
+
its semantics.
|
|
200
|
+
- **Accepted** — a live request carrying the parameter succeeds. This proves wire
|
|
201
|
+
compatibility only; an endpoint may accept and ignore a field.
|
|
202
|
+
- **Unknown** — neither semantics nor compatibility are established.
|
|
203
|
+
|
|
204
|
+
Measured evidence outranks documented evidence when they conflict. A portable
|
|
205
|
+
nonzero tuning default requires measured or documented semantics that actually
|
|
206
|
+
generalize across its scope. Accepted evidence permits transport but never
|
|
207
|
+
justifies a magnitude. PLURNK does not maintain a provider-characterization
|
|
208
|
+
harness; upstream provider contracts and focused regression specimens own that
|
|
209
|
+
knowledge.
|
|
210
|
+
|
|
194
211
|
Each provider's `fromEnv` reads these:
|
|
195
212
|
|
|
196
213
|
- **`PLURNK_PROVIDERS_REASONING`** — REQUIRED, one of `off | adaptive | on`. **`PLURNK_PROVIDERS_REASONING_BUDGET`** is a positive integer required when reasoning is `on`. The provider maps this intent to the backend's supported controls. Backends may omit, combine, or expose reasoning differently when grammar-constrained output is enabled; the provider reports the channels it actually receives rather than synthesizing a separate reasoning channel. `max_tokens` remains an output limit, not a reasoning budget. The model-facing `PLAN` operation is part of the grammar and is independent of provider reasoning controls.
|
|
197
214
|
|
|
198
215
|
Read via `reasoningFromEnv` and **fail hard when unset**: the budget is required only when `on` and carries no floor default. Configuration lives in the operator's env over the package's `.env.defaults` floor (which declares every var and ships its default); the framework never bakes a knob default into code.
|
|
199
216
|
- **`PLURNK_PROVIDERS_FETCH_TIMEOUT`** — service-wide ms ceiling on any single outbound request (**per attempt**, not shared across retries). Each `fromEnv` reads and passes as `AbortSignal.timeout`. Per-provider override envs are NOT part of the contract.
|
|
217
|
+
- **`PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT`** — REQUIRED non-negative milliseconds from the shipped floor; the maximum silence between streamed response-body chunks after the response body becomes available. The clock resets on every byte chunk, `0` disables it, and expiry is a retryable `network_failure`. The portable default is `0`: slow local inference may legitimately pause for minutes, so a nonzero deadline is a measured per-alias deployment policy, not a universal provider assumption. It does not shorten time-to-response for a large prefill; `FETCH_TIMEOUT` remains the whole-attempt ceiling. A successful retry records its attempt, elapsed time, reason, and message in `ProviderResponse.meta.transportRetries`.
|
|
200
218
|
- **`PLURNK_PROVIDERS_RETRY_ATTEMPTS`** — REQUIRED non-negative integer (read via `parseRequiredInt`). The transient-failure retry budget: **`0`** surfaces the first failure; **`N`** retries up to `N` times on a *transient* classification only (`rate_limit` / `network_failure` — 429, 5xx, timeout, connection reset), plus **`grammar_invalid`** (a 422 output reject a fresh sample may satisfy, #548). Terminal kinds (`unauthorized`, `quota_exceeded`, `invalid_response`, `model_refused`) are never retried. Backoff is exponential from a `2000ms` base (`base * 2^(attempt-1)`), unless the server sent a `Retry-After` (which wins). The caller's `signal` aborts both the in-flight request and the backoff sleep. Lives in the shared `OpenAICompatProvider` so every provider inherits it uniformly; rides on the existing `classifyProviderError` (#18).
|
|
201
219
|
- **`PLURNK_PROVIDERS_CONTEXT_WINDOW`** -- optional positive-integer override (alias-scopable) for the model's context window. Resolution (#419): this env var -> endpoint `n_ctx` probe (probing specs only) -> `@plurnk/plurnk-models` catalog -> then a PROBING provider degrades to `null`, a CLOUD provider (no probe) FAILS HARD (an uncataloged, unpinned cloud model is a config error, not a guessable window). See §11.
|
|
202
220
|
- **`PLURNK_PROVIDERS_REASONING_RESERVE` / `PLURNK_PROVIDERS_COMPLETION_RESERVE`** (#507, owner-ruled) -- the generation-envelope reserves, REQUIRED (floor ships `10%` / `25%`). A percentage derives from the DETECTED window (llama-server n_ctx, the plurnk.ai router, the catalog) so every advertising endpoint gets sane defaults with ZERO operator tuning; an absolute token count wins outright (per-alias-scopable -- the measured-envelope override). These MIGRATED from core's `PLURNK_SERVICE_{CONTEXT_WINDOW,REASONING,COMPLETION}` (provider quantities wearing a service prefix; core keeps only its own packing-safety margin). Surfaced as `Provider.reasoningReserve`/`completionReserve`.
|
|
203
|
-
- **`PLURNK_PROVIDERS_TEMPERATURE` / `PLURNK_PROVIDERS_REPEAT_PENALTY` / `PLURNK_PROVIDERS_FREQUENCY_PENALTY`** -- REQUIRED sampling
|
|
204
|
-
- **`
|
|
221
|
+
- **`PLURNK_PROVIDERS_TEMPERATURE` / `PLURNK_PROVIDERS_REPEAT_PENALTY` / `PLURNK_PROVIDERS_FREQUENCY_PENALTY`** -- REQUIRED sampling controls (read via `parseRequiredFloat`, values from the `.env.defaults` floor). `REPEAT_PENALTY` (canonical `1.15`) is the measured llama.cpp multiplier. `FREQUENCY_PENALTY` is the optional cloud analogue, but its portable floor is `0`: endpoint acceptance proves only that the field may ride, not that one magnitude has a portable semantic effect. Enable it per alias from provider documentation or controlled behavioral evidence. The controls are keyed per backend (§13).
|
|
222
|
+
- **`PLURNK_PROVIDERS_SERVICE_TIER`** -- optional fixed request tier, alias-scopable. Fireworks accepts its published `auto | default | flex | priority` vocabulary and owns those values' routing semantics; when configured, the value is validated at construction and wins over per-call sampling on every request. Unset delegates to the provider default. Other standard providers reject the knob rather than silently ignoring a paid routing choice.
|
|
223
|
+
- **`PLURNK_PROVIDERS_DRY_MULTIPLIER` / `_DRY_BASE` / `_DRY_ALLOWED_LENGTH`** -- the llama.cpp DRY loop-breaker (#567), customer-overridable per alias. The generic floor is deliberately **off** (`MULTIPLIER=0`): the turboderp/Gemma sweep proved the community-standard `0.8`/`1.75`/`2` settings worse than off for both runaway emissions and exact-identifier corruption. That same sweep measured `0.8`/`1.75`/`32` as a safe alias-specific deployment setting (zero corruption, about 6% runaways versus 19% off), so `.env.defaults` carries `BASE=1.75` and `ALLOWED_LENGTH=32` as inert override companions without pretending one model's multiplier is universal. DRY penalizes repeated sequences with a penalty escalating in run length -- the tool for a plan-restart loop a single-token `repeat_penalty` over a short window cannot see. **`PLURNK_PROVIDERS_REPEAT_LAST_N`** (optional) widens the older `repeat_penalty` window past the box's 64. These knobs are sent **only on the detected `llamacpp` path**; cloud providers parse but never emit them.
|
|
205
224
|
|
|
206
225
|
## §5 Alias cascade resolution
|
|
207
226
|
|
|
@@ -253,7 +272,7 @@ The framework is **contract-only**: it does not depend on provider plugins. The
|
|
|
253
272
|
- `signal` is wired to the worker's AbortController.
|
|
254
273
|
- `generate` is single-call per turn. No parallel calls on the same instance.
|
|
255
274
|
- `assistantRaw` is opaque to the consumer (forensics-only).
|
|
256
|
-
- `meta` is the per-turn provider→client metadata bag: the backend's
|
|
275
|
+
- `meta` is the per-turn provider→client metadata bag: the backend's non-standard top-level response fields pass through verbatim. Monetary metadata carries an explicit decimal-string `amount` and `currency`; the provider never guesses or converts its unit. Absent when the backend reported no extras. The consumer (service) merges `meta` into its Turn metadata and filters what reaches the client; it reads `meta`, never mines `assistantRaw` (#23).
|
|
257
276
|
- `countTokens` is cheap by contract; consumer calls frequently.
|
|
258
277
|
|
|
259
278
|
## §7 Provider → engine guarantees
|
|
@@ -265,7 +284,7 @@ The framework is **contract-only**: it does not depend on provider plugins. The
|
|
|
265
284
|
- **Atomic.** One `generate` call resolves with one complete `ProviderResponse`. No streaming partial resolves (v0).
|
|
266
285
|
- **Honors `signal`.** Aborted calls reject; resources free; no orphaned connections.
|
|
267
286
|
- **Single model.** One provider instance speaks to one model.
|
|
268
|
-
- **Synchronous `countTokens`, pure `
|
|
287
|
+
- **Synchronous `countTokens`, pure `calculateCost`.** No I/O, no async, no state beyond cached tokenizer artifacts.
|
|
269
288
|
|
|
270
289
|
## §8 Forbidden
|
|
271
290
|
|
|
@@ -306,9 +325,9 @@ A sibling package satisfies the contract when:
|
|
|
306
325
|
|
|
307
326
|
1. Default export is a class with `static fromEnv(env, model, options?)` factory.
|
|
308
327
|
2. Instance exposes `contextWindow: number | null` and `model: string` (non-empty).
|
|
309
|
-
3. Instance exposes `countTokens(text): number` and `
|
|
328
|
+
3. Instance exposes `countTokens(text): number` and `calculateCost(usage): number`.
|
|
310
329
|
4. `countTokens("")` returns `0`; `countTokens("…")` returns a non-negative integer.
|
|
311
|
-
5. `
|
|
330
|
+
5. `calculateCost({prompt:0,completion:0,reasoning:0,cached:0,total:0})` returns `0` (or non-negative USD for non-free models).
|
|
312
331
|
6. Identity getters return stable values across reads.
|
|
313
332
|
7. `generate` resolves with a valid `ProviderResponse` shape.
|
|
314
333
|
8. `generate` invoked with a pre-aborted `signal` rejects without making a wire call.
|
|
@@ -335,8 +354,8 @@ The framework ships the transport spine every OpenAI-compatible provider had bee
|
|
|
335
354
|
contextWindow, // number | null
|
|
336
355
|
reasoning, reasoningStyle, // {mode,budget} intent + style: "none"|"think"|"include_reasoning"|"effort"|"effort_explicit"|"template"|"anthropic"
|
|
337
356
|
temperature, repeatPenalty, frequencyPenalty, // sampling + anti-degeneration floor; frequency_penalty guards the plain cloud path (#426)
|
|
338
|
-
countTokens,
|
|
339
|
-
grammarStyle, // "none" | "llamacpp"
|
|
357
|
+
countTokens, calculateCost, // strategies; default heuristic / free
|
|
358
|
+
grammarStyle, // "none" | "llamacpp" — optional local GBNF transport (§13)
|
|
340
359
|
gbnfDebug, // PLURNK_PROVIDERS_GBNF_DEBUG: validate a grammar locally + throw on invalid, but DON'T send it (§13); default false
|
|
341
360
|
streaming, // SSE transport; default true (false → one non-streamed JSON)
|
|
342
361
|
supportsSlotPinning, slotCount, // INTERNAL slot-affinity wiring (run→id_slot); never consumer-facing
|
|
@@ -348,17 +367,17 @@ The framework ships the transport spine every OpenAI-compatible provider had bee
|
|
|
348
367
|
The `openai` standard provider sets `grammarStyle: "llamacpp"`, `supportsSlotPinning`, and `slotCount` from the same llama-server fingerprint (`/v1/models` `meta` block + `/props`). The worker→slot mapping lives inside `OpenAICompatProvider`: sticky per `workerId`, round-robin across new runs, LRU-bounded.
|
|
349
368
|
|
|
350
369
|
- **`chatCompletionStream` / `chatCompletion` / `OpenAiHttpError` / `StreamResponse`** — the shared HTTP client (`chatCompletionStream` for SSE, `chatCompletion` for the non-streamed JSON the `streaming: false` path uses). One shared copy.
|
|
351
|
-
- **`normalizeUsage(raw, reasoningText?, contentText?)` / `
|
|
370
|
+
- **`normalizeUsage(raw, reasoningText?, contentText?)` / `calculateCostUsd(usage, rates)`** — usage normalization to the §2 invariant (handles all three reasoning-reporting conventions; the optional text args feed the Fireworks re-split, #425) and the single cost formula. Rates use the Models.dev convention of USD per million tokens; billable output is `completion + reasoning`. `OpenAICompatProvider` applies `normalizeUsage` automatically; provider instances expose `calculateCost(usage)`.
|
|
352
371
|
- **`parseRequiredInt` / `parseOptionalInt` / `requireEnv`** — env helpers; each takes a provider `label` for error prefixing.
|
|
353
372
|
- **`effortFromBudget(budget)`** — the shared reasoning-budget → `low|medium|high` breakpoints.
|
|
354
373
|
|
|
355
374
|
A **bespoke sibling** therefore reduces to a thin class whose `fromEnv` probes whatever it needs (model catalog, pricing, context window), builds the config, and returns `new OpenAICompatProvider(config)`. A **standard provider** (§5 tier 1) needs no sibling at all — it's a frozen entry in `STANDARD_PROVIDERS` describing its key var, base-URL var, reasoning style, and tokenizer; `standardProviderFromEnv(name, env, model)` (async — returns `Promise<Provider | null>`) does the rest. The endpoint's **canonical URL ships as a floored default** in `.env.defaults` (set-if-unset, overridable in the operator's env or per-alias); it is read from the base-URL var (or a `baseUrlFromEnv` deriver) with **no in-code default**, the value living in the shipped floor, never baked into the table. Only the API **key** is required operator config (a secret with no default; fail-hard when unset).
|
|
356
375
|
|
|
357
|
-
The `plurnk` entry alone sets **`firstPartyMetadata: true`** — it forwards the consumer's per-turn `generate()` `attributions` (which installed plugin packages dispatched) and `client` (the originating frontend, e.g. `plurnk.nvim/1.4.0`) as `Plurnk-Attribution` / `Plurnk-Client` headers, and the opaque `workerId` as `Plurnk-Worker-Id` (#26, wire-name completed #511), and the lineage root `primaryWorkerId` as `Plurnk-Worker-Primary` (#522 — root-vs-descendant classification and worker-tree grouping key, always stamped when supplied). The gate lives on the provider, not the call site, so these first-party signals are
|
|
376
|
+
The `plurnk` entry alone sets **`firstPartyMetadata: true`** — it forwards the consumer's per-turn `generate()` `attributions` (which installed plugin packages dispatched) and `client` (the originating frontend, e.g. `plurnk.nvim/1.4.0`) as `Plurnk-Attribution` / `Plurnk-Client` headers, and the opaque `workerId` as `Plurnk-Worker-Id` (#26, wire-name completed #511), and the lineage root `primaryWorkerId` as `Plurnk-Worker-Primary` (#522 — root-vs-descendant classification and worker-tree grouping key, always stamped when supplied). The gate lives on the provider, not the call site, so these first-party signals are structurally incapable of reaching a third-party backend. Empty values emit no header. Endpoint response metadata follows the same general pass-through contract as every provider.
|
|
358
377
|
|
|
359
378
|
**Prompt-cache affinity (`promptCacheKey`, #518).** Standard providers send the OpenAI-standard `prompt_cache_key` set to the `workerId` on every request, **default-ON** (opt out per spec). Serverless backends prompt-cache automatically but the cache is REPLICA-LOCAL; without an affinity key a worker's turns scatter across replicas and the stable prefix never hits (verified live: `cached_tokens` 0 without the key). `workerId` -- already the slot-affinity identity -- is exactly the opaque per-conversation key the cache wants, so a worker's turns pin to one replica and its stable prefix caches. Managed + reserved from caller `sampling`. It's the OpenAI-standard field and broadly accepted -- verified live on fireworks, together, deepinfra, xai, openrouter, and llama-server (6/6, all accept it, every serverless one caches). A backend that caches by a DIFFERENT mechanism opts out (`anthropic`: cache_control breakpoints); a backend later found to strict-reject the field opts out the same way.
|
|
360
379
|
|
|
361
|
-
A spec may carry a **`modelPrefix`** — a constant model-id segment the backend requires but the operator's alias shouldn't repeat. `fireworks` sets `"accounts/fireworks/models/"`, so `PLURNK_MODEL_fast=fireworks/deepseek-v4-pro` carries only the distinctive tail; `standardProviderFromEnv` prepends it idempotently
|
|
380
|
+
A spec may carry a **`modelPrefix`** — a constant model-id segment the backend requires but the operator's alias shouldn't repeat. `fireworks` sets `"accounts/fireworks/models/"`, so `PLURNK_MODEL_fast=fireworks/deepseek-v4-pro` carries only the distinctive tail; `standardProviderFromEnv` prepends it idempotently to form the wire id, which is **also** the catalog key (models.dev keys fireworks-ai on the full id). A fully qualified Fireworks resource under `accounts/fireworks/` is preserved verbatim, including `routers/` and `deployments/`. Specs without a `modelPrefix` use the model string verbatim.
|
|
362
381
|
|
|
363
382
|
`contextWindow` for a standard provider resolves (#419): `PLURNK_PROVIDERS_CONTEXT_WINDOW` -> endpoint `n_ctx` (for `probeNctx`-flagged specs like `openai`, queried from `GET /v1/models`: llama-server reports its loaded window at `data[].meta.n_ctx`, vLLM top-level; cloud endpoints don't) -> the `@plurnk/plurnk-models` catalog -> **then the hybrid: a PROBING provider degrades to `null`, a CLOUD provider (no probe) FAILS HARD** (uncataloged + unpinned = config error, the #417 kimi case, not a guessed window). The same probe fingerprints llama-server (the `meta` block) to enable grammar transport (§13), and reads the row's `id` as `servedModel` (#37) — the real served name behind a local alias — so it runs even when the env var pins the window. The probe is best-effort: any failure resolves to `null` context / no grammar capability (a legitimate "unknown"), never throws. For a PROBING provider, an underivable window (env, probe, and catalog ALL missed) is surfaced once via a **`PLURNK_CONTEXT_UNKNOWN`** warning naming the model and the remediation (`PLURNK_PROVIDERS_CONTEXT_WINDOW`, alias-scopable) -- null stays legitimate but never silent (a CLOUD provider throws here instead, above). Operator-facing warnings (`PLURNK_TOKENIZER_HEURISTIC`, `PLURNK_PROBE_FAILED`, `PLURNK_GRAMMAR_UNVERIFIABLE`, `PLURNK_CONTEXT_UNKNOWN`, `PLURNK_FINISH_REASON_UNKNOWN`) are deduplicated **once per process per (code, message)** (#40) — repeat constructions don't re-fire them, but a *different* provider/model's first surfacing is never suppressed.
|
|
364
383
|
|
|
@@ -379,37 +398,39 @@ The `TelemetryEvent` shape is mirrored **locally** (`./telemetry.ts`), structura
|
|
|
379
398
|
|
|
380
399
|
## §13 Grammar-constrained sampling (GBNF)
|
|
381
400
|
|
|
382
|
-
|
|
401
|
+
GBNF is an optional aid for local llama.cpp hobbyists, not the PLURNK language
|
|
402
|
+
contract and not a baseline cloud capability. The canonical language is parsed
|
|
403
|
+
by `@plurnk/plurnk-grammar`'s ANTLR grammar. That package also ships a generated
|
|
404
|
+
`plurnk.gbnf` whose language is a tested subset of the canonical grammar.
|
|
383
405
|
|
|
384
406
|
- **plurnk-grammar** owns the artifact (canonical-form GBNF, `L(GBNF) ⊂ L(ANTLR)` invariant, tests).
|
|
385
|
-
- **This layer**
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
Zero grammar dependency (§11) is preserved: the GBNF string arrives per call; this package never imports the artifact.
|
|
407
|
+
- **This layer** detects or accepts an operator pin for llama-server, transports
|
|
408
|
+
the caller's GBNF verbatim as its top-level `grammar` field, and reports
|
|
409
|
+
conformance. Cloud providers never receive a grammar-related field.
|
|
410
|
+
- **The consumer** decides whether to configure a local constraint and which
|
|
411
|
+
artifact to send. Endpoint-managed constraints are endpoint settings, not a
|
|
412
|
+
provider capability inferred from their absence here.
|
|
413
|
+
|
|
414
|
+
`grammarStyle` is `"none"` or `"llamacpp"`. A llama-server fingerprint or
|
|
415
|
+
`PLURNK_PROVIDERS_LLAMA_SERVER=1` selects `"llamacpp"`; all other providers
|
|
416
|
+
remain `"none"`. `constrainsOutput` is true only for the former.
|
|
417
|
+
|
|
418
|
+
When a grammar is transported, the provider independently validates returned
|
|
419
|
+
content with `@plurnk/gbnf`. A non-accept verdict attaches
|
|
420
|
+
`grammar_unenforced` telemetry without discarding the completed response.
|
|
421
|
+
`meta.railsAttached` and `meta.railsVerdict` record the observed local
|
|
422
|
+
transport and verdict. If the validator cannot parse the supplied grammar, the
|
|
423
|
+
provider emits `PLURNK_GRAMMAR_UNVERIFIABLE`.
|
|
424
|
+
|
|
425
|
+
`PLURNK_PROVIDERS_GBNF_DEBUG` validates and withholds an otherwise transportable
|
|
426
|
+
local grammar, then reports how the unconstrained output diverges. It is a
|
|
427
|
+
development diagnostic, not a cloud compatibility mode.
|
|
428
|
+
|
|
429
|
+
Hard constraints can amplify repetition and do not replace the normal output
|
|
430
|
+
envelope. The llama.cpp path therefore carries its configured
|
|
431
|
+
`repeat_penalty`; ordinary cloud requests use the standard configured
|
|
432
|
+
`frequency_penalty`. The GBNF string still arrives per call, so this package
|
|
433
|
+
does not depend on the PLURNK grammar artifact.
|
|
413
434
|
|
|
414
435
|
## §14 Data capture — logprobs + verbatim body (#36)
|
|
415
436
|
|
|
@@ -425,7 +446,7 @@ Two OPT-IN knobs surface the full signal of a paid turn for downstream IQ scorin
|
|
|
425
446
|
|
|
426
447
|
`Pool` fronts **N interchangeable backends as one `Provider`** - capacity scaling, not model blend. It ships the MECHANISM (round-robin across workers, sticky within a worker, overflow to a healthy sibling); the blend/escalation DECISION (which SKU, when to switch) stays the **consumer's**, one level up, by choosing WHICH pool to call. `new Pool(backends: Provider[])` - the consumer resolves the backends (per-alias `instantiateProvider`) and composes them.
|
|
427
448
|
|
|
428
|
-
**Interchangeable, or it throws.** Construction fails on mixed `model`: a heterogeneous "pool" is the consumer's per-turn selection, not this primitive. The surface is the honest aggregate - `contextWindow` is the **safe floor** (min; `null` if any backend's window is unknown, so the consumer never improvises a cap, #421) with its matching reserves; `constrainsOutput` is claimed only if EVERY backend does; `requiresMaxTokens` if ANY does; `servedModel` the common id (else absent); `countTokens`/`tokenize`/`
|
|
449
|
+
**Interchangeable, or it throws.** Construction fails on mixed `model`: a heterogeneous "pool" is the consumer's per-turn selection, not this primitive. The surface is the honest aggregate - `contextWindow` is the **safe floor** (min; `null` if any backend's window is unknown, so the consumer never improvises a cap, #421) with its matching reserves; `constrainsOutput` is claimed only if EVERY backend does; `requiresMaxTokens` if ANY does; `servedModel` the common id (else absent); `countTokens`/`tokenize`/`calculateCost` delegate.
|
|
429
450
|
|
|
430
451
|
**Affinity is the point (§11, one level up).** A worker's turns stick to one backend so its stable prompt prefix keeps hitting the same KV cache; scattering a worker across backends shreds the prefix cache (#531). `worker -> backend` is the `worker -> slot` slot-affinity pattern (#11) across a fleet: round-robin assigns a NEW worker, a returning worker re-pins, the map is LRU-bounded (`N*8`).
|
|
431
452
|
|
package/dist/Mock.d.ts
CHANGED
|
@@ -34,7 +34,7 @@ export default class Mock implements Provider {
|
|
|
34
34
|
get completionReserve(): number | null;
|
|
35
35
|
get model(): string;
|
|
36
36
|
countTokens(text: string): number;
|
|
37
|
-
|
|
37
|
+
calculateCost(_usage: ProviderUsage): number;
|
|
38
38
|
generate({ signal }: {
|
|
39
39
|
messages: ChatMessage[];
|
|
40
40
|
workerId?: string;
|
package/dist/Mock.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"Mock.d.ts","sourceRoot":"","sources":["../src/Mock.ts"],"names":[],"mappings":"AAOA,OAAO,KAAK,EAAE,WAAW,EAAE,YAAY,EAAE,QAAQ,EAAE,iBAAiB,EAAE,aAAa,EAAE,MAAM,YAAY,CAAC;AAGxG,MAAM,MAAM,aAAa,GAAG;IACxB,OAAO,EAAE,MAAM,CAAC;IAChB,SAAS,EAAE,MAAM,GAAG,IAAI,CAAC;IAEzB,KAAK,CAAC,EAAE,OAAO,CAAC,aAAa,CAAC,CAAC;IAC/B,YAAY,CAAC,EAAE,YAAY,CAAC;IAC5B,KAAK,CAAC,EAAE,MAAM,CAAC;IAGf,kBAAkB,CAAC,EAAE,aAAa,CAAC;QAAE,EAAE,EAAE,MAAM,GAAG,IAAI,CAAC;QAAC,OAAO,EAAE,MAAM,CAAC;QAAC,SAAS,EAAE,aAAa,CAAC;YAAE,IAAI,EAAE,MAAM,CAAC;YAAC,MAAM,EAAE,MAAM,GAAG,IAAI,CAAA;SAAE,CAAC,CAAA;KAAE,CAAC,CAAC;IAK9I,GAAG,CAAC,EAAE,OAAO,EAAE,CAAC;CACnB,CAAC;AAEF,MAAM,MAAM,YAAY,GAAG;IACvB,SAAS,EAAE,aAAa,CAAC;IACzB,YAAY,CAAC,EAAE,OAAO,CAAC;CAC1B,CAAC;AAGF,MAAM,MAAM,qBAAqB,GAAG,iBAAiB,GAAG;IAAE,GAAG,CAAC,EAAE,OAAO,EAAE,CAAA;CAAE,CAAC;AAE5E,QAAA,MAAM,aAAa,EAAE,aAA+E,CAAC;AAErG,MAAM,CAAC,OAAO,OAAO,IAAK,YAAW,QAAQ;;IAazC,YAAY,EAAE,aAAa,EAAE,SAAS,EAAE,EAAE;QAAE,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;QAAC,SAAS,EAAE,YAAY,EAAE,CAAA;KAAE,EAMpG;IAED,IAAI,aAAa,IAAI,MAAM,GAAG,IAAI,CAAgC;IAClE,IAAI,gBAAgB,IAAI,MAAM,GAAG,IAAI,CAAmC;IACxE,IAAI,iBAAiB,IAAI,MAAM,GAAG,IAAI,CAAoC;IAC1E,IAAI,KAAK,IAAI,MAAM,CAAmB;IAItC,WAAW,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAEhC;IAGD,
|
|
1
|
+
{"version":3,"file":"Mock.d.ts","sourceRoot":"","sources":["../src/Mock.ts"],"names":[],"mappings":"AAOA,OAAO,KAAK,EAAE,WAAW,EAAE,YAAY,EAAE,QAAQ,EAAE,iBAAiB,EAAE,aAAa,EAAE,MAAM,YAAY,CAAC;AAGxG,MAAM,MAAM,aAAa,GAAG;IACxB,OAAO,EAAE,MAAM,CAAC;IAChB,SAAS,EAAE,MAAM,GAAG,IAAI,CAAC;IAEzB,KAAK,CAAC,EAAE,OAAO,CAAC,aAAa,CAAC,CAAC;IAC/B,YAAY,CAAC,EAAE,YAAY,CAAC;IAC5B,KAAK,CAAC,EAAE,MAAM,CAAC;IAGf,kBAAkB,CAAC,EAAE,aAAa,CAAC;QAAE,EAAE,EAAE,MAAM,GAAG,IAAI,CAAC;QAAC,OAAO,EAAE,MAAM,CAAC;QAAC,SAAS,EAAE,aAAa,CAAC;YAAE,IAAI,EAAE,MAAM,CAAC;YAAC,MAAM,EAAE,MAAM,GAAG,IAAI,CAAA;SAAE,CAAC,CAAA;KAAE,CAAC,CAAC;IAK9I,GAAG,CAAC,EAAE,OAAO,EAAE,CAAC;CACnB,CAAC;AAEF,MAAM,MAAM,YAAY,GAAG;IACvB,SAAS,EAAE,aAAa,CAAC;IACzB,YAAY,CAAC,EAAE,OAAO,CAAC;CAC1B,CAAC;AAGF,MAAM,MAAM,qBAAqB,GAAG,iBAAiB,GAAG;IAAE,GAAG,CAAC,EAAE,OAAO,EAAE,CAAA;CAAE,CAAC;AAE5E,QAAA,MAAM,aAAa,EAAE,aAA+E,CAAC;AAErG,MAAM,CAAC,OAAO,OAAO,IAAK,YAAW,QAAQ;;IAazC,YAAY,EAAE,aAAa,EAAE,SAAS,EAAE,EAAE;QAAE,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;QAAC,SAAS,EAAE,YAAY,EAAE,CAAA;KAAE,EAMpG;IAED,IAAI,aAAa,IAAI,MAAM,GAAG,IAAI,CAAgC;IAClE,IAAI,gBAAgB,IAAI,MAAM,GAAG,IAAI,CAAmC;IACxE,IAAI,iBAAiB,IAAI,MAAM,GAAG,IAAI,CAAoC;IAC1E,IAAI,KAAK,IAAI,MAAM,CAAmB;IAItC,WAAW,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAEhC;IAGD,aAAa,CAAC,MAAM,EAAE,aAAa,GAAG,MAAM,CAAc;IAEpD,QAAQ,CAAC,EAAE,MAAM,EAAE,EAAE;QAAE,QAAQ,EAAE,WAAW,EAAE,CAAC;QAAC,QAAQ,CAAC,EAAE,MAAM,CAAC;QAAC,MAAM,CAAC,EAAE,WAAW,CAAA;KAAE,GAAG,OAAO,CAAC;QAAE,SAAS,EAAE,qBAAqB,CAAC;QAAC,YAAY,EAAE,OAAO,CAAA;KAAE,CAAC,CAiBrK;IAED,IAAI,SAAS,IAAI,MAAM,CAA+B;CACzD;AAED,OAAO,EAAE,aAAa,IAAI,gBAAgB,EAAE,CAAC"}
|
package/dist/Mock.js
CHANGED
|
@@ -35,7 +35,7 @@ export default class Mock {
|
|
|
35
35
|
return text.length === 0 ? 0 : Math.ceil(text.length / 2);
|
|
36
36
|
}
|
|
37
37
|
// Mock is free.
|
|
38
|
-
|
|
38
|
+
calculateCost(_usage) { return 0; }
|
|
39
39
|
async generate({ signal }) {
|
|
40
40
|
// Honor abort before consuming the queue — an aborted call makes no
|
|
41
41
|
// "wire call" and must not exhaust a queued response (SPEC §10.8).
|
package/dist/Mock.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"Mock.js","sourceRoot":"","sources":["../src/Mock.ts"],"names":[],"mappings":"AAAA,2DAA2D;AAC3D,EAAE;AACF,wEAAwE;AACxE,wEAAwE;AACxE,wEAAwE;AACxE,2CAA2C;AAG3C,OAAO,EAAE,sBAAsB,EAAE,MAAM,UAAU,CAAC;AA2BlD,MAAM,aAAa,GAAkB,EAAE,MAAM,EAAE,CAAC,EAAE,UAAU,EAAE,CAAC,EAAE,SAAS,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,KAAK,EAAE,CAAC,EAAE,CAAC;AAErG,MAAM,CAAC,OAAO,OAAO,IAAI;IACrB,cAAc,CAAgB;IAC9B,iBAAiB,CAAgB;IACjC,kBAAkB,CAAgB;IAClC,MAAM,CAAiB;IAEvB,8EAA8E;IAC9E,6EAA6E;IAC7E,2EAA2E;IAC3E,wEAAwE;IACxE,sEAAsE;IACtE,gFAAgF;IAChF,gFAAgF;IAChF,YAAY,EAAE,aAAa,EAAE,SAAS,EAA+D;QACjG,IAAI,CAAC,cAAc,GAAG,aAAa,CAAC;QACpC,MAAM,GAAG,GAAG,sBAAsB,CAAC,OAAO,CAAC,GAAG,EAAE,aAAa,CAAC,CAAC;QAC/D,IAAI,CAAC,iBAAiB,GAAG,GAAG,CAAC,gBAAgB,CAAC;QAC9C,IAAI,CAAC,kBAAkB,GAAG,GAAG,CAAC,iBAAiB,CAAC;QAChD,IAAI,CAAC,MAAM,GAAG,CAAC,GAAG,SAAS,CAAC,CAAC;IACjC,CAAC;IAED,IAAI,aAAa,KAAoB,OAAO,IAAI,CAAC,cAAc,CAAC,CAAC,CAAC;IAClE,IAAI,gBAAgB,KAAoB,OAAO,IAAI,CAAC,iBAAiB,CAAC,CAAC,CAAC;IACxE,IAAI,iBAAiB,KAAoB,OAAO,IAAI,CAAC,kBAAkB,CAAC,CAAC,CAAC;IAC1E,IAAI,KAAK,KAAa,OAAO,MAAM,CAAC,CAAC,CAAC;IAEtC,qEAAqE;IACrE,0EAA0E;IAC1E,WAAW,CAAC,IAAY;QACpB,OAAO,IAAI,CAAC,MAAM,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IAC9D,CAAC;IAED,gBAAgB;IAChB,
|
|
1
|
+
{"version":3,"file":"Mock.js","sourceRoot":"","sources":["../src/Mock.ts"],"names":[],"mappings":"AAAA,2DAA2D;AAC3D,EAAE;AACF,wEAAwE;AACxE,wEAAwE;AACxE,wEAAwE;AACxE,2CAA2C;AAG3C,OAAO,EAAE,sBAAsB,EAAE,MAAM,UAAU,CAAC;AA2BlD,MAAM,aAAa,GAAkB,EAAE,MAAM,EAAE,CAAC,EAAE,UAAU,EAAE,CAAC,EAAE,SAAS,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,KAAK,EAAE,CAAC,EAAE,CAAC;AAErG,MAAM,CAAC,OAAO,OAAO,IAAI;IACrB,cAAc,CAAgB;IAC9B,iBAAiB,CAAgB;IACjC,kBAAkB,CAAgB;IAClC,MAAM,CAAiB;IAEvB,8EAA8E;IAC9E,6EAA6E;IAC7E,2EAA2E;IAC3E,wEAAwE;IACxE,sEAAsE;IACtE,gFAAgF;IAChF,gFAAgF;IAChF,YAAY,EAAE,aAAa,EAAE,SAAS,EAA+D;QACjG,IAAI,CAAC,cAAc,GAAG,aAAa,CAAC;QACpC,MAAM,GAAG,GAAG,sBAAsB,CAAC,OAAO,CAAC,GAAG,EAAE,aAAa,CAAC,CAAC;QAC/D,IAAI,CAAC,iBAAiB,GAAG,GAAG,CAAC,gBAAgB,CAAC;QAC9C,IAAI,CAAC,kBAAkB,GAAG,GAAG,CAAC,iBAAiB,CAAC;QAChD,IAAI,CAAC,MAAM,GAAG,CAAC,GAAG,SAAS,CAAC,CAAC;IACjC,CAAC;IAED,IAAI,aAAa,KAAoB,OAAO,IAAI,CAAC,cAAc,CAAC,CAAC,CAAC;IAClE,IAAI,gBAAgB,KAAoB,OAAO,IAAI,CAAC,iBAAiB,CAAC,CAAC,CAAC;IACxE,IAAI,iBAAiB,KAAoB,OAAO,IAAI,CAAC,kBAAkB,CAAC,CAAC,CAAC;IAC1E,IAAI,KAAK,KAAa,OAAO,MAAM,CAAC,CAAC,CAAC;IAEtC,qEAAqE;IACrE,0EAA0E;IAC1E,WAAW,CAAC,IAAY;QACpB,OAAO,IAAI,CAAC,MAAM,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IAC9D,CAAC;IAED,gBAAgB;IAChB,aAAa,CAAC,MAAqB,IAAY,OAAO,CAAC,CAAC,CAAC,CAAC;IAE1D,KAAK,CAAC,QAAQ,CAAC,EAAE,MAAM,EAAwE;QAC3F,oEAAoE;QACpE,mEAAmE;QACnE,MAAM,EAAE,cAAc,EAAE,CAAC;QACzB,MAAM,IAAI,GAAG,IAAI,CAAC,MAAM,CAAC,KAAK,EAAE,CAAC;QACjC,IAAI,IAAI,KAAK,SAAS;YAAE,MAAM,IAAI,KAAK,CAAC,mDAAmD,CAAC,CAAC;QAC7F,MAAM,CAAC,GAAG,IAAI,CAAC,SAAS,CAAC;QACzB,MAAM,SAAS,GAA0B;YACrC,OAAO,EAAE,CAAC,CAAC,OAAO;YAClB,SAAS,EAAE,CAAC,CAAC,SAAS;YACtB,KAAK,EAAE,EAAE,GAAG,aAAa,EAAE,GAAG,CAAC,CAAC,KAAK,EAAE;YACvC,GAAG,CAAC,CAAC,CAAC,kBAAkB,KAAK,SAAS,CAAC,CAAC,CAAC,EAAE,kBAAkB,EAAE,CAAC,CAAC,kBAAkB,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,yDAAyD;YACtJ,YAAY,EAAE,CAAC,CAAC,YAAY,IAAI,MAAM;YACtC,KAAK,EAAE,CAAC,CAAC,KAAK,IAAI,MAAM;YACxB,GAAG,CAAC,CAAC,CAAC,GAAG,KAAK,SAAS,CAAC,CAAC,CAAC,EAAE,GAAG,EAAE,CAAC,CAAC,GAAG,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC;SACjD,CAAC;QACF,OAAO,EAAE,SAAS,EAAE,YAAY,EAAE,IAAI,CAAC,YAAY,IAAI,IAAI,EAAE,CAAC;IAClE,CAAC;IAED,IAAI,SAAS,KAAa,OAAO,IAAI,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC,CAAC;CACzD;AAED,OAAO,EAAE,aAAa,IAAI,gBAAgB,EAAE,CAAC"}
|
package/dist/OpenAICompat.d.ts
CHANGED
|
@@ -2,26 +2,27 @@ import type { ChatMessage, Provider, ProviderResponse, ProviderUsage } from "./t
|
|
|
2
2
|
import type { Reasoning, ReserveSpec } from "./env.ts";
|
|
3
3
|
import { type ProviderFetch } from "./openaiStream.ts";
|
|
4
4
|
export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "template" | "anthropic";
|
|
5
|
-
export type GrammarStyle = "none" | "llamacpp"
|
|
5
|
+
export type GrammarStyle = "none" | "llamacpp";
|
|
6
6
|
export type OpenAICompatConfig = {
|
|
7
7
|
model: string;
|
|
8
8
|
url: string;
|
|
9
9
|
fetchTimeoutMs: number;
|
|
10
|
+
streamIdleTimeoutMs?: number;
|
|
10
11
|
headers?: Record<string, string>;
|
|
11
12
|
fetch?: ProviderFetch;
|
|
12
13
|
contextWindow?: number | null;
|
|
13
14
|
reasoningStyle?: ReasoningStyle;
|
|
14
15
|
countTokens?: (text: string) => number;
|
|
15
|
-
|
|
16
|
+
calculateCost?: (usage: ProviderUsage) => number;
|
|
16
17
|
source?: string;
|
|
17
18
|
grammarStyle?: GrammarStyle;
|
|
18
19
|
promptCacheKey?: boolean;
|
|
20
|
+
serviceTier?: string;
|
|
19
21
|
gbnfDebug?: boolean;
|
|
20
22
|
streaming?: boolean;
|
|
21
23
|
firstPartyMetadata?: boolean;
|
|
22
24
|
apiKeyRejectedMessage?: string;
|
|
23
25
|
eosText?: string;
|
|
24
|
-
balanceMetaKey?: string;
|
|
25
26
|
supportsSlotPinning?: boolean;
|
|
26
27
|
slotCount?: number | null;
|
|
27
28
|
tokenizeUrl?: string;
|
|
@@ -56,7 +57,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
56
57
|
get requiresMaxTokens(): boolean | undefined;
|
|
57
58
|
get constrainsOutput(): boolean;
|
|
58
59
|
countTokens(text: string): number;
|
|
59
|
-
|
|
60
|
+
calculateCost(usage: ProviderUsage): number;
|
|
60
61
|
generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }: {
|
|
61
62
|
messages: ChatMessage[];
|
|
62
63
|
workerId: string;
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"OpenAICompat.d.ts","sourceRoot":"","sources":["../src/OpenAICompat.ts"],"names":[],"mappings":"AAUA,OAAO,KAAK,EAAE,WAAW,EAAgB,QAAQ,EAAE,gBAAgB,EAAE,aAAa,EAAE,MAAM,YAAY,CAAC;AACvG,OAAO,KAAK,EAAE,SAAS,EAAE,WAAW,EAAE,MAAM,UAAU,CAAC;AACvD,OAAO,EAAuE,KAAK,aAAa,EAAuB,MAAM,mBAAmB,CAAC;AAiBjJ,MAAM,MAAM,cAAc,GAAG,MAAM,GAAG,OAAO,GAAG,mBAAmB,GAAG,QAAQ,GAAG,iBAAiB,GAAG,UAAU,GAAG,WAAW,CAAC;
|
|
1
|
+
{"version":3,"file":"OpenAICompat.d.ts","sourceRoot":"","sources":["../src/OpenAICompat.ts"],"names":[],"mappings":"AAUA,OAAO,KAAK,EAAE,WAAW,EAAgB,QAAQ,EAAE,gBAAgB,EAAE,aAAa,EAAE,MAAM,YAAY,CAAC;AACvG,OAAO,KAAK,EAAE,SAAS,EAAE,WAAW,EAAE,MAAM,UAAU,CAAC;AACvD,OAAO,EAAuE,KAAK,aAAa,EAAuB,MAAM,mBAAmB,CAAC;AAiBjJ,MAAM,MAAM,cAAc,GAAG,MAAM,GAAG,OAAO,GAAG,mBAAmB,GAAG,QAAQ,GAAG,iBAAiB,GAAG,UAAU,GAAG,WAAW,CAAC;AAI9H,MAAM,MAAM,YAAY,GAAG,MAAM,GAAG,UAAU,CAAC;AAE/C,MAAM,MAAM,kBAAkB,GAAG;IAC7B,KAAK,EAAE,MAAM,CAAC;IACd,GAAG,EAAE,MAAM,CAAC;IACZ,cAAc,EAAE,MAAM,CAAC;IACvB,mBAAmB,CAAC,EAAE,MAAM,CAAC;IAC7B,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACjC,KAAK,CAAC,EAAE,aAAa,CAAC;IACtB,aAAa,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC9B,cAAc,CAAC,EAAE,cAAc,CAAC;IAChC,WAAW,CAAC,EAAE,CAAC,IAAI,EAAE,MAAM,KAAK,MAAM,CAAC;IACvC,aAAa,CAAC,EAAE,CAAC,KAAK,EAAE,aAAa,KAAK,MAAM,CAAC;IACjD,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,YAAY,CAAC,EAAE,YAAY,CAAC;IAM5B,cAAc,CAAC,EAAE,OAAO,CAAC;IAGzB,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,SAAS,CAAC,EAAE,OAAO,CAAC;IACpB,SAAS,CAAC,EAAE,OAAO,CAAC;IACpB,kBAAkB,CAAC,EAAE,OAAO,CAAC;IAC7B,qBAAqB,CAAC,EAAE,MAAM,CAAC;IAC/B,OAAO,CAAC,EAAE,MAAM,CAAC;IAEjB,mBAAmB,CAAC,EAAE,OAAO,CAAC;IAC9B,SAAS,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAI1B,WAAW,CAAC,EAAE,MAAM,CAAC;IAKrB,WAAW,CAAC,EAAE,MAAM,CAAC;IAIrB,iBAAiB,CAAC,EAAE,OAAO,CAAC;IAM5B,SAAS,EAAE,SAAS,CAAC;IAQrB,WAAW,EAAE,MAAM,CAAC;IACpB,aAAa,EAAE,MAAM,CAAC;IAKtB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAO1B,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,YAAY,EAAE,MAAM,CAAC;IAIrB,aAAa,EAAE,MAAM,CAAC;IAQtB,WAAW,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC5B,OAAO,CAAC,EAAE,OAAO,CAAC;IAOlB,gBAAgB,CAAC,EAAE,WAAW,CAAC;IAC/B,iBAAiB,CAAC,EAAE,WAAW,CAAC;IAIhC,YAAY,CAAC,EAAE,OAAO,CAAC;CAC1B,CAAC;AA0DF,eAAO,MAAM,gBAAgB,WAAY,MAAM,KAAG,KAAK,GAAG,QAAQ,GAAG,MAIpE,CAAC;AAwCF,MAAM,CAAC,OAAO,OAAO,oBAAqB,YAAW,QAAQ;;IA6CzD,QAAQ,CAAC,EAAE,CAAC,IAAI,EAAE,MAAM,KAAK,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;IAE/C,YAAY,MAAM,EAAE,kBAAkB,EA+DrC;IAED,IAAI,aAAa,IAAI,MAAM,GAAG,IAAI,CAAgC;IAQlE,IAAI,gBAAgB,IAAI,MAAM,GAAG,IAAI,CAAyD;IAC9F,IAAI,iBAAiB,IAAI,MAAM,GAAG,IAAI,CAA0D;IAChG,IAAI,KAAK,IAAI,MAAM,CAAwB;IAE3C,IAAI,WAAW,IAAI,MAAM,GAAG,SAAS,CAA8B;IAEnE,IAAI,iBAAiB,IAAI,OAAO,GAAG,SAAS,CAAoC;IAIhF,IAAI,gBAAgB,IAAI,OAAO,CAA0C;IAEzE,WAAW,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAAoC;IACrE,aAAa,CAAC,KAAK,EAAE,aAAa,GAAG,MAAM,CAAuC;IAuN5E,QAAQ,CAAC,EAAE,QAAQ,EAAE,QAAQ,EAAE,eAAe,EAAE,MAAM,EAAE,OAAO,EAAE,SAAS,EAAE,YAAY,EAAE,MAAM,EAAE,OAAO,EAAE,WAAW,EAAE,IAAI,EAAE,IAAI,EAAE,QAAQ,EAAE,EAAE;QAAE,QAAQ,EAAE,WAAW,EAAE,CAAC;QAAC,QAAQ,EAAE,MAAM,CAAC;QAAC,eAAe,CAAC,EAAE,MAAM,CAAC;QAAC,MAAM,CAAC,EAAE,WAAW,CAAC;QAAC,OAAO,CAAC,EAAE,MAAM,CAAC;QAAC,SAAS,CAAC,EAAE,MAAM,CAAC;QAAC,YAAY,CAAC,EAAE,MAAM,EAAE,CAAC;QAAC,MAAM,CAAC,EAAE,MAAM,CAAC;QAAC,OAAO,CAAC,EAAE,MAAM,CAAC;QAAC,WAAW,CAAC,EAAE,MAAM,CAAC;QAAC,IAAI,CAAC,EAAE,MAAM,CAAC;QAAC,IAAI,CAAC,EAAE,MAAM,CAAC;QAAC,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAA;KAAE,GAAG,OAAO,CAAC,gBAAgB,CAAC,CA8Kxc;CACJ"}
|