@plurnk/plurnk-providers 1.0.5 → 1.0.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +20 -4
- package/README.md +3 -3
- package/SPEC.md +55 -37
- package/dist/Mock.d.ts +10 -4
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +20 -4
- package/dist/Mock.js.map +1 -1
- package/dist/OpenAICompat.d.ts +13 -7
- package/dist/OpenAICompat.d.ts.map +1 -1
- package/dist/OpenAICompat.js +150 -54
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/env.d.ts +16 -1
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +66 -10
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +3 -3
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -1
- package/dist/index.js.map +1 -1
- package/dist/openaiStream.d.ts +5 -0
- package/dist/openaiStream.d.ts.map +1 -1
- package/dist/openaiStream.js +31 -1
- package/dist/openaiStream.js.map +1 -1
- package/dist/standardProviders.d.ts +1 -0
- package/dist/standardProviders.d.ts.map +1 -1
- package/dist/standardProviders.js +30 -19
- package/dist/standardProviders.js.map +1 -1
- package/dist/types.d.ts +9 -3
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +1 -1
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +16 -2
- package/dist/usage.js.map +1 -1
- package/package.json +6 -6
package/.env.defaults
CHANGED
|
@@ -32,6 +32,13 @@ PLURNK_PROVIDERS_REASONING=adaptive
|
|
|
32
32
|
# the floor the provider manages wherever a grammar rides (greedy-under-mask loops without it).
|
|
33
33
|
PLURNK_PROVIDERS_TEMPERATURE=0.2
|
|
34
34
|
PLURNK_PROVIDERS_REPEAT_PENALTY=1.15
|
|
35
|
+
# FREQUENCY_PENALTY (#426): the anti-degeneration guard on the CLOUD path. repeat_penalty is a
|
|
36
|
+
# llama.cpp/vLLM MULTIPLIER the plain cloud path (grammarStyle "none") can't use; frequency_penalty
|
|
37
|
+
# is OpenAI-standard, so every OpenAI-compat backend accepts it (verified live: together, deepinfra,
|
|
38
|
+
# fireworks). Without it a cloud alias ran the sampler bare and looped to the token cap. 0 = off.
|
|
39
|
+
# NOTE: 0.4 is a STARTING value (acceptance-verified, not yet effectiveness-benched) - raise it if a
|
|
40
|
+
# firefast/deepseek re-bench still degenerates.
|
|
41
|
+
PLURNK_PROVIDERS_FREQUENCY_PENALTY=0.4
|
|
35
42
|
|
|
36
43
|
# --- Transport budgets (§4, #18) ---
|
|
37
44
|
# Per-attempt fetch timeout (ms); the caller's abort signal spans retries.
|
|
@@ -68,7 +75,7 @@ PLURNK_PROVIDERS_PROBE_DELAY=250
|
|
|
68
75
|
# Unset = derive: env -> live endpoint probe (n_ctx) -> models.dev catalog -> null (surfaced
|
|
69
76
|
# once via PLURNK_CONTEXT_UNKNOWN naming the model). Set to pin the window deliberately -
|
|
70
77
|
# per-alias (_<alias>) for a box whose model the catalog doesn't know.
|
|
71
|
-
#
|
|
78
|
+
# PLURNK_PROVIDERS_CONTEXT_WINDOW=200000
|
|
72
79
|
|
|
73
80
|
# --- llama-server detection pin (#34) ---
|
|
74
81
|
# Unset = auto-detect from the /v1/models fingerprint. 1 = pin llama-server capabilities
|
|
@@ -78,10 +85,10 @@ PLURNK_PROVIDERS_PROBE_DELAY=250
|
|
|
78
85
|
|
|
79
86
|
# --- Data capture (#36, SPEC §14) - OFF by default; the flag IS the isolation ---
|
|
80
87
|
# Enable ONLY on a dataset-scraping alias (append _<alias>); serving turns then request and
|
|
81
|
-
# carry nothing.
|
|
82
|
-
# raw model logprob, the sampling-invariant confidence). RAWBODY: truthy -> verbatim wire
|
|
88
|
+
# carry nothing. TOP_LOGPROBS: non-negative int = the OpenAI top_logprobs count (surfaced on
|
|
89
|
+
# assistant.logprobs; raw model logprob, the sampling-invariant confidence). RAWBODY: truthy -> verbatim wire
|
|
83
90
|
# body on response.rawBody (keeps sampling_logprob/token_id/bytes the digest drops).
|
|
84
|
-
#
|
|
91
|
+
# PLURNK_PROVIDERS_TOP_LOGPROBS_myscraper=3
|
|
85
92
|
# PLURNK_PROVIDERS_RAWBODY_myscraper=1
|
|
86
93
|
|
|
87
94
|
# --- Alias cascade (SPEC §5) - declare aliases, then pick the active one ---
|
|
@@ -140,3 +147,12 @@ ANTHROPIC_BASE_URL=https://api.anthropic.com/v1
|
|
|
140
147
|
# plurnk hosted model - PLURNK_API_KEY is an OPTIONAL bearer (sent only when set).
|
|
141
148
|
PLURNK_BASE_URL=https://plurnk.ai/v1
|
|
142
149
|
# PLURNK_BASE_URL=http://plurnksnr2kihuukt6v22ko72r34dxeatbsfhgow3hvnlw6btanxphad.onion/v1 # Tor
|
|
150
|
+
|
|
151
|
+
# --- Generation envelope (#507, owner-ruled) - sane defaults from the DETECTED window ---
|
|
152
|
+
# When a backend advertises its context window (llama-server n_ctx, the plurnk.ai router,
|
|
153
|
+
# a cataloged cloud model), the reserves derive from it automatically - ZERO operator
|
|
154
|
+
# tuning. Each accepts a percentage of the window ("10%") or an absolute token count
|
|
155
|
+
# ("4096"; absolutes win outright, alias-scopable for measured envelopes). The prompt
|
|
156
|
+
# budget is window - reasoning - completion - the consumer's own safety margin.
|
|
157
|
+
PLURNK_PROVIDERS_REASONING_RESERVE=10%
|
|
158
|
+
PLURNK_PROVIDERS_COMPLETION_RESERVE=25%
|
package/README.md
CHANGED
|
@@ -27,13 +27,13 @@ One package is **one** provider identity — the `<name>` segment of `PLURNK_MOD
|
|
|
27
27
|
The framework calls `YourClass.fromEnv(env, model, options?)` (sync or async) and expects a `Provider`. `options.baseUrl` is the per-alias endpoint override (`PLURNK_BASEURL_<alias>`) — honor it if you're a self-hosted provider so two aliases can reach two boxes; ignore it otherwise. Two ways in:
|
|
28
28
|
|
|
29
29
|
- **OpenAI-compatible backends** (the common case): `fromEnv` reads its env (base URL, key), probes whatever it needs (catalog, context window, pricing), and returns **`new OpenAICompatProvider(config)`**. You write a `fromEnv` and a config object — the transport spine (SSE, usage normalization, `finishReason`, grammar transport, slot affinity) is inherited. See `OpenAICompatConfig` / SPEC §11.
|
|
30
|
-
- **Non-OpenAI backends**: `implements Provider` directly — `generate`, `
|
|
30
|
+
- **Non-OpenAI backends**: `implements Provider` directly — `generate`, `contextWindow`, `model`, `countTokens(text)`, `costFor(usage)`.
|
|
31
31
|
|
|
32
32
|
`fromEnv` **MUST fail fast with a named error** when required env is missing — name the var the operator must set. (Why a factory, not a base-class constructor like execs/mimes: a provider often async-probes at construction — SPEC §3.)
|
|
33
33
|
|
|
34
34
|
### 3. What `generate` receives — and returns
|
|
35
35
|
|
|
36
|
-
`generate({ messages,
|
|
36
|
+
`generate({ messages, workerId, signal?, grammar?, maxTokens?, attributions?, client? }) → Promise<ProviderResponse>`. Return **raw** wire output: `content` unparsed (the consumer parses the plurnk DSL — never parse it yourself), `reasoning` is the wire-reported CoT only. Honor `signal`. The provider never mutates `messages` or injects turns. `grammar` (GBNF) is attached only by backends that support it; all others ignore it (SPEC §13). When a grammar *is* transported, the provider verifies the backend actually enforced it — non-conforming output rejects with a `grammar_unenforced` `ProviderError` (a conformance check via `@plurnk/gbnf`, never a plurnk-DSL parse). In **GBNF-filter mode** (`PLURNK_GBNF_DEBUG`, grammar withheld) the same non-conformance is **non-fatal**: `generate` returns the model's bytes and attaches a `grammar_unenforced` event to `ProviderResponse.telemetry` with the divergence `position`, so the consumer can drive self-correction instead of losing the turn (#24). `attributions`/`client` are per-turn first-party metadata, forwarded as `Plurnk-*` headers **only** by a provider configured with `firstPartyMetadata` (the plurnk endpoint); every other provider drops them, so they can never reach a third-party backend (SPEC §11). The response carries a `meta?` bag — the backend's extra top-level fields passed through, plus validated known keys (e.g. `meta.balancePico`, pico-USD, normalized only from the plurnk endpoint) — for the service's per-turn metadata (#23).
|
|
37
37
|
|
|
38
38
|
## Discovery & trust
|
|
39
39
|
|
|
@@ -51,7 +51,7 @@ First-party daughters install flat via [`@plurnk/plurnk-providers-all`](https://
|
|
|
51
51
|
- `parseAliasesFromEnv`, `resolveActiveAlias`, `instantiateProvider`, `loadActiveProvider`, `discover`, `resetDiscoveryCache` — alias-cascade resolution + two-tier provider instantiation (`resetDiscoveryCache` clears the memoized tier-2 scan; for tests). Tier 1 is the standard table; tier 2 is a scope-agnostic `node_modules` scan for `plurnk.kind:"provider"` packages — first-party daughters (flat via `@plurnk/plurnk-providers-all`) and third-party providers under any scope, gated by the host `PLURNK_PLUGINS_TRUSTED_ONLY` allowlist. The framework is contract-only (SPEC §5).
|
|
52
52
|
- `OpenAICompatProvider` (+ `OpenAICompatConfig`, `ReasoningStyle`, `GrammarStyle`, `effortFromBudget`) — shared OpenAI-compatible transport spine; siblings extend it (SPEC §11). Transports a GBNF grammar via `grammarStyle` — `llamacpp` (top-level `grammar` field) or `response_format` (Fireworks); `none` drops it — and verifies the backend enforced it against `@plurnk/gbnf`, rejecting non-conforming output as `grammar_unenforced` (SPEC §13).
|
|
53
53
|
- `chatCompletionStream`, `chatCompletion`, `OpenAiHttpError`, `StreamResponse` — the shared SSE client (`chatCompletion` is the non-streaming variant).
|
|
54
|
-
- `parseRequiredInt`, `parseOptionalInt`, `requireEnv`, `
|
|
54
|
+
- `parseRequiredInt`, `parseOptionalInt`, `parseRequiredFloat`, `requireEnv`, `reasoningFromEnv` — env helpers (SPEC §4; all required-with-named-errors, no in-code defaults).
|
|
55
55
|
- `normalizeUsage`, `computeCost` (+ `RawUsage`, `TokenRates`) — usage normalization to the §2 invariant and the single cost formula (SPEC §11).
|
|
56
56
|
- `ProviderError`, `classifyProviderError`, `toProviderError`, `providerSource` (+ `TelemetryEvent`, `ProviderTelemetryKind`) — the TelemetryEvent envelope for transport failures (SPEC §12).
|
|
57
57
|
- `tokenizerFor`, `tokenizerByPublisher`, `parseTokenizerFamily` (+ `TokenizerFamily`, `CountTokens`) — synchronous tokenizer strategies.
|
package/SPEC.md
CHANGED
|
@@ -23,7 +23,7 @@ Collision on `(kind: "provider", name)` at discovery: fail-hard.
|
|
|
23
23
|
```ts
|
|
24
24
|
interface Provider {
|
|
25
25
|
// Identity (immutable across lifetime)
|
|
26
|
-
readonly
|
|
26
|
+
readonly contextWindow: number | null; // context tokens, null if unresolved.
|
|
27
27
|
// PER SLOT under llama-server --parallel N
|
|
28
28
|
// (the server splits --ctx-size and reports
|
|
29
29
|
// the divided value; verified live).
|
|
@@ -53,8 +53,15 @@ interface Provider {
|
|
|
53
53
|
// backends never set it; undefined = no claim.
|
|
54
54
|
readonly constrainsOutput?: boolean;
|
|
55
55
|
readonly requiresMaxTokens?: boolean;
|
|
56
|
-
|
|
57
|
-
//
|
|
56
|
+
// #507 (owner-ruled): generation-envelope reserves derived from the DETECTED
|
|
57
|
+
// window (floor percentages; absolute per-alias pins win outright). null =
|
|
58
|
+
// underivable -> the consumer's no-cap path. The consumer's prompt budget is
|
|
59
|
+
// window - reasoningReserve - completionReserve - its OWN safety margin;
|
|
60
|
+
// its generation cap is the two reserves pooled. Absent = no claim.
|
|
61
|
+
readonly reasoningReserve?: number | null;
|
|
62
|
+
readonly completionReserve?: number | null;
|
|
63
|
+
|
|
64
|
+
// Transport. `workerId` is REQUIRED: the opaque, stable identity of the
|
|
58
65
|
// consumer's work stream — providers may key backend affinity on it and
|
|
59
66
|
// never interpret it. `grammar` is an optional GBNF string for
|
|
60
67
|
// grammar-constrained sampling (§13) — attached verbatim by capable
|
|
@@ -67,25 +74,31 @@ interface Provider {
|
|
|
67
74
|
// `sampling` is an optional bag of standard OpenAI-compat sampling params
|
|
68
75
|
// (temperature, top_p, top_k, min_p, penalties, stop, seed, …) merged into the
|
|
69
76
|
// body UNDER the managed fields — model/messages/grammar/reasoning/max_tokens/
|
|
70
|
-
// slot always win, and
|
|
71
|
-
// grammar, id_slot)
|
|
72
|
-
//
|
|
77
|
+
// slot always win, and reserved keys are stripped (#477): transport/protocol
|
|
78
|
+
// (stream, response_format, grammar, id_slot, logprobs), paradigm breakers
|
|
79
|
+
// (n, the tools/functions family, modalities/audio, prediction), and the
|
|
80
|
+
// token caps (max_tokens/max_completion_tokens -- the envelope is the managed
|
|
81
|
+
// maxTokens, never bypassable). It carries sampling intent + platform knobs
|
|
82
|
+
// only (§8). For a PROXY consumer forwarding its own
|
|
73
83
|
// caller's sampling knobs (the plurnk endpoint fronting gemma/Fireworks); a
|
|
74
84
|
// direct consumer leaves it unset.
|
|
75
|
-
// `strikes` is the
|
|
85
|
+
// `strikes` is the worker's CURRENT rail-strike streak at time-of-generate
|
|
76
86
|
// (0 = clean, distinct from absent = unreported; contract plurnk-service#313).
|
|
77
87
|
// Forwarded as `Plurnk-Strikes` ONLY under the firstPartyMetadata gate,
|
|
78
88
|
// dropped everywhere else. Headers only — never placed in the packet.
|
|
79
|
-
// `
|
|
80
|
-
// `Plurnk-
|
|
89
|
+
// `workspaceId`/`loop`/`turn` (#404): the turn coordinate, stamped as
|
|
90
|
+
// `Plurnk-Workspace-Id`/`Plurnk-Loop`/`Plurnk-Turn` under the SAME gate.
|
|
81
91
|
// 1-based; absent/0 emits no header. Headers only, never the packet.
|
|
82
|
-
generate(args: { messages: ChatMessage[];
|
|
92
|
+
generate(args: { messages: ChatMessage[]; workerId: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse>;
|
|
83
93
|
}
|
|
84
94
|
|
|
85
95
|
interface ProviderResponse {
|
|
86
96
|
assistant: {
|
|
87
97
|
content: string; // raw model emission; consumer parses
|
|
88
|
-
reasoning: string | null; // wire-reported
|
|
98
|
+
reasoning: string | null; // wire-reported reasoning content; null if absent
|
|
99
|
+
// sealed relay reasoning (#482): { data, format } blobs verbatim, never
|
|
100
|
+
// decoded; absent when none. agui projects REASONING_ENCRYPTED_VALUE.
|
|
101
|
+
reasoningEncrypted?: Array<{ data: string; format: string | null }>;
|
|
89
102
|
usage: ProviderUsage; // { prompt, completion, reasoning, cached, total }
|
|
90
103
|
finishReason: "stop" | "length" | "tool_calls" | "content_filter" | null;
|
|
91
104
|
model: string; // wire-reported (may differ from requested for relay providers)
|
|
@@ -97,13 +110,13 @@ interface ProviderResponse {
|
|
|
97
110
|
interface ProviderUsage {
|
|
98
111
|
prompt: number; // input tokens (cached ones included)
|
|
99
112
|
completion: number; // visible output tokens, EXCLUDING reasoning
|
|
100
|
-
reasoning: number; // reasoning
|
|
113
|
+
reasoning: number; // reasoning tokens, billed as output
|
|
101
114
|
cached: number; // subset of prompt served from cache
|
|
102
115
|
total: number; // prompt + completion + reasoning
|
|
103
116
|
}
|
|
104
117
|
```
|
|
105
118
|
|
|
106
|
-
Usage invariant: `total = prompt + completion + reasoning`; `cached ⊆ prompt`; `completion` excludes reasoning; **billable output = `completion + reasoning`**. Providers report reasoning
|
|
119
|
+
Usage invariant: `total = prompt + completion + reasoning`; `cached ⊆ prompt`; `completion` excludes reasoning; **billable output = `completion + reasoning`**. Providers report reasoning THREE ways: inside `completion_tokens` (OpenAI, via `completion_tokens_details.reasoning_tokens`), only as the `total - prompt - completion` gap (Gemini), or folded into `completion_tokens` with NO itemization while shipping the reasoning as TEXT (Fireworks, #425). The framework's `normalizeUsage` (§11) collapses all three to this invariant -- re-splitting the Fireworks case by the emitted text lengths, sum-preserving so cost is unchanged -- so siblings on `OpenAICompatProvider` get it for free.
|
|
107
120
|
|
|
108
121
|
### Promises
|
|
109
122
|
|
|
@@ -114,11 +127,11 @@ Usage invariant: `total = prompt + completion + reasoning`; `cached ⊆ prompt`;
|
|
|
114
127
|
- `countTokens` is **synchronous**, returns a non-negative integer, deterministic for the same input. Without an exact tokenizer family configured it is the **chars/2 UPPER BOUND** — deliberately conservative (real agentic text measures ~2.9–3.2 chars/token on gemma/deepseek, so the former chars/4 silently UNDERcounted 20–27%; a fallback may overcount, never under) — and it is **surfaced at construction** (`process.emitWarning`, code `PLURNK_TOKENIZER_HEURISTIC`), never silent. Exact counting is the tokenizer seam's job (mimetypes family), fed by `tokenize()` where available.
|
|
115
128
|
- `tokenize?` is an **optional async capability**: token ids in the model's real vocabulary, served by the backend itself (llama-server's native root `/tokenize`, surfaced when the §11 probe fingerprints a llama-server and `detectLlamaServer` isn't false). `tokenize === undefined` is the honest "backend can't" signal. Exact-counting consumers prefer it over any client-side tokenizer data — the local model's own vocab needs no bundled `tokenizer.json` at all.
|
|
116
129
|
- `costFor` is **pure**, returns pico-USD non-negative integer. Returns `0` for siblings with no known rates (local Ollama, generic OpenAI-compat shims).
|
|
117
|
-
- `
|
|
130
|
+
- `contextWindow` resolves to `null` when a PROBING provider (openai/llama-server) can't determine the window (consumer treats null as "no budget info"); a CLOUD provider with no window source FAILS HARD instead (#419, §11).
|
|
118
131
|
- `generate` rejects on signal abort — does NOT resolve with partial content.
|
|
119
132
|
- `generate` transports `grammar` verbatim when the backend supports grammar-constrained sampling, and silently ignores it otherwise (§13). The provider never chooses or modifies the grammar.
|
|
120
133
|
- `generate` **returns for every completed exchange — bytes always present, the conformance verdict attached as an observation.** When a grammar was transported (or validated in filter mode), the returned `content` is checked against it; a non-accept verdict rides `response.telemetry` as a `grammar_unenforced` event (message + divergence `position`) and the response returns normally. The provider transports and observes; it never adjudicates — discard, retry, escalate, or feed-back is consumer policy. This is a grammar-**conformance** check against the grammar the provider already holds — *not* a plurnk-DSL parse (that stays consumer-side, below) — so it remains backend- and DSL-agnostic. `ProviderError` remains reserved for exchanges that did NOT complete (transport failure, abort, boundary violations).
|
|
121
|
-
- **Backend affinity is the provider's internal guarantee, keyed by `
|
|
134
|
+
- **Backend affinity is the provider's internal guarantee, keyed by `workerId`.** The consumer says *which run this is*, never *which backend resource serves it* — raw resource identifiers (slot integers, connections) never cross the contract in either direction. On slot-pinning backends (llama-server `--parallel N>1`), the provider keeps each worker sticky to one slot and spreads distinct runs across slots, so each concurrent run keeps its KV-cache prefix warm (un-pinned routing is the server's similarity heuristic — slot hops re-pay full prefills). Backends without affinity semantics ignore `workerId` entirely.
|
|
122
135
|
|
|
123
136
|
## §3 `fromEnv(env, model, options?)` factory
|
|
124
137
|
|
|
@@ -129,7 +142,7 @@ class OpenAI {
|
|
|
129
142
|
static fromEnv(env: NodeJS.ProcessEnv, model: string, options?: ProviderOptions): OpenAI | Promise<OpenAI> {
|
|
130
143
|
// Read provider-specific env (OPENAI_BASE_URL, OPENAI_API_KEY, ...)
|
|
131
144
|
// plus universal operator knobs (PLURNK_PROVIDERS_REASONING, PLURNK_PROVIDERS_FETCH_TIMEOUT,
|
|
132
|
-
//
|
|
145
|
+
// PLURNK_PROVIDERS_CONTEXT_WINDOW). `options.baseUrl`, when set, is the per-alias
|
|
133
146
|
// endpoint override (PLURNK_BASEURL_<alias>, §5) and wins over the env base URL.
|
|
134
147
|
return new OpenAI({ /* ... */ });
|
|
135
148
|
}
|
|
@@ -149,10 +162,12 @@ Each provider's `fromEnv` reads these:
|
|
|
149
162
|
|
|
150
163
|
- **`PLURNK_PROVIDERS_REASONING`** — REQUIRED, one of `off | adaptive | on` (#33). Pure ACTIVATION intent; **`PLURNK_PROVIDERS_REASONING_BUDGET`** (positive int, REQUIRED iff `on`) carries the magnitude separately, so a number can never silently flip wire flags. The provider maps intent to the active backend's mechanism — llama-server `chat_template_kwargs: { enable_thinking }` (always emitted; the explicit FALSE is the only working off-switch; the NUMERIC clamp is the box's `--reasoning-budget` LAUNCH flag — per-request numerics are ignored, so env budget and launch flag are the same number, changed together), Ollama `think`, relay `include_reasoning`, cloud `reasoning_effort` (tier from budget when `on`; `off`/`adaptive` omit), fireworks `reasoning_effort` **explicit** (`effort_explicit`: `off` SENDS `"none"` — omission is NOT off on reason-by-default models like DeepSeek V4, #30; `adaptive` OMITS the field — the backend default IS adaptive, and the literal `"adaptive"` is MiniMax-M3-only, 400 elsewhere, #403. Intent maps **identically under a transported grammar** (the former #32 clamp-to-`"none"` is lifted — it caused plan-less turns, service#331; the lift enables best-effort reasoning). BUT reasoning and hard-masked GBNF rails do **NOT reliably coexist** on fireworks (TUNING-EPIC **F9-final**, ~60 live tests). Fireworks' docs: `response_format` (which includes grammar mode) **disables reasoning output**; measured, reasoning reaches the usable `reasoning_content` channel only stochastically (~1/8) and otherwise collapses or leaks into `content` — best lever (prefill `<think>\n`) tops out at 7/10 with a runaway tail. So under a transported grammar, reasoning is **best-effort and unsegregated** (lands in `content`, not `reasoning_content`), and NOT tunable — a consumer wanting the reasoning parses the preplan region of `content`. Fireworks is rails-first; single-call reasoning+rails needs a llama.cpp lazy-grammar host. Envelope: `max_tokens` is a cost circuit-breaker (reasoning-on can run away), not a reasoning reserve), Anthropic `thinking: disabled | enabled+budget_tokens(budget) | omit`. Consumers state intent, never mechanism. (In-DSL `<<PLAN>>` reasoning has **no provider footprint** and no operator knob: `PLAN` is a required element of the canonical plurnk grammar the consumer attaches via `generate`'s `grammar`. The provider transports that grammar verbatim and never forces, prefills, or toggles `PLAN`.)
|
|
151
164
|
|
|
152
|
-
Read via `
|
|
165
|
+
Read via `reasoningFromEnv` and **fail hard when unset**: the budget is required only when `on` and carries no floor default. Configuration lives in the operator's env over the package's `.env.defaults` floor (which declares every var and ships its default); the framework never bakes a knob default into code.
|
|
153
166
|
- **`PLURNK_PROVIDERS_FETCH_TIMEOUT`** — service-wide ms ceiling on any single outbound request (**per attempt**, not shared across retries). Each `fromEnv` reads and passes as `AbortSignal.timeout`. Per-provider override envs are NOT part of the contract.
|
|
154
167
|
- **`PLURNK_PROVIDERS_RETRY_ATTEMPTS`** — REQUIRED non-negative integer (read via `parseRequiredInt`). The transient-failure retry budget: **`0`** surfaces the first failure; **`N`** retries up to `N` times on a *transient* classification only (`rate_limit` / `network_failure` — 429, 5xx, timeout, connection reset). Terminal kinds (`unauthorized`, `quota_exceeded`, `invalid_response`, `model_refused`) are never retried. Backoff is exponential from a `2000ms` base (`base * 2^(attempt-1)`), unless the server sent a `Retry-After` (which wins). The caller's `signal` aborts both the in-flight request and the backoff sleep. Lives in the shared `OpenAICompatProvider` so every provider inherits it uniformly; rides on the existing `classifyProviderError` (#18).
|
|
155
|
-
- **`
|
|
168
|
+
- **`PLURNK_PROVIDERS_CONTEXT_WINDOW`** -- optional positive-integer override (alias-scopable) for the model's context window. Resolution (#419): this env var -> endpoint `n_ctx` probe (probing specs only) -> `@plurnk/plurnk-models` catalog -> then a PROBING provider degrades to `null`, a CLOUD provider (no probe) FAILS HARD (an uncataloged, unpinned cloud model is a config error, not a guessable window). See §11.
|
|
169
|
+
- **`PLURNK_PROVIDERS_REASONING_RESERVE` / `PLURNK_PROVIDERS_COMPLETION_RESERVE`** (#507, owner-ruled) -- the generation-envelope reserves, REQUIRED (floor ships `10%` / `25%`). A percentage derives from the DETECTED window (llama-server n_ctx, the plurnk.ai router, the catalog) so every advertising endpoint gets sane defaults with ZERO operator tuning; an absolute token count wins outright (per-alias-scopable -- the measured-envelope override). These MIGRATED from core's `PLURNK_SERVICE_{CONTEXT_WINDOW,REASONING,COMPLETION}` (provider quantities wearing a service prefix; core keeps only its own packing-safety margin). Surfaced as `Provider.reasoningReserve`/`completionReserve`.
|
|
170
|
+
- **`PLURNK_PROVIDERS_TEMPERATURE` / `PLURNK_PROVIDERS_REPEAT_PENALTY` / `PLURNK_PROVIDERS_FREQUENCY_PENALTY`** -- REQUIRED sampling + anti-degeneration floors (read via `parseRequiredFloat`, values from the `.env.defaults` floor). `REPEAT_PENALTY` (canonical `1.15`) is the llama.cpp/Fireworks multiplier; `FREQUENCY_PENALTY` (canonical `0.4`, #426) is the OpenAI-standard guard on the plain cloud path, which has no `repeat_penalty`. Applied to EVERY request, keyed per backend (§13).
|
|
156
171
|
|
|
157
172
|
## §5 Alias cascade resolution
|
|
158
173
|
|
|
@@ -190,18 +205,18 @@ This package's exported resolution surface:
|
|
|
190
205
|
|
|
191
206
|
**Two-tier provider resolution — owned entirely by this package.** A provider name resolves in this order:
|
|
192
207
|
|
|
193
|
-
1. **Standard provider** (`§11`) — if `isStandardProvider(name)`, instantiated directly via `standardProviderFromEnv(name, env, model)`. No package is imported. Covers every plain OpenAI-compatible endpoint (`openai`, `groq`, `deepseek`, `mistral`, `together`, `fireworks`, `deepinfra`, `anthropic`, `bedrock`, `plurnk`, …) — including first-party Claude (Anthropic's compat endpoint, bearer auth, the `thinking` reasoning param), AWS Bedrock (its compat endpoint at the `/openai/v1` path, Bedrock-API-key bearer; the base is `BEDROCK_BASE_URL` if set, else derived as `https://bedrock-runtime.{region}.amazonaws.com/openai/v1` from the standard `AWS_REGION`/`AWS_DEFAULT_REGION`; its inference-profile model ids resolve a catalog **context window** by stripping the region and looking the model up under its publisher (#22), while cost stays unknown — bedrock marks up over the native rate), and the **`plurnk` hosted model** (a plain remote OpenAI endpoint at its `PLURNK_BASE_URL` base, `.env.defaults` → `https://plurnk.ai/v1`; reads its server-controlled context window from upstream, sends no grammar and no tuning); none needs a daughter. `plurnk` authenticates with a single optional bearer (`PLURNK_API_KEY`, sent only when set), like any other standard bearer. A spec's `apiKeyVar`/`baseUrlVar` may be a **list** of accepted env-var aliases — the conventional names the wild uses for one credential/base (e.g. `deepinfra` → `DEEPINFRA_API_KEY` / `DEEPINFRA_API_TOKEN` / `DEEPINFRA_TOKEN`; `openai` base → `OPENAI_BASE_URL` / `OPENAI_API_BASE`). First set non-empty wins; a required key unset across *all* aliases fails hard naming each. This is alias resolution over operator-set values, never a fabricated default. (An entry MAY instead supply a custom `headersFromEnv` builder for auth a bearer can't express — multi-header/credential schemes — or a `baseUrlFromEnv` builder to template the base from env, as bedrock does.) The standard table is **authoritative**: a scanned package whose name duplicates a standard one is shadowed (tier 1 returns first).
|
|
208
|
+
1. **Standard provider** (`§11`) — if `isStandardProvider(name)`, instantiated directly via `standardProviderFromEnv(name, env, model)`. No package is imported. Covers every plain OpenAI-compatible endpoint (`openai`, `groq`, `deepseek`, `mistral`, `together`, `fireworks`, `deepinfra`, `anthropic`, `bedrock`, `plurnk`, …) — including first-party Claude (Anthropic's compat endpoint, bearer auth, the `thinking` reasoning param), AWS Bedrock (its compat endpoint at the `/openai/v1` path, Bedrock-API-key bearer; the base is `BEDROCK_BASE_URL` if set, else derived as `https://bedrock-runtime.{region}.amazonaws.com/openai/v1` from the standard `AWS_REGION`/`AWS_DEFAULT_REGION`; its inference-profile model ids resolve a catalog **context window** by stripping the region and looking the model up under its publisher (#22), while cost stays unknown — bedrock marks up over the native rate), and the **`plurnk` hosted model** (a plain remote OpenAI endpoint at its `PLURNK_BASE_URL` base, `.env.defaults` → `https://plurnk.ai/v1`; reads its server-controlled context window from upstream, sends no grammar and no tuning -- `suppressTuningFloors` drops the client temperature/penalty floors so the router's per-model tuning is never overridden (#507); caller `sampling` still passes); none needs a daughter. `plurnk` authenticates with a single optional bearer (`PLURNK_API_KEY`, sent only when set), like any other standard bearer. A spec's `apiKeyVar`/`baseUrlVar` may be a **list** of accepted env-var aliases — the conventional names the wild uses for one credential/base (e.g. `deepinfra` → `DEEPINFRA_API_KEY` / `DEEPINFRA_API_TOKEN` / `DEEPINFRA_TOKEN`; `openai` base → `OPENAI_BASE_URL` / `OPENAI_API_BASE`). First set non-empty wins; a required key unset across *all* aliases fails hard naming each. This is alias resolution over operator-set values, never a fabricated default. (An entry MAY instead supply a custom `headersFromEnv` builder for auth a bearer can't express — multi-header/credential schemes — or a `baseUrlFromEnv` builder to template the base from env, as bedrock does.) The standard table is **authoritative**: a scanned package whose name duplicates a standard one is shadowed (tier 1 returns first).
|
|
194
209
|
2. **Discovered package** — otherwise, a **scope-agnostic `node_modules` scan** (`discover()`) maps the provider name to the installed package that declares `plurnk: { kind: "provider", name }`, which is dynamic-imported and `fromEnv(env, model, options?)`-called. Covers first-party daughters with real runtime surface (`openrouter`, `ollama`, `google`, `xai`, `cloudflare`, the planned `vertex`/`cohere`) **and any third-party provider published under its own scope** (`@acme/llm-provider-foo`) — no involvement from us, no `@plurnk` scope assumption. Two installed packages claiming the same name → fail-hard naming both. Unknown name → fail-hard. The scan runs once per process and is memoized.
|
|
195
210
|
|
|
196
211
|
Discovery honors the host **trust gate** `PLURNK_PLUGINS_TRUSTED_ONLY` — the same env var the execs/mimes/schemes families read (plurnk-service#229, #15). OFF (unset/empty/`0`) trusts every installed provider; ON (any value) trusts `@plurnk/*` plus a comma-separated allowlist and declines the rest. A declined package is recorded in `skipped` (never registered, never thrown), so requesting its name yields a precise *untrusted* error rather than *unknown*.
|
|
197
212
|
|
|
198
|
-
The framework is **contract-only**
|
|
213
|
+
The framework is **contract-only** -- it does **not** depend on its daughters. First-party daughters install flat (via the [`@plurnk/plurnk-providers-all`](https://github.com/plurnk/plurnk-providers-all) aggregator) so npm hoists them where the scan finds them; a framework that instead aggregated its own daughters would nest and hide them from the scan (the execs #12/#14 lesson). Each daughter declares the framework as a `peerDependency` (`^1`) so one shared copy lives in the tree -- class identity for `instanceof ProviderError` included -- without a framework minor forcing a daughter republish.
|
|
199
214
|
|
|
200
215
|
## §6 Engine → provider guarantees (consumer side)
|
|
201
216
|
|
|
202
217
|
- `messages` is a complete prompt. Consumer has pre-assembled all sections. Provider does not add, reorder, or inject turns — the wire `messages` are exactly what the consumer passed. (The provider injects no `PLAN` turn — `PLAN` is part of the consumer's grammar contract, §4.)
|
|
203
|
-
- Every `generate` carries `
|
|
204
|
-
- `signal` is wired to the
|
|
218
|
+
- Every `generate` carries `workerId` — the worker's stable, opaque identity. Same run → same string across its turns; distinct runs → distinct strings.
|
|
219
|
+
- `signal` is wired to the worker's AbortController.
|
|
205
220
|
- `generate` is single-call per turn. No parallel calls on the same instance.
|
|
206
221
|
- `assistantRaw` is opaque to the consumer (forensics-only).
|
|
207
222
|
- `meta` is the per-turn provider→client metadata bag: the backend's **non-standard top-level response fields passed through verbatim** (every provider), PLUS **validated known keys** the framework holds a contract for — currently `balancePico` (a finite pico-USD number normalized from the plurnk endpoint's balance field, renamed off its raw key; dropped if non-numeric). Absent when the backend reported no extras. The consumer (service) merges `meta` into its Turn metadata and filters what reaches the client; it reads `meta`, never mines `assistantRaw` (#23).
|
|
@@ -243,7 +258,7 @@ The framework is **contract-only** — it does **not** depend on its daughters.
|
|
|
243
258
|
import { Mock } from "@plurnk/plurnk-providers";
|
|
244
259
|
|
|
245
260
|
const mock = new Mock({
|
|
246
|
-
|
|
261
|
+
contextWindow: 100000,
|
|
247
262
|
responses: [{ assistant: { content: "<<SEND[200]:hi:SEND", reasoning: null } }],
|
|
248
263
|
});
|
|
249
264
|
const result = await mock.generate({ messages: [] });
|
|
@@ -256,7 +271,7 @@ const result = await mock.generate({ messages: [] });
|
|
|
256
271
|
A sibling package satisfies the contract when:
|
|
257
272
|
|
|
258
273
|
1. Default export is a class with `static fromEnv(env, model, options?)` factory.
|
|
259
|
-
2. Instance exposes `
|
|
274
|
+
2. Instance exposes `contextWindow: number | null` and `model: string` (non-empty).
|
|
260
275
|
3. Instance exposes `countTokens(text): number` and `costFor(usage): number`.
|
|
261
276
|
4. `countTokens("")` returns `0`; `countTokens("…")` returns a non-negative integer.
|
|
262
277
|
5. `costFor({prompt:0,completion:0,reasoning:0,cached:0,total:0})` returns `0` (or non-negative pico-USD for non-free models).
|
|
@@ -276,39 +291,40 @@ Sibling-specific behavioral tests (wire-format compliance, model-family quirks,
|
|
|
276
291
|
|
|
277
292
|
The framework ships the transport spine every OpenAI-compatible provider had been duplicating. Build a sibling *on top of these* — don't re-implement them.
|
|
278
293
|
|
|
279
|
-
- **`OpenAICompatProvider`** — a `Provider` implementation built by composition. Its `generate` does the universal work (merge `signal` with a `PLURNK_PROVIDERS_FETCH_TIMEOUT` deadline, stream the completion, map `usage`, normalize `finishReason` to the §2 set, assemble the response). Per-provider deltas arrive as config:
|
|
294
|
+
- **`OpenAICompatProvider`** — a `Provider` implementation built by composition. Its `generate` does the universal work (merge `signal` with a `PLURNK_PROVIDERS_FETCH_TIMEOUT` deadline, stream the completion, map `usage`, normalize `finishReason` to the §2 set (translating known per-backend cap synonyms -- `max_tokens`, `MAX_TOKENS`, ... -> `length` -- so a consumer's `=== "length"` holds across backends, warning once on any unmapped value, #425), assemble the response). Per-provider deltas arrive as config:
|
|
280
295
|
|
|
281
296
|
```ts
|
|
282
297
|
new OpenAICompatProvider({
|
|
283
298
|
model, url, // fully-resolved chat-completions URL
|
|
284
299
|
fetchTimeoutMs,
|
|
285
300
|
headers, // fully-resolved request headers (incl. auth)
|
|
286
|
-
|
|
287
|
-
|
|
301
|
+
contextWindow, // number | null
|
|
302
|
+
reasoning, reasoningStyle, // {mode,budget} intent + style: "none"|"think"|"include_reasoning"|"effort"|"effort_explicit"|"template"|"anthropic"
|
|
303
|
+
temperature, repeatPenalty, frequencyPenalty, // sampling + anti-degeneration floor; frequency_penalty guards the plain cloud path (#426)
|
|
288
304
|
countTokens, costFor, // strategies; default heuristic / free
|
|
289
305
|
grammarStyle, // "none" | "llamacpp" | "response_format" — GBNF wire shape (§13); default "none"
|
|
290
306
|
gbnfDebug, // PLURNK_PROVIDERS_GBNF_DEBUG: validate a grammar locally + throw on invalid, but DON'T send it (§13); default false
|
|
291
307
|
streaming, // SSE transport; default true (false → one non-streamed JSON)
|
|
292
308
|
supportsSlotPinning, slotCount, // INTERNAL slot-affinity wiring (run→id_slot); never consumer-facing
|
|
293
|
-
|
|
309
|
+
topLogprobs, rawBody, // #36 opt-in data capture (PLURNK_PROVIDERS_TOP_LOGPROBS / _RAWBODY); default off
|
|
294
310
|
servedModel, // #37 backend's self-reported served id (from the probe) → Provider.servedModel
|
|
295
311
|
});
|
|
296
312
|
```
|
|
297
313
|
|
|
298
|
-
The `openai` standard provider sets `grammarStyle: "llamacpp"`, `supportsSlotPinning`, and `slotCount` from the same llama-server fingerprint (`/v1/models` `meta` block + `/props`). The
|
|
314
|
+
The `openai` standard provider sets `grammarStyle: "llamacpp"`, `supportsSlotPinning`, and `slotCount` from the same llama-server fingerprint (`/v1/models` `meta` block + `/props`). The worker→slot mapping lives inside `OpenAICompatProvider`: sticky per `workerId`, round-robin across new runs, LRU-bounded.
|
|
299
315
|
|
|
300
316
|
- **`chatCompletionStream` / `chatCompletion` / `OpenAiHttpError` / `StreamResponse`** — the shared HTTP client (`chatCompletionStream` for SSE, `chatCompletion` for the non-streamed JSON the `streaming: false` path uses). One shared copy.
|
|
301
|
-
- **`normalizeUsage(raw)` / `computeCost(usage, {input, output, cached})`** — usage normalization to the §2 invariant (handles
|
|
317
|
+
- **`normalizeUsage(raw, reasoningText?, contentText?)` / `computeCost(usage, {input, output, cached})`** — usage normalization to the §2 invariant (handles all three reasoning-reporting conventions; the optional text args feed the Fireworks re-split, #425) and the single cost formula (bills `completion + reasoning` at the output rate). `OpenAICompatProvider` applies `normalizeUsage` automatically; siblings pass their per-token rates to `computeCost` in their `costFor`.
|
|
302
318
|
- **`parseRequiredInt` / `parseOptionalInt` / `requireEnv`** — env helpers; each takes a provider `label` for error prefixing.
|
|
303
|
-
- **`effortFromBudget(budget)`** — the shared
|
|
319
|
+
- **`effortFromBudget(budget)`** — the shared reasoning-budget → `low|medium|high` breakpoints.
|
|
304
320
|
|
|
305
321
|
A **bespoke sibling** therefore reduces to a thin class whose `fromEnv` probes whatever it needs (model catalog, pricing, context window), builds the config, and returns `new OpenAICompatProvider(config)`. A **standard provider** (§5 tier 1) needs no sibling at all — it's a frozen entry in `STANDARD_PROVIDERS` describing its key var, base-URL var, reasoning style, and tokenizer; `standardProviderFromEnv(name, env, model)` (async — returns `Promise<Provider | null>`) does the rest. The endpoint's **canonical URL ships as a floored default** in `.env.defaults` (set-if-unset, overridable in the operator's env or per-alias); it is read from the base-URL var (or a `baseUrlFromEnv` deriver) with **no in-code default**, the value living in the shipped floor, never baked into the table. Only the API **key** is required operator config (a secret with no default; fail-hard when unset).
|
|
306
322
|
|
|
307
|
-
The `plurnk` entry alone sets **`firstPartyMetadata: true`** — it forwards the consumer's per-turn `generate()` `attributions` (which installed plugin packages dispatched) and `client` (the originating frontend, e.g. `plurnk.nvim/1.4.0`) as `Plurnk-Attribution` / `Plurnk-Client` headers, and the opaque `
|
|
323
|
+
The `plurnk` entry alone sets **`firstPartyMetadata: true`** — it forwards the consumer's per-turn `generate()` `attributions` (which installed plugin packages dispatched) and `client` (the originating frontend, e.g. `plurnk.nvim/1.4.0`) as `Plurnk-Attribution` / `Plurnk-Client` headers, and the opaque `workerId` as `Plurnk-Run-Id` (#26). The gate lives on the provider, not the call site, so these first-party signals are **structurally incapable** of reaching a third-party backend. Empty values emit no header. It also alone sets **`balanceMetaKey: "balance_pico"`** — the top-level response field the plurnk endpoint reports the running account balance (pico-USD) in. `OpenAICompatProvider` surfaces the backend's extra top-level fields as `ProviderResponse.meta` for **every** provider (passed through), and for a provider that set `balanceMetaKey` it additionally normalizes that field into a validated `meta.balancePico`. So `balancePico` appears only from plurnk; the raw pass-through `meta` is general (#23).
|
|
308
324
|
|
|
309
325
|
A spec may carry a **`modelPrefix`** — a constant model-id segment the backend requires but the operator's alias shouldn't repeat. `fireworks` sets `"accounts/fireworks/models/"`, so `PLURNK_MODEL_fast=fireworks/deepseek-v4-pro` carries only the distinctive tail; `standardProviderFromEnv` prepends it idempotently (an already-prefixed id is untouched) to form the wire id, which is **also** the catalog key (models.dev keys fireworks-ai on the full id). Specs without a `modelPrefix` use the model string verbatim.
|
|
310
326
|
|
|
311
|
-
`
|
|
327
|
+
`contextWindow` for a standard provider resolves (#419): `PLURNK_PROVIDERS_CONTEXT_WINDOW` -> endpoint `n_ctx` (for `probeNctx`-flagged specs like `openai`, queried from `GET /v1/models`: llama-server reports its loaded window at `data[].meta.n_ctx`, vLLM top-level; cloud endpoints don't) -> the `@plurnk/plurnk-models` catalog -> **then the hybrid: a PROBING provider degrades to `null`, a CLOUD provider (no probe) FAILS HARD** (uncataloged + unpinned = config error, the #417 kimi case, not a guessed window). The same probe fingerprints llama-server (the `meta` block) to enable grammar transport (§13), and reads the row's `id` as `servedModel` (#37) — the real served name behind a local alias — so it runs even when the env var pins the window. The probe is best-effort: any failure resolves to `null` context / no grammar capability (a legitimate "unknown"), never throws. For a PROBING provider, an underivable window (env, probe, and catalog ALL missed) is surfaced once via a **`PLURNK_CONTEXT_UNKNOWN`** warning naming the model and the remediation (`PLURNK_PROVIDERS_CONTEXT_WINDOW`, alias-scopable) -- null stays legitimate but never silent (a CLOUD provider throws here instead, above). Operator-facing warnings (`PLURNK_TOKENIZER_HEURISTIC`, `PLURNK_PROBE_FAILED`, `PLURNK_GRAMMAR_UNVERIFIABLE`, `PLURNK_CONTEXT_UNKNOWN`, `PLURNK_FINISH_REASON_UNKNOWN`) are deduplicated **once per process per (code, message)** (#40) — repeat constructions don't re-fire them, but a *different* provider/model's first surfacing is never suppressed.
|
|
312
328
|
|
|
313
329
|
## §12 Telemetry — provider failures
|
|
314
330
|
|
|
@@ -333,11 +349,13 @@ The `TelemetryEvent` shape is mirrored **locally** (`./telemetry.ts`), structura
|
|
|
333
349
|
- **This layer** owns capability detection, transport, **and enforcement verification**: `generate({ …, grammar })` attaches the string **verbatim** as the `grammar` body field when the backend supports it, sends no grammar-related field otherwise (cloud APIs reject unknown params), and — when it did transport a grammar — checks that the response actually conforms. The provider never chooses or modifies the grammar.
|
|
334
350
|
- **The consumer** owns policy: whether to constrain a given call, and which root variant to send (e.g. the `root ::= statement` single-statement substitution that forces EOS at the close tag — the shipped `statement+` root never forces EOS, so greedy generation runs to `max_tokens`).
|
|
335
351
|
|
|
336
|
-
**Sampling guard.**
|
|
352
|
+
**Sampling guard (anti-degeneration, #426).** Degeneration into repetition loops is NOT grammar-specific -- greedy/low-temperature decoding loops on the plain cloud path too -- so the penalty rides **every** request, not just grammar'd ones, and never relies on server launch flags. The wire field is keyed on the backend (`grammarStyle`): `repeat_penalty` on `llamacpp` (the managed multiplier floor, `PLURNK_PROVIDERS_REPEAT_PENALTY`, canonical `1.15`), `repetition_penalty` on `response_format`/Fireworks (same floor, the OpenAI-compat spelling, verified honored #20), and `frequency_penalty` on `none`/plain cloud (`PLURNK_PROVIDERS_FREQUENCY_PENALTY`, canonical `0.4` -- the OpenAI-standard field, since the llama.cpp `repeat_penalty` multiplier isn't accepted there). A hard grammar makes the loop WORSE (masking removes the model's natural exits), which is why the floor was born on the grammar path; #426 promoted it to the universal guard it always needed to be. (Probed live on llama.cpp b894 + gemma-4-26B; reference: plurnk-grammar `test/llama/gbnf-live.test.ts`.)
|
|
337
353
|
|
|
338
|
-
**The cap is the
|
|
354
|
+
**The cap is still required -- and the provider now supplies its default (#507 doctrine revision).** The repeat-penalty floor suppresses short repetition cycles, NOT long-cycle degeneration: under the multi-op root (optional EOS) at near-greedy temperatures, a constrained emission can answer correctly in its first tokens and then loop to the **context wall** (observed live: 30,736 junk tokens to `finish_reason: length` -- providers#10). The former letter ("no layer defaults a cap; the consumer must bring the envelope") is revised by the owner-ruled #507 migration: the provider OWNS the default envelope (`reasoningReserve + completionReserve`, derived from the detected window), the consumer applies it as its `maxTokens` and MAY override per call -- the caller's explicit cap always wins on the wire. The spirit strengthens: nothing decodes unbounded silently, and bounding no longer depends on every consumer hand-tuning knobs.
|
|
339
355
|
|
|
340
|
-
**Native reasoning and the grammar
|
|
356
|
+
**Native reasoning and the grammar COEXIST on llama-server; the sanctioned think block is the protection, not the hazard (#488 postmortem).** With `enable_thinking: true`, the server auto-gates the grammar around the SANCTIONED reasoning block — the model thinks up front (routed to `reasoning_content`), then content decodes constrained. Verified: a 26-run baseline green under exactly this configuration, and ZERO grammar rejects across every #488 "railless" specimen (`@plurnk/gbnf` verdicts: accept or cap-truncated incomplete — the rail never left). The #488 rails-win-the-channel clamp (grammar forces `enable_thinking:false`) is REVERTED: closing the channel starves a reasoning-tuned model of its outlet, and it ESCAPES mid-content into the raw thought channel — which the server then discards while the decode runs unconstrained and billed (measured: 12,288 completion tokens billed, 1,033 chars visible, reasoning empty). So intent maps identically under a transported grammar, and the failure mode is SURFACED instead of traded against:
|
|
357
|
+
- **Per-request rail state on `meta` (#488):** every observed-grammar response carries `meta.railsAttached` (was the grammar transported) and `meta.railsVerdict` (`accept | incomplete | reject | unverifiable` from the conformance check) — the consumer's turn row then answers "did the rail ride, and did the output conform" PER TURN from the run db, so rail presence is never again inferred from output shape.
|
|
358
|
+
- **Channel-escape telemetry (#488):** billed completion tokens vastly exceeding every visible channel (`usage.completion > countTokens(content) + countTokens(reasoning) + 64`; countTokens overcounts text, so the excess is real vanishing) attaches a `grammar_unenforced` event naming the vanished balance — the run105 class (escape into a discarded reasoning block) is loud, not invisible. With the channel closed, the model reasons **inside the DSL**: the canonical grammar's required `PLAN` statement is a free-text body the model fills with genuine step-by-step reasoning before acting (probed live, b894+gemma: correct chain-of-thought inside `<<PLAN:…:PLAN`, then a clean `SEND`, `finish_reason: stop`). The `PLAN` element belongs to the consumer's grammar contract; the provider's only job is closing the native channel deterministically.
|
|
341
359
|
|
|
342
360
|
**Grammar transport — same GBNF, different wire shapes (`grammarStyle`, default `"none"`).** Backends carry the *same* grammar string differently:
|
|
343
361
|
- **`"llamacpp"`** — top-level `grammar` field + the repeat-penalty floor (`PLURNK_PROVIDERS_REPEAT_PENALTY`, §9). Detected from the §11 probe: only llama-server rows on `GET /v1/models` carry a `meta` block. This is any local llama-server (e.g. the generic `openai` provider fronting llama.cpp). `plurnk` opts out via `detectLlamaServer: false` — it reads a window but is treated as a plain OpenAI endpoint, never fingerprinted and never sent a grammar.
|
|
@@ -361,7 +379,7 @@ Zero grammar dependency (§11) is preserved: the GBNF string arrives per call; t
|
|
|
361
379
|
|
|
362
380
|
Two OPT-IN knobs surface the full signal of a paid turn for downstream IQ scoring and model distillation. Both are **OFF by default** and **per-alias-scopable** (`PLURNK_PROVIDERS_<KNOB>_<alias>`): the flag *is* the isolation, so a serving turn requests nothing on the wire and carries nothing on the response — only a dataset-scraping alias opts in. Universal: any provider (standard or daughter), any backend that returns the data.
|
|
363
381
|
|
|
364
|
-
**`
|
|
382
|
+
**`PLURNK_PROVIDERS_TOP_LOGPROBS`** (non-negative int = the OpenAI `top_logprobs`; unset = off). When set, `generate` requests `logprobs:true, top_logprobs:<n>` and surfaces `response.assistant.logprobs: Array<{ token, logprob, top? }>` plus `assistant.meanLogprob`. These are **managed fields** — reserved from caller `sampling`, so the env flag is the single control (a proxy consumer can't forge them). A backend that returns no logprobs yields an absent field — **never synthesized**.
|
|
365
383
|
|
|
366
384
|
**The `logprob` vs `sampling_logprob` decision (the honest-confidence call).** Fireworks returns both per token: `logprob` (raw model log-probability) and `sampling_logprob` (post-sampling-transform). The structured `logprob` we surface is the **raw** value — the sampling-transform-invariant measure of the model's native belief, the correct confidence signal AND distillation target. This was settled empirically, not by assumption: under grammar the two measured **identical to full float precision**, including an *adversarial* mask (grammar forcing a token the model assigned ~8%: `logprob` −2.5229365 == `sampling_logprob` −2.5229365). A post-mask renormalization would inflate confidence toward the constraint; the raw value stays honest. Anyone wanting `sampling_logprob` reads it from `rawBody`.
|
|
367
385
|
|
package/dist/Mock.d.ts
CHANGED
|
@@ -5,6 +5,10 @@ export type MockAssistant = {
|
|
|
5
5
|
usage?: Partial<ProviderUsage>;
|
|
6
6
|
finishReason?: FinishReason;
|
|
7
7
|
model?: string;
|
|
8
|
+
reasoningEncrypted?: ReadonlyArray<{
|
|
9
|
+
data: string;
|
|
10
|
+
format: string | null;
|
|
11
|
+
}>;
|
|
8
12
|
ops?: unknown[];
|
|
9
13
|
};
|
|
10
14
|
export type MockResponse = {
|
|
@@ -17,17 +21,19 @@ export type MockReturnedAssistant = ProviderAssistant & {
|
|
|
17
21
|
declare const DEFAULT_USAGE: ProviderUsage;
|
|
18
22
|
export default class Mock implements Provider {
|
|
19
23
|
#private;
|
|
20
|
-
constructor({
|
|
21
|
-
|
|
24
|
+
constructor({ contextWindow, responses }: {
|
|
25
|
+
contextWindow: number | null;
|
|
22
26
|
responses: MockResponse[];
|
|
23
27
|
});
|
|
24
|
-
get
|
|
28
|
+
get contextWindow(): number | null;
|
|
29
|
+
get reasoningReserve(): number | null;
|
|
30
|
+
get completionReserve(): number | null;
|
|
25
31
|
get model(): string;
|
|
26
32
|
countTokens(text: string): number;
|
|
27
33
|
costFor(_usage: ProviderUsage): number;
|
|
28
34
|
generate({ signal }: {
|
|
29
35
|
messages: ChatMessage[];
|
|
30
|
-
|
|
36
|
+
workerId?: string;
|
|
31
37
|
signal?: AbortSignal;
|
|
32
38
|
}): Promise<{
|
|
33
39
|
assistant: MockReturnedAssistant;
|
package/dist/Mock.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"Mock.d.ts","sourceRoot":"","sources":["../src/Mock.ts"],"names":[],"mappings":"AAOA,OAAO,KAAK,EAAE,WAAW,EAAE,YAAY,EAAE,QAAQ,EAAE,iBAAiB,EAAE,aAAa,EAAE,MAAM,YAAY,CAAC;
|
|
1
|
+
{"version":3,"file":"Mock.d.ts","sourceRoot":"","sources":["../src/Mock.ts"],"names":[],"mappings":"AAOA,OAAO,KAAK,EAAE,WAAW,EAAE,YAAY,EAAE,QAAQ,EAAE,iBAAiB,EAAE,aAAa,EAAE,MAAM,YAAY,CAAC;AAGxG,MAAM,MAAM,aAAa,GAAG;IACxB,OAAO,EAAE,MAAM,CAAC;IAChB,SAAS,EAAE,MAAM,GAAG,IAAI,CAAC;IAEzB,KAAK,CAAC,EAAE,OAAO,CAAC,aAAa,CAAC,CAAC;IAC/B,YAAY,CAAC,EAAE,YAAY,CAAC;IAC5B,KAAK,CAAC,EAAE,MAAM,CAAC;IACf,kBAAkB,CAAC,EAAE,aAAa,CAAC;QAAE,IAAI,EAAE,MAAM,CAAC;QAAC,MAAM,EAAE,MAAM,GAAG,IAAI,CAAA;KAAE,CAAC,CAAC;IAK5E,GAAG,CAAC,EAAE,OAAO,EAAE,CAAC;CACnB,CAAC;AAEF,MAAM,MAAM,YAAY,GAAG;IACvB,SAAS,EAAE,aAAa,CAAC;IACzB,YAAY,CAAC,EAAE,OAAO,CAAC;CAC1B,CAAC;AAGF,MAAM,MAAM,qBAAqB,GAAG,iBAAiB,GAAG;IAAE,GAAG,CAAC,EAAE,OAAO,EAAE,CAAA;CAAE,CAAC;AAE5E,QAAA,MAAM,aAAa,EAAE,aAA+E,CAAC;AAErG,MAAM,CAAC,OAAO,OAAO,IAAK,YAAW,QAAQ;;IAazC,YAAY,EAAE,aAAa,EAAE,SAAS,EAAE,EAAE;QAAE,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;QAAC,SAAS,EAAE,YAAY,EAAE,CAAA;KAAE,EAMpG;IAED,IAAI,aAAa,IAAI,MAAM,GAAG,IAAI,CAAgC;IAClE,IAAI,gBAAgB,IAAI,MAAM,GAAG,IAAI,CAAmC;IACxE,IAAI,iBAAiB,IAAI,MAAM,GAAG,IAAI,CAAoC;IAC1E,IAAI,KAAK,IAAI,MAAM,CAAmB;IAItC,WAAW,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAEhC;IAGD,OAAO,CAAC,MAAM,EAAE,aAAa,GAAG,MAAM,CAAc;IAE9C,QAAQ,CAAC,EAAE,MAAM,EAAE,EAAE;QAAE,QAAQ,EAAE,WAAW,EAAE,CAAC;QAAC,QAAQ,CAAC,EAAE,MAAM,CAAC;QAAC,MAAM,CAAC,EAAE,WAAW,CAAA;KAAE,GAAG,OAAO,CAAC;QAAE,SAAS,EAAE,qBAAqB,CAAC;QAAC,YAAY,EAAE,OAAO,CAAA;KAAE,CAAC,CAiBrK;IAED,IAAI,SAAS,IAAI,MAAM,CAA+B;CACzD;AAED,OAAO,EAAE,aAAa,IAAI,gBAAgB,EAAE,CAAC"}
|
package/dist/Mock.js
CHANGED
|
@@ -4,15 +4,30 @@
|
|
|
4
4
|
// engine tests; (b) worked example for sibling authors implementing the
|
|
5
5
|
// Provider contract. Production providers don't expose the `ops` escape
|
|
6
6
|
// hatch — that's an intg-only convenience.
|
|
7
|
+
import { resolveEnvelopeFromEnv } from "./env.js";
|
|
7
8
|
const DEFAULT_USAGE = { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 };
|
|
8
9
|
export default class Mock {
|
|
9
|
-
#
|
|
10
|
+
#contextWindow;
|
|
11
|
+
#reasoningReserve;
|
|
12
|
+
#completionReserve;
|
|
10
13
|
#queue;
|
|
11
|
-
|
|
12
|
-
|
|
14
|
+
// #507: reserves resolve the SAME way a real provider's do — the TOLERANT env
|
|
15
|
+
// read (the service's partition/budget suite sets PLURNK_PROVIDERS_*_RESERVE
|
|
16
|
+
// and Mock reflects it, resolved against contextWindow). Tolerant, NOT the
|
|
17
|
+
// fail-hard envelopeFromEnv: Mock is the universal fixture, constructed
|
|
18
|
+
// without the reserves in most base tests, so absent env → null → the
|
|
19
|
+
// consumer's no-cap path, and the ~100 `new Mock({ contextWindow, responses })`
|
|
20
|
+
// sites stay untouched. Mock has no alias identity, so it reads the BARE knobs.
|
|
21
|
+
constructor({ contextWindow, responses }) {
|
|
22
|
+
this.#contextWindow = contextWindow;
|
|
23
|
+
const env = resolveEnvelopeFromEnv(process.env, contextWindow);
|
|
24
|
+
this.#reasoningReserve = env.reasoningReserve;
|
|
25
|
+
this.#completionReserve = env.completionReserve;
|
|
13
26
|
this.#queue = [...responses];
|
|
14
27
|
}
|
|
15
|
-
get
|
|
28
|
+
get contextWindow() { return this.#contextWindow; }
|
|
29
|
+
get reasoningReserve() { return this.#reasoningReserve; }
|
|
30
|
+
get completionReserve() { return this.#completionReserve; }
|
|
16
31
|
get model() { return "mock"; }
|
|
17
32
|
// Heuristic tokenizer (chars/2 upper bound, matching the framework's
|
|
18
33
|
// fallback). Mock is test-only; real provider siblings ship exact counts.
|
|
@@ -33,6 +48,7 @@ export default class Mock {
|
|
|
33
48
|
content: a.content,
|
|
34
49
|
reasoning: a.reasoning,
|
|
35
50
|
usage: { ...DEFAULT_USAGE, ...a.usage },
|
|
51
|
+
...(a.reasoningEncrypted !== undefined ? { reasoningEncrypted: a.reasoningEncrypted } : {}), // #482 — sealed blobs ride the contract through Mock too
|
|
36
52
|
finishReason: a.finishReason ?? "stop",
|
|
37
53
|
model: a.model ?? "mock",
|
|
38
54
|
...(a.ops !== undefined ? { ops: a.ops } : {}),
|
package/dist/Mock.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"Mock.js","sourceRoot":"","sources":["../src/Mock.ts"],"names":[],"mappings":"AAAA,2DAA2D;AAC3D,EAAE;AACF,wEAAwE;AACxE,wEAAwE;AACxE,wEAAwE;AACxE,2CAA2C;
|
|
1
|
+
{"version":3,"file":"Mock.js","sourceRoot":"","sources":["../src/Mock.ts"],"names":[],"mappings":"AAAA,2DAA2D;AAC3D,EAAE;AACF,wEAAwE;AACxE,wEAAwE;AACxE,wEAAwE;AACxE,2CAA2C;AAG3C,OAAO,EAAE,sBAAsB,EAAE,MAAM,UAAU,CAAC;AAyBlD,MAAM,aAAa,GAAkB,EAAE,MAAM,EAAE,CAAC,EAAE,UAAU,EAAE,CAAC,EAAE,SAAS,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,KAAK,EAAE,CAAC,EAAE,CAAC;AAErG,MAAM,CAAC,OAAO,OAAO,IAAI;IACrB,cAAc,CAAgB;IAC9B,iBAAiB,CAAgB;IACjC,kBAAkB,CAAgB;IAClC,MAAM,CAAiB;IAEvB,8EAA8E;IAC9E,6EAA6E;IAC7E,2EAA2E;IAC3E,wEAAwE;IACxE,sEAAsE;IACtE,gFAAgF;IAChF,gFAAgF;IAChF,YAAY,EAAE,aAAa,EAAE,SAAS,EAA+D;QACjG,IAAI,CAAC,cAAc,GAAG,aAAa,CAAC;QACpC,MAAM,GAAG,GAAG,sBAAsB,CAAC,OAAO,CAAC,GAAG,EAAE,aAAa,CAAC,CAAC;QAC/D,IAAI,CAAC,iBAAiB,GAAG,GAAG,CAAC,gBAAgB,CAAC;QAC9C,IAAI,CAAC,kBAAkB,GAAG,GAAG,CAAC,iBAAiB,CAAC;QAChD,IAAI,CAAC,MAAM,GAAG,CAAC,GAAG,SAAS,CAAC,CAAC;IACjC,CAAC;IAED,IAAI,aAAa,KAAoB,OAAO,IAAI,CAAC,cAAc,CAAC,CAAC,CAAC;IAClE,IAAI,gBAAgB,KAAoB,OAAO,IAAI,CAAC,iBAAiB,CAAC,CAAC,CAAC;IACxE,IAAI,iBAAiB,KAAoB,OAAO,IAAI,CAAC,kBAAkB,CAAC,CAAC,CAAC;IAC1E,IAAI,KAAK,KAAa,OAAO,MAAM,CAAC,CAAC,CAAC;IAEtC,qEAAqE;IACrE,0EAA0E;IAC1E,WAAW,CAAC,IAAY;QACpB,OAAO,IAAI,CAAC,MAAM,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IAC9D,CAAC;IAED,gBAAgB;IAChB,OAAO,CAAC,MAAqB,IAAY,OAAO,CAAC,CAAC,CAAC,CAAC;IAEpD,KAAK,CAAC,QAAQ,CAAC,EAAE,MAAM,EAAwE;QAC3F,oEAAoE;QACpE,mEAAmE;QACnE,MAAM,EAAE,cAAc,EAAE,CAAC;QACzB,MAAM,IAAI,GAAG,IAAI,CAAC,MAAM,CAAC,KAAK,EAAE,CAAC;QACjC,IAAI,IAAI,KAAK,SAAS;YAAE,MAAM,IAAI,KAAK,CAAC,mDAAmD,CAAC,CAAC;QAC7F,MAAM,CAAC,GAAG,IAAI,CAAC,SAAS,CAAC;QACzB,MAAM,SAAS,GAA0B;YACrC,OAAO,EAAE,CAAC,CAAC,OAAO;YAClB,SAAS,EAAE,CAAC,CAAC,SAAS;YACtB,KAAK,EAAE,EAAE,GAAG,aAAa,EAAE,GAAG,CAAC,CAAC,KAAK,EAAE;YACvC,GAAG,CAAC,CAAC,CAAC,kBAAkB,KAAK,SAAS,CAAC,CAAC,CAAC,EAAE,kBAAkB,EAAE,CAAC,CAAC,kBAAkB,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,yDAAyD;YACtJ,YAAY,EAAE,CAAC,CAAC,YAAY,IAAI,MAAM;YACtC,KAAK,EAAE,CAAC,CAAC,KAAK,IAAI,MAAM;YACxB,GAAG,CAAC,CAAC,CAAC,GAAG,KAAK,SAAS,CAAC,CAAC,CAAC,EAAE,GAAG,EAAE,CAAC,CAAC,GAAG,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC;SACjD,CAAC;QACF,OAAO,EAAE,SAAS,EAAE,YAAY,EAAE,IAAI,CAAC,YAAY,IAAI,IAAI,EAAE,CAAC;IAClE,CAAC;IAED,IAAI,SAAS,KAAa,OAAO,IAAI,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC,CAAC;CACzD;AAED,OAAO,EAAE,aAAa,IAAI,gBAAgB,EAAE,CAAC"}
|
package/dist/OpenAICompat.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { ChatMessage, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
2
|
-
import type { Reasoning } from "./env.ts";
|
|
2
|
+
import type { Reasoning, ReserveSpec } from "./env.ts";
|
|
3
3
|
export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "template" | "anthropic";
|
|
4
4
|
export type GrammarStyle = "none" | "llamacpp" | "response_format";
|
|
5
5
|
export type OpenAICompatConfig = {
|
|
@@ -7,7 +7,7 @@ export type OpenAICompatConfig = {
|
|
|
7
7
|
url: string;
|
|
8
8
|
fetchTimeoutMs: number;
|
|
9
9
|
headers?: Record<string, string>;
|
|
10
|
-
|
|
10
|
+
contextWindow?: number | null;
|
|
11
11
|
reasoningStyle?: ReasoningStyle;
|
|
12
12
|
countTokens?: (text: string) => number;
|
|
13
13
|
costFor?: (usage: ProviderUsage) => number;
|
|
@@ -25,33 +25,39 @@ export type OpenAICompatConfig = {
|
|
|
25
25
|
reasoning: Reasoning;
|
|
26
26
|
temperature: number;
|
|
27
27
|
repeatPenalty: number;
|
|
28
|
+
frequencyPenalty?: number;
|
|
28
29
|
retryDelayMs: number;
|
|
29
30
|
retryAttempts: number;
|
|
30
|
-
|
|
31
|
+
topLogprobs?: number | null;
|
|
31
32
|
rawBody?: boolean;
|
|
33
|
+
reasoningReserve?: ReserveSpec;
|
|
34
|
+
completionReserve?: ReserveSpec;
|
|
35
|
+
tuningFloors?: boolean;
|
|
32
36
|
};
|
|
33
37
|
export declare const effortFromBudget: (budget: number) => "low" | "medium" | "high";
|
|
34
38
|
export default class OpenAICompatProvider implements Provider {
|
|
35
39
|
#private;
|
|
36
40
|
tokenize?: (text: string) => Promise<number[]>;
|
|
37
41
|
constructor(config: OpenAICompatConfig);
|
|
38
|
-
get
|
|
42
|
+
get contextWindow(): number | null;
|
|
43
|
+
get reasoningReserve(): number | null;
|
|
44
|
+
get completionReserve(): number | null;
|
|
39
45
|
get model(): string;
|
|
40
46
|
get servedModel(): string | undefined;
|
|
41
47
|
get requiresMaxTokens(): boolean | undefined;
|
|
42
48
|
get constrainsOutput(): boolean;
|
|
43
49
|
countTokens(text: string): number;
|
|
44
50
|
costFor(usage: ProviderUsage): number;
|
|
45
|
-
generate({ messages,
|
|
51
|
+
generate({ messages, workerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }: {
|
|
46
52
|
messages: ChatMessage[];
|
|
47
|
-
|
|
53
|
+
workerId: string;
|
|
48
54
|
signal?: AbortSignal;
|
|
49
55
|
grammar?: string;
|
|
50
56
|
maxTokens?: number;
|
|
51
57
|
attributions?: string[];
|
|
52
58
|
client?: string;
|
|
53
59
|
strikes?: number;
|
|
54
|
-
|
|
60
|
+
workspaceId?: string;
|
|
55
61
|
loop?: number;
|
|
56
62
|
turn?: number;
|
|
57
63
|
sampling?: Record<string, unknown>;
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"OpenAICompat.d.ts","sourceRoot":"","sources":["../src/OpenAICompat.ts"],"names":[],"mappings":"AAUA,OAAO,KAAK,EAAE,WAAW,EAAgB,QAAQ,EAAE,gBAAgB,EAAE,aAAa,EAAE,MAAM,YAAY,CAAC;AACvG,OAAO,KAAK,EAAE,SAAS,EAAE,MAAM,UAAU,CAAC;
|
|
1
|
+
{"version":3,"file":"OpenAICompat.d.ts","sourceRoot":"","sources":["../src/OpenAICompat.ts"],"names":[],"mappings":"AAUA,OAAO,KAAK,EAAE,WAAW,EAAgB,QAAQ,EAAE,gBAAgB,EAAE,aAAa,EAAE,MAAM,YAAY,CAAC;AACvG,OAAO,KAAK,EAAE,SAAS,EAAE,WAAW,EAAE,MAAM,UAAU,CAAC;AAkBvD,MAAM,MAAM,cAAc,GAAG,MAAM,GAAG,OAAO,GAAG,mBAAmB,GAAG,QAAQ,GAAG,iBAAiB,GAAG,UAAU,GAAG,WAAW,CAAC;AAO9H,MAAM,MAAM,YAAY,GAAG,MAAM,GAAG,UAAU,GAAG,iBAAiB,CAAC;AAEnE,MAAM,MAAM,kBAAkB,GAAG;IAC7B,KAAK,EAAE,MAAM,CAAC;IACd,GAAG,EAAE,MAAM,CAAC;IACZ,cAAc,EAAE,MAAM,CAAC;IACvB,OAAO,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,MAAM,CAAC,CAAC;IACjC,aAAa,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC9B,cAAc,CAAC,EAAE,cAAc,CAAC;IAChC,WAAW,CAAC,EAAE,CAAC,IAAI,EAAE,MAAM,KAAK,MAAM,CAAC;IACvC,OAAO,CAAC,EAAE,CAAC,KAAK,EAAE,aAAa,KAAK,MAAM,CAAC;IAC3C,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,YAAY,CAAC,EAAE,YAAY,CAAC;IAC5B,SAAS,CAAC,EAAE,OAAO,CAAC;IACpB,SAAS,CAAC,EAAE,OAAO,CAAC;IACpB,kBAAkB,CAAC,EAAE,OAAO,CAAC;IAC7B,cAAc,CAAC,EAAE,MAAM,CAAC;IAExB,mBAAmB,CAAC,EAAE,OAAO,CAAC;IAC9B,SAAS,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAI1B,WAAW,CAAC,EAAE,MAAM,CAAC;IAKrB,WAAW,CAAC,EAAE,MAAM,CAAC;IAIrB,iBAAiB,CAAC,EAAE,OAAO,CAAC;IAM5B,SAAS,EAAE,SAAS,CAAC;IAQrB,WAAW,EAAE,MAAM,CAAC;IACpB,aAAa,EAAE,MAAM,CAAC;IAKtB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,YAAY,EAAE,MAAM,CAAC;IAIrB,aAAa,EAAE,MAAM,CAAC;IAQtB,WAAW,CAAC,EAAE,MAAM,GAAG,IAAI,CAAC;IAC5B,OAAO,CAAC,EAAE,OAAO,CAAC;IAOlB,gBAAgB,CAAC,EAAE,WAAW,CAAC;IAC/B,iBAAiB,CAAC,EAAE,WAAW,CAAC;IAIhC,YAAY,CAAC,EAAE,OAAO,CAAC;CAC1B,CAAC;AA6CF,eAAO,MAAM,gBAAgB,WAAY,MAAM,KAAG,KAAK,GAAG,QAAQ,GAAG,MAIpE,CAAC;AAuCF,MAAM,CAAC,OAAO,OAAO,oBAAqB,YAAW,QAAQ;;IAmCzD,QAAQ,CAAC,EAAE,CAAC,IAAI,EAAE,MAAM,KAAK,OAAO,CAAC,MAAM,EAAE,CAAC,CAAC;IAE/C,YAAY,MAAM,EAAE,kBAAkB,EAqDrC;IAED,IAAI,aAAa,IAAI,MAAM,GAAG,IAAI,CAAgC;IAQlE,IAAI,gBAAgB,IAAI,MAAM,GAAG,IAAI,CAAyD;IAC9F,IAAI,iBAAiB,IAAI,MAAM,GAAG,IAAI,CAA0D;IAChG,IAAI,KAAK,IAAI,MAAM,CAAwB;IAE3C,IAAI,WAAW,IAAI,MAAM,GAAG,SAAS,CAA8B;IAEnE,IAAI,iBAAiB,IAAI,OAAO,GAAG,SAAS,CAAoC;IAIhF,IAAI,gBAAgB,IAAI,OAAO,CAA0C;IAEzE,WAAW,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAAoC;IACrE,OAAO,CAAC,KAAK,EAAE,aAAa,GAAG,MAAM,CAAiC;IAmNhE,QAAQ,CAAC,EAAE,QAAQ,EAAE,QAAQ,EAAE,MAAM,EAAE,OAAO,EAAE,SAAS,EAAE,YAAY,EAAE,MAAM,EAAE,OAAO,EAAE,WAAW,EAAE,IAAI,EAAE,IAAI,EAAE,QAAQ,EAAE,EAAE;QAAE,QAAQ,EAAE,WAAW,EAAE,CAAC;QAAC,QAAQ,EAAE,MAAM,CAAC;QAAC,MAAM,CAAC,EAAE,WAAW,CAAC;QAAC,OAAO,CAAC,EAAE,MAAM,CAAC;QAAC,SAAS,CAAC,EAAE,MAAM,CAAC;QAAC,YAAY,CAAC,EAAE,MAAM,EAAE,CAAC;QAAC,MAAM,CAAC,EAAE,MAAM,CAAC;QAAC,OAAO,CAAC,EAAE,MAAM,CAAC;QAAC,WAAW,CAAC,EAAE,MAAM,CAAC;QAAC,IAAI,CAAC,EAAE,MAAM,CAAC;QAAC,IAAI,CAAC,EAAE,MAAM,CAAC;QAAC,QAAQ,CAAC,EAAE,MAAM,CAAC,MAAM,EAAE,OAAO,CAAC,CAAA;KAAE,GAAG,OAAO,CAAC,gBAAgB,CAAC,CAuI7Z;CACJ"}
|