@plurnk/plurnk-providers 1.0.4 → 1.0.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +93 -25
- package/README.md +4 -4
- package/SPEC.md +46 -36
- package/dist/Mock.d.ts +4 -4
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +4 -4
- package/dist/Mock.js.map +1 -1
- package/dist/OpenAICompat.d.ts +7 -6
- package/dist/OpenAICompat.d.ts.map +1 -1
- package/dist/OpenAICompat.js +131 -53
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/env.d.ts +2 -1
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +30 -11
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +2 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -1
- package/dist/index.js.map +1 -1
- package/dist/openaiStream.d.ts +5 -0
- package/dist/openaiStream.d.ts.map +1 -1
- package/dist/openaiStream.js +31 -1
- package/dist/openaiStream.js.map +1 -1
- package/dist/standardProviders.d.ts.map +1 -1
- package/dist/standardProviders.js +27 -20
- package/dist/standardProviders.js.map +1 -1
- package/dist/types.d.ts +7 -3
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +1 -1
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +16 -2
- package/dist/usage.js.map +1 -1
- package/package.json +6 -6
package/.env.defaults
CHANGED
|
@@ -1,24 +1,28 @@
|
|
|
1
|
-
#
|
|
2
|
-
# PLURNK_PROVIDERS_* knobs
|
|
3
|
-
#
|
|
4
|
-
#
|
|
5
|
-
# ~/.plurnk/.env or ./.env. A key
|
|
1
|
+
# REFERENCE - @plurnk/plurnk-providers' shipped .env.defaults: the operative floor for the
|
|
2
|
+
# PLURNK_PROVIDERS_* knobs AND the standard providers' endpoints + alias config (ecosystem
|
|
3
|
+
# standard, providers#44: every package owns what it reads; the file IS the documentation).
|
|
4
|
+
# The daemon assembles every installed member's file into one floor (set-if-unset under the
|
|
5
|
+
# operator's env) - do NOT edit this file; put YOUR config in ~/.plurnk/.env or ./.env. A key
|
|
6
|
+
# claimed by two packages crashes boot naming both.
|
|
6
7
|
#
|
|
7
|
-
# EVERY knob
|
|
8
|
-
#
|
|
9
|
-
#
|
|
10
|
-
#
|
|
11
|
-
#
|
|
8
|
+
# EVERY PLURNK_PROVIDERS_* knob is per-alias-scopable: PLURNK_PROVIDERS_<KNOB>_<alias> wins over
|
|
9
|
+
# the bare name (suffix case-folds; SPEC §4). Set lines are the floor; commented lines are values
|
|
10
|
+
# whose UNSET state is meaningful (derive/auto/off, or a required secret) - uncomment to pin.
|
|
11
|
+
#
|
|
12
|
+
# Base URLs are SHIPPED DEFAULTS - the canonical vendor endpoint lives HERE (overridable in your
|
|
13
|
+
# .env, or per-alias via PLURNK_BASEURL_<alias>), so the happy path needs ZERO endpoint config.
|
|
14
|
+
# The ONLY required per-provider input is the API KEY: a secret with no default, set in your own
|
|
15
|
+
# .env; an unset key for a provider you actually use fails hard at construction naming the var.
|
|
12
16
|
|
|
13
17
|
# --- Side-channel reasoning (SPEC §4, #32/#33/#399) ---
|
|
14
18
|
# ACTIVATION and BUDGET are separate so a numeric can never silently flip wire flags.
|
|
15
19
|
# off | adaptive | on. The provider maps intent to each backend's native mechanism
|
|
16
|
-
# (reasoning_effort, enable_thinking, think,
|
|
20
|
+
# (reasoning_effort, enable_thinking, think, ...). Default ADAPTIVE (owner ruling,
|
|
17
21
|
# #399): reasoning ACTIVE on a fresh install, each backend's own adaptive depth,
|
|
18
|
-
# no shipped magnitude
|
|
19
|
-
# declares done (svc#396 A/B: 25 EXECs
|
|
22
|
+
# no shipped magnitude - a reasoning model that ships un-reasoning blind-edits and
|
|
23
|
+
# declares done (svc#396 A/B: 25 EXECs -> 0, build broken, confident SEND[200]).
|
|
20
24
|
PLURNK_PROVIDERS_REASONING=adaptive
|
|
21
|
-
# Positive int, REQUIRED iff REASONING=on
|
|
25
|
+
# Positive int, REQUIRED iff REASONING=on - the magnitude for tier/budget mapping. On
|
|
22
26
|
# llama.cpp the ENFORCEMENT is the box's --reasoning-budget LAUNCH flag (per-request
|
|
23
27
|
# numerics are ignored): env budget and launch flag are the same number, changed together.
|
|
24
28
|
# PLURNK_PROVIDERS_REASONING_BUDGET=4096
|
|
@@ -28,6 +32,13 @@ PLURNK_PROVIDERS_REASONING=adaptive
|
|
|
28
32
|
# the floor the provider manages wherever a grammar rides (greedy-under-mask loops without it).
|
|
29
33
|
PLURNK_PROVIDERS_TEMPERATURE=0.2
|
|
30
34
|
PLURNK_PROVIDERS_REPEAT_PENALTY=1.15
|
|
35
|
+
# FREQUENCY_PENALTY (#426): the anti-degeneration guard on the CLOUD path. repeat_penalty is a
|
|
36
|
+
# llama.cpp/vLLM MULTIPLIER the plain cloud path (grammarStyle "none") can't use; frequency_penalty
|
|
37
|
+
# is OpenAI-standard, so every OpenAI-compat backend accepts it (verified live: together, deepinfra,
|
|
38
|
+
# fireworks). Without it a cloud alias ran the sampler bare and looped to the token cap. 0 = off.
|
|
39
|
+
# NOTE: 0.4 is a STARTING value (acceptance-verified, not yet effectiveness-benched) - raise it if a
|
|
40
|
+
# firefast/deepseek re-bench still degenerates.
|
|
41
|
+
PLURNK_PROVIDERS_FREQUENCY_PENALTY=0.4
|
|
31
42
|
|
|
32
43
|
# --- Transport budgets (§4, #18) ---
|
|
33
44
|
# Per-attempt fetch timeout (ms); the caller's abort signal spans retries.
|
|
@@ -45,26 +56,26 @@ PLURNK_PROVIDERS_PROBE_DELAY=250
|
|
|
45
56
|
|
|
46
57
|
# --- Grammar (GBNF, SPEC §13) ---
|
|
47
58
|
# The grammar artifact the consumer resolves and passes per call (providers transport it
|
|
48
|
-
# verbatim; enforcement is per-backend
|
|
59
|
+
# verbatim; enforcement is per-backend - see constrainsOutput). Read by the service.
|
|
49
60
|
# OFF by default (bare unset = off): core opts IN per alias via PLURNK_PROVIDERS_GBNF_<alias>
|
|
50
|
-
# (Engine §353). A GLOBAL rails default is unsafe
|
|
61
|
+
# (Engine §353). A GLOBAL rails default is unsafe - providers that ACCEPT a grammar quietly
|
|
51
62
|
# disable reasoning as a side effect, invisible to the constrainsOutput gate (it verifies rails
|
|
52
63
|
# are ENFORCED, not that reasoning SURVIVED). The reasoning model then blind-edits and looks
|
|
53
64
|
# like plurnk's fault on their platform (svc#396). Off degrades gracefully (free output, still
|
|
54
|
-
# verified; divergence
|
|
65
|
+
# verified; divergence -> grammar_unenforced telemetry, #24); rails-on-where-unsafe degrades
|
|
55
66
|
# silently. Turn ON per alias where rails coexist (local llama-server) or the reasoning
|
|
56
67
|
# tradeoff is deliberate.
|
|
57
68
|
# PLURNK_PROVIDERS_GBNF=plurnk.gbnf
|
|
58
69
|
# Debug toggle (0/empty = off): validate a transported grammar locally, throw on malformed,
|
|
59
|
-
# and WITHHOLD it so the model runs unconstrained
|
|
70
|
+
# and WITHHOLD it so the model runs unconstrained - the free output is still verified and a
|
|
60
71
|
# divergence surfaces as grammar_unenforced telemetry. Dev aid; leave off in production.
|
|
61
72
|
# PLURNK_PROVIDERS_GBNF_DEBUG=0
|
|
62
73
|
|
|
63
74
|
# --- Window (SPEC §11) ---
|
|
64
|
-
# Unset = derive: env
|
|
65
|
-
# once via PLURNK_CONTEXT_UNKNOWN naming the model). Set to pin the window deliberately
|
|
75
|
+
# Unset = derive: env -> live endpoint probe (n_ctx) -> models.dev catalog -> null (surfaced
|
|
76
|
+
# once via PLURNK_CONTEXT_UNKNOWN naming the model). Set to pin the window deliberately -
|
|
66
77
|
# per-alias (_<alias>) for a box whose model the catalog doesn't know.
|
|
67
|
-
#
|
|
78
|
+
# PLURNK_PROVIDERS_CONTEXT_WINDOW=200000
|
|
68
79
|
|
|
69
80
|
# --- llama-server detection pin (#34) ---
|
|
70
81
|
# Unset = auto-detect from the /v1/models fingerprint. 1 = pin llama-server capabilities
|
|
@@ -72,10 +83,67 @@ PLURNK_PROVIDERS_PROBE_DELAY=250
|
|
|
72
83
|
# remote even when the fingerprint matches. Usually per-alias: _<alias>.
|
|
73
84
|
# PLURNK_PROVIDERS_LLAMA_SERVER=1
|
|
74
85
|
|
|
75
|
-
# --- Data capture (#36, SPEC §14)
|
|
86
|
+
# --- Data capture (#36, SPEC §14) - OFF by default; the flag IS the isolation ---
|
|
76
87
|
# Enable ONLY on a dataset-scraping alias (append _<alias>); serving turns then request and
|
|
77
|
-
# carry nothing.
|
|
78
|
-
# raw model logprob, the sampling-invariant confidence). RAWBODY: truthy
|
|
88
|
+
# carry nothing. TOP_LOGPROBS: non-negative int = the OpenAI top_logprobs count (surfaced on
|
|
89
|
+
# assistant.logprobs; raw model logprob, the sampling-invariant confidence). RAWBODY: truthy -> verbatim wire
|
|
79
90
|
# body on response.rawBody (keeps sampling_logprob/token_id/bytes the digest drops).
|
|
80
|
-
#
|
|
91
|
+
# PLURNK_PROVIDERS_TOP_LOGPROBS_myscraper=3
|
|
81
92
|
# PLURNK_PROVIDERS_RAWBODY_myscraper=1
|
|
93
|
+
|
|
94
|
+
# --- Alias cascade (SPEC §5) - declare aliases, then pick the active one ---
|
|
95
|
+
# First path segment is the provider name; the rest is the provider-native model id (may
|
|
96
|
+
# contain "/"). Operator-specific, no default - the examples are commented.
|
|
97
|
+
# PLURNK_MODEL_gemma=openai/macher.gguf
|
|
98
|
+
# PLURNK_MODEL_opus=anthropic/claude-opus-4-8
|
|
99
|
+
# PLURNK_MODEL_sonnet=openrouter/anthropic/claude-sonnet-latest
|
|
100
|
+
# PLURNK_MODEL=gemma
|
|
101
|
+
#
|
|
102
|
+
# PLURNK_BASEURL_<alias> - bind an endpoint to ONE alias (the only way to run N self-hosted
|
|
103
|
+
# boxes of one provider: openai=llama.cpp/vLLM, ollama). WINS over the provider's *_BASE_URL;
|
|
104
|
+
# a dangling override fails hard.
|
|
105
|
+
# PLURNK_MODEL_HAZEL1=openai/qwen2.5-coder
|
|
106
|
+
# PLURNK_BASEURL_HAZEL1=http://hazel1:8080/v1
|
|
107
|
+
# PLURNK_MODEL_NOOK=ollama/qwen2.5-coder
|
|
108
|
+
# PLURNK_BASEURL_NOOK=http://nook:11434
|
|
109
|
+
|
|
110
|
+
# --- Standard provider endpoints (frozen table) ---
|
|
111
|
+
# Base URLs ship at their canonical vendor default - zero endpoint config for the happy path.
|
|
112
|
+
# Override in your .env (proxy/self-host/regional twin), or per-alias with PLURNK_BASEURL_<alias>
|
|
113
|
+
# (which wins over these). Each provider ALSO needs its API KEY - a required secret with no
|
|
114
|
+
# default, set in ~/.plurnk/.env; the key var mirrors the *_BASE_URL prefix (FIREWORKS_API_KEY,
|
|
115
|
+
# ...) except where noted. An unset key fails hard at construction naming the var.
|
|
116
|
+
|
|
117
|
+
# Generic OpenAI-compat - OpenAI proper by default; for a local llama-server/vLLM, override the
|
|
118
|
+
# endpoint per-alias (PLURNK_BASEURL_<alias>). OPENAI_API_BASE is a legacy URL alias.
|
|
119
|
+
OPENAI_BASE_URL=https://api.openai.com/v1
|
|
120
|
+
GROQ_BASE_URL=https://api.groq.com/openai/v1
|
|
121
|
+
DEEPSEEK_BASE_URL=https://api.deepseek.com/v1
|
|
122
|
+
MISTRAL_BASE_URL=https://api.mistral.ai/v1
|
|
123
|
+
TOGETHER_BASE_URL=https://api.together.xyz/v1
|
|
124
|
+
FIREWORKS_BASE_URL=https://api.fireworks.ai/inference/v1
|
|
125
|
+
DEEPINFRA_BASE_URL=https://api.deepinfra.com/v1/openai # key aliases: DEEPINFRA_API_TOKEN, DEEPINFRA_TOKEN
|
|
126
|
+
|
|
127
|
+
# Chinese cloud hosts - base ships INTERNATIONAL; mainland operators repoint *_BASE_URL at the `.cn` twin.
|
|
128
|
+
MOONSHOT_BASE_URL=https://api.moonshot.ai/v1
|
|
129
|
+
DASHSCOPE_BASE_URL=https://dashscope-intl.aliyuncs.com/compatible-mode/v1 # Alibaba Qwen
|
|
130
|
+
ZHIPU_BASE_URL=https://api.z.ai/api/paas/v4 # Zhipu GLM; key ZHIPUAI_API_KEY (alias ZAI_API_KEY)
|
|
131
|
+
ARK_BASE_URL=https://ark.ap-southeast.bytepluses.com/api/v3 # ByteDance Doubao (BytePlus ModelArk)
|
|
132
|
+
HUNYUAN_BASE_URL=https://api.hunyuan.cloud.tencent.com/v1 # Tencent Hunyuan
|
|
133
|
+
MINIMAX_BASE_URL=https://api.minimax.io/v1
|
|
134
|
+
STEPFUN_BASE_URL=https://api.stepfun.com/v1 # StepFun; key STEP_API_KEY
|
|
135
|
+
BAICHUAN_BASE_URL=https://api.baichuan-ai.com/v1
|
|
136
|
+
QIANFAN_BASE_URL=https://qianfan.baidubce.com/v2 # Baidu ERNIE (Qianfan v2)
|
|
137
|
+
SILICONFLOW_BASE_URL=https://api.siliconflow.com/v1
|
|
138
|
+
MODELSCOPE_BASE_URL=https://api-inference.modelscope.cn/v1 # key alias: MODELSCOPE_TOKEN
|
|
139
|
+
|
|
140
|
+
# First-party Claude (Anthropic's OpenAI-compat endpoint).
|
|
141
|
+
ANTHROPIC_BASE_URL=https://api.anthropic.com/v1
|
|
142
|
+
|
|
143
|
+
# AWS Bedrock (OpenAI-compat path /openai/v1) - base is REGION-templated: leave unset to derive
|
|
144
|
+
# from AWS_REGION / AWS_DEFAULT_REGION, or pin explicitly. No static default. Key AWS_BEARER_TOKEN_BEDROCK.
|
|
145
|
+
# BEDROCK_BASE_URL=https://bedrock-runtime.us-east-1.amazonaws.com/openai/v1
|
|
146
|
+
|
|
147
|
+
# plurnk hosted model - PLURNK_API_KEY is an OPTIONAL bearer (sent only when set).
|
|
148
|
+
PLURNK_BASE_URL=https://plurnk.ai/v1
|
|
149
|
+
# PLURNK_BASE_URL=http://plurnksnr2kihuukt6v22ko72r34dxeatbsfhgow3hvnlw6btanxphad.onion/v1 # Tor
|
package/README.md
CHANGED
|
@@ -5,7 +5,7 @@ Framework + contract for `@plurnk/plurnk-providers-*` sibling packages (LLM tran
|
|
|
5
5
|
## Documentation
|
|
6
6
|
|
|
7
7
|
- [`SPEC.md`](./SPEC.md) — author-facing contract for sibling implementers.
|
|
8
|
-
- [`.env.
|
|
8
|
+
- [`.env.defaults`](./.env.defaults) — the shipped floor AND operator config reference: every var the provider layer reads, each with its default (canonical endpoints + measured tuning); only the API keys are secrets with no default (fail-hard). Assembled into the daemon floor at boot (#44).
|
|
9
9
|
- Constellation: [plurnk-grammar](https://github.com/plurnk/plurnk-grammar) (HEREDOC + AST), [plurnk-mimetypes](https://github.com/plurnk/plurnk-mimetypes), [plurnk-schemes](https://github.com/plurnk/plurnk-schemes), [plurnk-execs](https://github.com/plurnk/plurnk-execs) (the reference family this one mirrors).
|
|
10
10
|
|
|
11
11
|
## Write a provider
|
|
@@ -27,13 +27,13 @@ One package is **one** provider identity — the `<name>` segment of `PLURNK_MOD
|
|
|
27
27
|
The framework calls `YourClass.fromEnv(env, model, options?)` (sync or async) and expects a `Provider`. `options.baseUrl` is the per-alias endpoint override (`PLURNK_BASEURL_<alias>`) — honor it if you're a self-hosted provider so two aliases can reach two boxes; ignore it otherwise. Two ways in:
|
|
28
28
|
|
|
29
29
|
- **OpenAI-compatible backends** (the common case): `fromEnv` reads its env (base URL, key), probes whatever it needs (catalog, context window, pricing), and returns **`new OpenAICompatProvider(config)`**. You write a `fromEnv` and a config object — the transport spine (SSE, usage normalization, `finishReason`, grammar transport, slot affinity) is inherited. See `OpenAICompatConfig` / SPEC §11.
|
|
30
|
-
- **Non-OpenAI backends**: `implements Provider` directly — `generate`, `
|
|
30
|
+
- **Non-OpenAI backends**: `implements Provider` directly — `generate`, `contextWindow`, `model`, `countTokens(text)`, `costFor(usage)`.
|
|
31
31
|
|
|
32
32
|
`fromEnv` **MUST fail fast with a named error** when required env is missing — name the var the operator must set. (Why a factory, not a base-class constructor like execs/mimes: a provider often async-probes at construction — SPEC §3.)
|
|
33
33
|
|
|
34
34
|
### 3. What `generate` receives — and returns
|
|
35
35
|
|
|
36
|
-
`generate({ messages,
|
|
36
|
+
`generate({ messages, workerId, signal?, grammar?, maxTokens?, attributions?, client? }) → Promise<ProviderResponse>`. Return **raw** wire output: `content` unparsed (the consumer parses the plurnk DSL — never parse it yourself), `reasoning` is the wire-reported CoT only. Honor `signal`. The provider never mutates `messages` or injects turns. `grammar` (GBNF) is attached only by backends that support it; all others ignore it (SPEC §13). When a grammar *is* transported, the provider verifies the backend actually enforced it — non-conforming output rejects with a `grammar_unenforced` `ProviderError` (a conformance check via `@plurnk/gbnf`, never a plurnk-DSL parse). In **GBNF-filter mode** (`PLURNK_GBNF_DEBUG`, grammar withheld) the same non-conformance is **non-fatal**: `generate` returns the model's bytes and attaches a `grammar_unenforced` event to `ProviderResponse.telemetry` with the divergence `position`, so the consumer can drive self-correction instead of losing the turn (#24). `attributions`/`client` are per-turn first-party metadata, forwarded as `Plurnk-*` headers **only** by a provider configured with `firstPartyMetadata` (the plurnk endpoint); every other provider drops them, so they can never reach a third-party backend (SPEC §11). The response carries a `meta?` bag — the backend's extra top-level fields passed through, plus validated known keys (e.g. `meta.balancePico`, pico-USD, normalized only from the plurnk endpoint) — for the service's per-turn metadata (#23).
|
|
37
37
|
|
|
38
38
|
## Discovery & trust
|
|
39
39
|
|
|
@@ -51,7 +51,7 @@ First-party daughters install flat via [`@plurnk/plurnk-providers-all`](https://
|
|
|
51
51
|
- `parseAliasesFromEnv`, `resolveActiveAlias`, `instantiateProvider`, `loadActiveProvider`, `discover`, `resetDiscoveryCache` — alias-cascade resolution + two-tier provider instantiation (`resetDiscoveryCache` clears the memoized tier-2 scan; for tests). Tier 1 is the standard table; tier 2 is a scope-agnostic `node_modules` scan for `plurnk.kind:"provider"` packages — first-party daughters (flat via `@plurnk/plurnk-providers-all`) and third-party providers under any scope, gated by the host `PLURNK_PLUGINS_TRUSTED_ONLY` allowlist. The framework is contract-only (SPEC §5).
|
|
52
52
|
- `OpenAICompatProvider` (+ `OpenAICompatConfig`, `ReasoningStyle`, `GrammarStyle`, `effortFromBudget`) — shared OpenAI-compatible transport spine; siblings extend it (SPEC §11). Transports a GBNF grammar via `grammarStyle` — `llamacpp` (top-level `grammar` field) or `response_format` (Fireworks); `none` drops it — and verifies the backend enforced it against `@plurnk/gbnf`, rejecting non-conforming output as `grammar_unenforced` (SPEC §13).
|
|
53
53
|
- `chatCompletionStream`, `chatCompletion`, `OpenAiHttpError`, `StreamResponse` — the shared SSE client (`chatCompletion` is the non-streaming variant).
|
|
54
|
-
- `parseRequiredInt`, `parseOptionalInt`, `requireEnv`, `
|
|
54
|
+
- `parseRequiredInt`, `parseOptionalInt`, `parseRequiredFloat`, `requireEnv`, `reasoningFromEnv` — env helpers (SPEC §4; all required-with-named-errors, no in-code defaults).
|
|
55
55
|
- `normalizeUsage`, `computeCost` (+ `RawUsage`, `TokenRates`) — usage normalization to the §2 invariant and the single cost formula (SPEC §11).
|
|
56
56
|
- `ProviderError`, `classifyProviderError`, `toProviderError`, `providerSource` (+ `TelemetryEvent`, `ProviderTelemetryKind`) — the TelemetryEvent envelope for transport failures (SPEC §12).
|
|
57
57
|
- `tokenizerFor`, `tokenizerByPublisher`, `parseTokenizerFamily` (+ `TokenizerFamily`, `CountTokens`) — synchronous tokenizer strategies.
|
package/SPEC.md
CHANGED
|
@@ -23,7 +23,7 @@ Collision on `(kind: "provider", name)` at discovery: fail-hard.
|
|
|
23
23
|
```ts
|
|
24
24
|
interface Provider {
|
|
25
25
|
// Identity (immutable across lifetime)
|
|
26
|
-
readonly
|
|
26
|
+
readonly contextWindow: number | null; // context tokens, null if unresolved.
|
|
27
27
|
// PER SLOT under llama-server --parallel N
|
|
28
28
|
// (the server splits --ctx-size and reports
|
|
29
29
|
// the divided value; verified live).
|
|
@@ -54,7 +54,7 @@ interface Provider {
|
|
|
54
54
|
readonly constrainsOutput?: boolean;
|
|
55
55
|
readonly requiresMaxTokens?: boolean;
|
|
56
56
|
|
|
57
|
-
// Transport. `
|
|
57
|
+
// Transport. `workerId` is REQUIRED: the opaque, stable identity of the
|
|
58
58
|
// consumer's work stream — providers may key backend affinity on it and
|
|
59
59
|
// never interpret it. `grammar` is an optional GBNF string for
|
|
60
60
|
// grammar-constrained sampling (§13) — attached verbatim by capable
|
|
@@ -67,25 +67,31 @@ interface Provider {
|
|
|
67
67
|
// `sampling` is an optional bag of standard OpenAI-compat sampling params
|
|
68
68
|
// (temperature, top_p, top_k, min_p, penalties, stop, seed, …) merged into the
|
|
69
69
|
// body UNDER the managed fields — model/messages/grammar/reasoning/max_tokens/
|
|
70
|
-
// slot always win, and
|
|
71
|
-
// grammar, id_slot)
|
|
72
|
-
//
|
|
70
|
+
// slot always win, and reserved keys are stripped (#477): transport/protocol
|
|
71
|
+
// (stream, response_format, grammar, id_slot, logprobs), paradigm breakers
|
|
72
|
+
// (n, the tools/functions family, modalities/audio, prediction), and the
|
|
73
|
+
// token caps (max_tokens/max_completion_tokens -- the envelope is the managed
|
|
74
|
+
// maxTokens, never bypassable). It carries sampling intent + platform knobs
|
|
75
|
+
// only (§8). For a PROXY consumer forwarding its own
|
|
73
76
|
// caller's sampling knobs (the plurnk endpoint fronting gemma/Fireworks); a
|
|
74
77
|
// direct consumer leaves it unset.
|
|
75
|
-
// `strikes` is the
|
|
78
|
+
// `strikes` is the worker's CURRENT rail-strike streak at time-of-generate
|
|
76
79
|
// (0 = clean, distinct from absent = unreported; contract plurnk-service#313).
|
|
77
80
|
// Forwarded as `Plurnk-Strikes` ONLY under the firstPartyMetadata gate,
|
|
78
81
|
// dropped everywhere else. Headers only — never placed in the packet.
|
|
79
|
-
// `
|
|
80
|
-
// `Plurnk-
|
|
82
|
+
// `workspaceId`/`loop`/`turn` (#404): the turn coordinate, stamped as
|
|
83
|
+
// `Plurnk-Workspace-Id`/`Plurnk-Loop`/`Plurnk-Turn` under the SAME gate.
|
|
81
84
|
// 1-based; absent/0 emits no header. Headers only, never the packet.
|
|
82
|
-
generate(args: { messages: ChatMessage[];
|
|
85
|
+
generate(args: { messages: ChatMessage[]; workerId: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse>;
|
|
83
86
|
}
|
|
84
87
|
|
|
85
88
|
interface ProviderResponse {
|
|
86
89
|
assistant: {
|
|
87
90
|
content: string; // raw model emission; consumer parses
|
|
88
|
-
reasoning: string | null; // wire-reported
|
|
91
|
+
reasoning: string | null; // wire-reported reasoning content; null if absent
|
|
92
|
+
// sealed relay reasoning (#482): { data, format } blobs verbatim, never
|
|
93
|
+
// decoded; absent when none. agui projects REASONING_ENCRYPTED_VALUE.
|
|
94
|
+
reasoningEncrypted?: Array<{ data: string; format: string | null }>;
|
|
89
95
|
usage: ProviderUsage; // { prompt, completion, reasoning, cached, total }
|
|
90
96
|
finishReason: "stop" | "length" | "tool_calls" | "content_filter" | null;
|
|
91
97
|
model: string; // wire-reported (may differ from requested for relay providers)
|
|
@@ -97,13 +103,13 @@ interface ProviderResponse {
|
|
|
97
103
|
interface ProviderUsage {
|
|
98
104
|
prompt: number; // input tokens (cached ones included)
|
|
99
105
|
completion: number; // visible output tokens, EXCLUDING reasoning
|
|
100
|
-
reasoning: number; // reasoning
|
|
106
|
+
reasoning: number; // reasoning tokens, billed as output
|
|
101
107
|
cached: number; // subset of prompt served from cache
|
|
102
108
|
total: number; // prompt + completion + reasoning
|
|
103
109
|
}
|
|
104
110
|
```
|
|
105
111
|
|
|
106
|
-
Usage invariant: `total = prompt + completion + reasoning`; `cached ⊆ prompt`; `completion` excludes reasoning; **billable output = `completion + reasoning`**. Providers report reasoning
|
|
112
|
+
Usage invariant: `total = prompt + completion + reasoning`; `cached ⊆ prompt`; `completion` excludes reasoning; **billable output = `completion + reasoning`**. Providers report reasoning THREE ways: inside `completion_tokens` (OpenAI, via `completion_tokens_details.reasoning_tokens`), only as the `total - prompt - completion` gap (Gemini), or folded into `completion_tokens` with NO itemization while shipping the reasoning as TEXT (Fireworks, #425). The framework's `normalizeUsage` (§11) collapses all three to this invariant -- re-splitting the Fireworks case by the emitted text lengths, sum-preserving so cost is unchanged -- so siblings on `OpenAICompatProvider` get it for free.
|
|
107
113
|
|
|
108
114
|
### Promises
|
|
109
115
|
|
|
@@ -114,11 +120,11 @@ Usage invariant: `total = prompt + completion + reasoning`; `cached ⊆ prompt`;
|
|
|
114
120
|
- `countTokens` is **synchronous**, returns a non-negative integer, deterministic for the same input. Without an exact tokenizer family configured it is the **chars/2 UPPER BOUND** — deliberately conservative (real agentic text measures ~2.9–3.2 chars/token on gemma/deepseek, so the former chars/4 silently UNDERcounted 20–27%; a fallback may overcount, never under) — and it is **surfaced at construction** (`process.emitWarning`, code `PLURNK_TOKENIZER_HEURISTIC`), never silent. Exact counting is the tokenizer seam's job (mimetypes family), fed by `tokenize()` where available.
|
|
115
121
|
- `tokenize?` is an **optional async capability**: token ids in the model's real vocabulary, served by the backend itself (llama-server's native root `/tokenize`, surfaced when the §11 probe fingerprints a llama-server and `detectLlamaServer` isn't false). `tokenize === undefined` is the honest "backend can't" signal. Exact-counting consumers prefer it over any client-side tokenizer data — the local model's own vocab needs no bundled `tokenizer.json` at all.
|
|
116
122
|
- `costFor` is **pure**, returns pico-USD non-negative integer. Returns `0` for siblings with no known rates (local Ollama, generic OpenAI-compat shims).
|
|
117
|
-
- `
|
|
123
|
+
- `contextWindow` resolves to `null` when a PROBING provider (openai/llama-server) can't determine the window (consumer treats null as "no budget info"); a CLOUD provider with no window source FAILS HARD instead (#419, §11).
|
|
118
124
|
- `generate` rejects on signal abort — does NOT resolve with partial content.
|
|
119
125
|
- `generate` transports `grammar` verbatim when the backend supports grammar-constrained sampling, and silently ignores it otherwise (§13). The provider never chooses or modifies the grammar.
|
|
120
126
|
- `generate` **returns for every completed exchange — bytes always present, the conformance verdict attached as an observation.** When a grammar was transported (or validated in filter mode), the returned `content` is checked against it; a non-accept verdict rides `response.telemetry` as a `grammar_unenforced` event (message + divergence `position`) and the response returns normally. The provider transports and observes; it never adjudicates — discard, retry, escalate, or feed-back is consumer policy. This is a grammar-**conformance** check against the grammar the provider already holds — *not* a plurnk-DSL parse (that stays consumer-side, below) — so it remains backend- and DSL-agnostic. `ProviderError` remains reserved for exchanges that did NOT complete (transport failure, abort, boundary violations).
|
|
121
|
-
- **Backend affinity is the provider's internal guarantee, keyed by `
|
|
127
|
+
- **Backend affinity is the provider's internal guarantee, keyed by `workerId`.** The consumer says *which run this is*, never *which backend resource serves it* — raw resource identifiers (slot integers, connections) never cross the contract in either direction. On slot-pinning backends (llama-server `--parallel N>1`), the provider keeps each worker sticky to one slot and spreads distinct runs across slots, so each concurrent run keeps its KV-cache prefix warm (un-pinned routing is the server's similarity heuristic — slot hops re-pay full prefills). Backends without affinity semantics ignore `workerId` entirely.
|
|
122
128
|
|
|
123
129
|
## §3 `fromEnv(env, model, options?)` factory
|
|
124
130
|
|
|
@@ -129,7 +135,7 @@ class OpenAI {
|
|
|
129
135
|
static fromEnv(env: NodeJS.ProcessEnv, model: string, options?: ProviderOptions): OpenAI | Promise<OpenAI> {
|
|
130
136
|
// Read provider-specific env (OPENAI_BASE_URL, OPENAI_API_KEY, ...)
|
|
131
137
|
// plus universal operator knobs (PLURNK_PROVIDERS_REASONING, PLURNK_PROVIDERS_FETCH_TIMEOUT,
|
|
132
|
-
//
|
|
138
|
+
// PLURNK_PROVIDERS_CONTEXT_WINDOW). `options.baseUrl`, when set, is the per-alias
|
|
133
139
|
// endpoint override (PLURNK_BASEURL_<alias>, §5) and wins over the env base URL.
|
|
134
140
|
return new OpenAI({ /* ... */ });
|
|
135
141
|
}
|
|
@@ -149,10 +155,11 @@ Each provider's `fromEnv` reads these:
|
|
|
149
155
|
|
|
150
156
|
- **`PLURNK_PROVIDERS_REASONING`** — REQUIRED, one of `off | adaptive | on` (#33). Pure ACTIVATION intent; **`PLURNK_PROVIDERS_REASONING_BUDGET`** (positive int, REQUIRED iff `on`) carries the magnitude separately, so a number can never silently flip wire flags. The provider maps intent to the active backend's mechanism — llama-server `chat_template_kwargs: { enable_thinking }` (always emitted; the explicit FALSE is the only working off-switch; the NUMERIC clamp is the box's `--reasoning-budget` LAUNCH flag — per-request numerics are ignored, so env budget and launch flag are the same number, changed together), Ollama `think`, relay `include_reasoning`, cloud `reasoning_effort` (tier from budget when `on`; `off`/`adaptive` omit), fireworks `reasoning_effort` **explicit** (`effort_explicit`: `off` SENDS `"none"` — omission is NOT off on reason-by-default models like DeepSeek V4, #30; `adaptive` OMITS the field — the backend default IS adaptive, and the literal `"adaptive"` is MiniMax-M3-only, 400 elsewhere, #403. Intent maps **identically under a transported grammar** (the former #32 clamp-to-`"none"` is lifted — it caused plan-less turns, service#331; the lift enables best-effort reasoning). BUT reasoning and hard-masked GBNF rails do **NOT reliably coexist** on fireworks (TUNING-EPIC **F9-final**, ~60 live tests). Fireworks' docs: `response_format` (which includes grammar mode) **disables reasoning output**; measured, reasoning reaches the usable `reasoning_content` channel only stochastically (~1/8) and otherwise collapses or leaks into `content` — best lever (prefill `<think>\n`) tops out at 7/10 with a runaway tail. So under a transported grammar, reasoning is **best-effort and unsegregated** (lands in `content`, not `reasoning_content`), and NOT tunable — a consumer wanting the reasoning parses the preplan region of `content`. Fireworks is rails-first; single-call reasoning+rails needs a llama.cpp lazy-grammar host. Envelope: `max_tokens` is a cost circuit-breaker (reasoning-on can run away), not a reasoning reserve), Anthropic `thinking: disabled | enabled+budget_tokens(budget) | omit`. Consumers state intent, never mechanism. (In-DSL `<<PLAN>>` reasoning has **no provider footprint** and no operator knob: `PLAN` is a required element of the canonical plurnk grammar the consumer attaches via `generate`'s `grammar`. The provider transports that grammar verbatim and never forces, prefills, or toggles `PLAN`.)
|
|
151
157
|
|
|
152
|
-
Read via `
|
|
158
|
+
Read via `reasoningFromEnv` and **fail hard when unset**: the budget is required only when `on` and carries no floor default. Configuration lives in the operator's env over the package's `.env.defaults` floor (which declares every var and ships its default); the framework never bakes a knob default into code.
|
|
153
159
|
- **`PLURNK_PROVIDERS_FETCH_TIMEOUT`** — service-wide ms ceiling on any single outbound request (**per attempt**, not shared across retries). Each `fromEnv` reads and passes as `AbortSignal.timeout`. Per-provider override envs are NOT part of the contract.
|
|
154
160
|
- **`PLURNK_PROVIDERS_RETRY_ATTEMPTS`** — REQUIRED non-negative integer (read via `parseRequiredInt`). The transient-failure retry budget: **`0`** surfaces the first failure; **`N`** retries up to `N` times on a *transient* classification only (`rate_limit` / `network_failure` — 429, 5xx, timeout, connection reset). Terminal kinds (`unauthorized`, `quota_exceeded`, `invalid_response`, `model_refused`) are never retried. Backoff is exponential from a `2000ms` base (`base * 2^(attempt-1)`), unless the server sent a `Retry-After` (which wins). The caller's `signal` aborts both the in-flight request and the backoff sleep. Lives in the shared `OpenAICompatProvider` so every provider inherits it uniformly; rides on the existing `classifyProviderError` (#18).
|
|
155
|
-
- **`
|
|
161
|
+
- **`PLURNK_PROVIDERS_CONTEXT_WINDOW`** -- optional positive-integer override (alias-scopable) for the model's context window. Resolution (#419): this env var -> endpoint `n_ctx` probe (probing specs only) -> `@plurnk/plurnk-models` catalog -> then a PROBING provider degrades to `null`, a CLOUD provider (no probe) FAILS HARD (an uncataloged, unpinned cloud model is a config error, not a guessable window). See §11.
|
|
162
|
+
- **`PLURNK_PROVIDERS_TEMPERATURE` / `PLURNK_PROVIDERS_REPEAT_PENALTY` / `PLURNK_PROVIDERS_FREQUENCY_PENALTY`** -- REQUIRED sampling + anti-degeneration floors (read via `parseRequiredFloat`, values from the `.env.defaults` floor). `REPEAT_PENALTY` (canonical `1.15`) is the llama.cpp/Fireworks multiplier; `FREQUENCY_PENALTY` (canonical `0.4`, #426) is the OpenAI-standard guard on the plain cloud path, which has no `repeat_penalty`. Applied to EVERY request, keyed per backend (§13).
|
|
156
163
|
|
|
157
164
|
## §5 Alias cascade resolution
|
|
158
165
|
|
|
@@ -190,18 +197,18 @@ This package's exported resolution surface:
|
|
|
190
197
|
|
|
191
198
|
**Two-tier provider resolution — owned entirely by this package.** A provider name resolves in this order:
|
|
192
199
|
|
|
193
|
-
1. **Standard provider** (`§11`) — if `isStandardProvider(name)`, instantiated directly via `standardProviderFromEnv(name, env, model)`. No package is imported. Covers every plain OpenAI-compatible endpoint (`openai`, `groq`, `deepseek`, `mistral`, `together`, `fireworks`, `deepinfra`, `anthropic`, `bedrock`, `plurnk`, …) — including first-party Claude (Anthropic's compat endpoint, bearer auth, the `thinking` reasoning param), AWS Bedrock (its compat endpoint at the `/openai/v1` path, Bedrock-API-key bearer; the base is `BEDROCK_BASE_URL` if set, else derived as `https://bedrock-runtime.{region}.amazonaws.com/openai/v1` from the standard `AWS_REGION`/`AWS_DEFAULT_REGION`; its inference-profile model ids resolve a catalog **context window** by stripping the region and looking the model up under its publisher (#22), while cost stays unknown — bedrock marks up over the native rate), and the **`plurnk` hosted model** (a plain remote OpenAI endpoint at its `PLURNK_BASE_URL` base, `.env.
|
|
200
|
+
1. **Standard provider** (`§11`) — if `isStandardProvider(name)`, instantiated directly via `standardProviderFromEnv(name, env, model)`. No package is imported. Covers every plain OpenAI-compatible endpoint (`openai`, `groq`, `deepseek`, `mistral`, `together`, `fireworks`, `deepinfra`, `anthropic`, `bedrock`, `plurnk`, …) — including first-party Claude (Anthropic's compat endpoint, bearer auth, the `thinking` reasoning param), AWS Bedrock (its compat endpoint at the `/openai/v1` path, Bedrock-API-key bearer; the base is `BEDROCK_BASE_URL` if set, else derived as `https://bedrock-runtime.{region}.amazonaws.com/openai/v1` from the standard `AWS_REGION`/`AWS_DEFAULT_REGION`; its inference-profile model ids resolve a catalog **context window** by stripping the region and looking the model up under its publisher (#22), while cost stays unknown — bedrock marks up over the native rate), and the **`plurnk` hosted model** (a plain remote OpenAI endpoint at its `PLURNK_BASE_URL` base, `.env.defaults` → `https://plurnk.ai/v1`; reads its server-controlled context window from upstream, sends no grammar and no tuning); none needs a daughter. `plurnk` authenticates with a single optional bearer (`PLURNK_API_KEY`, sent only when set), like any other standard bearer. A spec's `apiKeyVar`/`baseUrlVar` may be a **list** of accepted env-var aliases — the conventional names the wild uses for one credential/base (e.g. `deepinfra` → `DEEPINFRA_API_KEY` / `DEEPINFRA_API_TOKEN` / `DEEPINFRA_TOKEN`; `openai` base → `OPENAI_BASE_URL` / `OPENAI_API_BASE`). First set non-empty wins; a required key unset across *all* aliases fails hard naming each. This is alias resolution over operator-set values, never a fabricated default. (An entry MAY instead supply a custom `headersFromEnv` builder for auth a bearer can't express — multi-header/credential schemes — or a `baseUrlFromEnv` builder to template the base from env, as bedrock does.) The standard table is **authoritative**: a scanned package whose name duplicates a standard one is shadowed (tier 1 returns first).
|
|
194
201
|
2. **Discovered package** — otherwise, a **scope-agnostic `node_modules` scan** (`discover()`) maps the provider name to the installed package that declares `plurnk: { kind: "provider", name }`, which is dynamic-imported and `fromEnv(env, model, options?)`-called. Covers first-party daughters with real runtime surface (`openrouter`, `ollama`, `google`, `xai`, `cloudflare`, the planned `vertex`/`cohere`) **and any third-party provider published under its own scope** (`@acme/llm-provider-foo`) — no involvement from us, no `@plurnk` scope assumption. Two installed packages claiming the same name → fail-hard naming both. Unknown name → fail-hard. The scan runs once per process and is memoized.
|
|
195
202
|
|
|
196
203
|
Discovery honors the host **trust gate** `PLURNK_PLUGINS_TRUSTED_ONLY` — the same env var the execs/mimes/schemes families read (plurnk-service#229, #15). OFF (unset/empty/`0`) trusts every installed provider; ON (any value) trusts `@plurnk/*` plus a comma-separated allowlist and declines the rest. A declined package is recorded in `skipped` (never registered, never thrown), so requesting its name yields a precise *untrusted* error rather than *unknown*.
|
|
197
204
|
|
|
198
|
-
The framework is **contract-only**
|
|
205
|
+
The framework is **contract-only** -- it does **not** depend on its daughters. First-party daughters install flat (via the [`@plurnk/plurnk-providers-all`](https://github.com/plurnk/plurnk-providers-all) aggregator) so npm hoists them where the scan finds them; a framework that instead aggregated its own daughters would nest and hide them from the scan (the execs #12/#14 lesson). Each daughter declares the framework as a `peerDependency` (`^1`) so one shared copy lives in the tree -- class identity for `instanceof ProviderError` included -- without a framework minor forcing a daughter republish.
|
|
199
206
|
|
|
200
207
|
## §6 Engine → provider guarantees (consumer side)
|
|
201
208
|
|
|
202
209
|
- `messages` is a complete prompt. Consumer has pre-assembled all sections. Provider does not add, reorder, or inject turns — the wire `messages` are exactly what the consumer passed. (The provider injects no `PLAN` turn — `PLAN` is part of the consumer's grammar contract, §4.)
|
|
203
|
-
- Every `generate` carries `
|
|
204
|
-
- `signal` is wired to the
|
|
210
|
+
- Every `generate` carries `workerId` — the worker's stable, opaque identity. Same run → same string across its turns; distinct runs → distinct strings.
|
|
211
|
+
- `signal` is wired to the worker's AbortController.
|
|
205
212
|
- `generate` is single-call per turn. No parallel calls on the same instance.
|
|
206
213
|
- `assistantRaw` is opaque to the consumer (forensics-only).
|
|
207
214
|
- `meta` is the per-turn provider→client metadata bag: the backend's **non-standard top-level response fields passed through verbatim** (every provider), PLUS **validated known keys** the framework holds a contract for — currently `balancePico` (a finite pico-USD number normalized from the plurnk endpoint's balance field, renamed off its raw key; dropped if non-numeric). Absent when the backend reported no extras. The consumer (service) merges `meta` into its Turn metadata and filters what reaches the client; it reads `meta`, never mines `assistantRaw` (#23).
|
|
@@ -243,7 +250,7 @@ The framework is **contract-only** — it does **not** depend on its daughters.
|
|
|
243
250
|
import { Mock } from "@plurnk/plurnk-providers";
|
|
244
251
|
|
|
245
252
|
const mock = new Mock({
|
|
246
|
-
|
|
253
|
+
contextWindow: 100000,
|
|
247
254
|
responses: [{ assistant: { content: "<<SEND[200]:hi:SEND", reasoning: null } }],
|
|
248
255
|
});
|
|
249
256
|
const result = await mock.generate({ messages: [] });
|
|
@@ -256,7 +263,7 @@ const result = await mock.generate({ messages: [] });
|
|
|
256
263
|
A sibling package satisfies the contract when:
|
|
257
264
|
|
|
258
265
|
1. Default export is a class with `static fromEnv(env, model, options?)` factory.
|
|
259
|
-
2. Instance exposes `
|
|
266
|
+
2. Instance exposes `contextWindow: number | null` and `model: string` (non-empty).
|
|
260
267
|
3. Instance exposes `countTokens(text): number` and `costFor(usage): number`.
|
|
261
268
|
4. `countTokens("")` returns `0`; `countTokens("…")` returns a non-negative integer.
|
|
262
269
|
5. `costFor({prompt:0,completion:0,reasoning:0,cached:0,total:0})` returns `0` (or non-negative pico-USD for non-free models).
|
|
@@ -276,39 +283,40 @@ Sibling-specific behavioral tests (wire-format compliance, model-family quirks,
|
|
|
276
283
|
|
|
277
284
|
The framework ships the transport spine every OpenAI-compatible provider had been duplicating. Build a sibling *on top of these* — don't re-implement them.
|
|
278
285
|
|
|
279
|
-
- **`OpenAICompatProvider`** — a `Provider` implementation built by composition. Its `generate` does the universal work (merge `signal` with a `PLURNK_PROVIDERS_FETCH_TIMEOUT` deadline, stream the completion, map `usage`, normalize `finishReason` to the §2 set, assemble the response). Per-provider deltas arrive as config:
|
|
286
|
+
- **`OpenAICompatProvider`** — a `Provider` implementation built by composition. Its `generate` does the universal work (merge `signal` with a `PLURNK_PROVIDERS_FETCH_TIMEOUT` deadline, stream the completion, map `usage`, normalize `finishReason` to the §2 set (translating known per-backend cap synonyms -- `max_tokens`, `MAX_TOKENS`, ... -> `length` -- so a consumer's `=== "length"` holds across backends, warning once on any unmapped value, #425), assemble the response). Per-provider deltas arrive as config:
|
|
280
287
|
|
|
281
288
|
```ts
|
|
282
289
|
new OpenAICompatProvider({
|
|
283
290
|
model, url, // fully-resolved chat-completions URL
|
|
284
291
|
fetchTimeoutMs,
|
|
285
292
|
headers, // fully-resolved request headers (incl. auth)
|
|
286
|
-
|
|
287
|
-
|
|
293
|
+
contextWindow, // number | null
|
|
294
|
+
reasoning, reasoningStyle, // {mode,budget} intent + style: "none"|"think"|"include_reasoning"|"effort"|"effort_explicit"|"template"|"anthropic"
|
|
295
|
+
temperature, repeatPenalty, frequencyPenalty, // sampling + anti-degeneration floor; frequency_penalty guards the plain cloud path (#426)
|
|
288
296
|
countTokens, costFor, // strategies; default heuristic / free
|
|
289
297
|
grammarStyle, // "none" | "llamacpp" | "response_format" — GBNF wire shape (§13); default "none"
|
|
290
298
|
gbnfDebug, // PLURNK_PROVIDERS_GBNF_DEBUG: validate a grammar locally + throw on invalid, but DON'T send it (§13); default false
|
|
291
299
|
streaming, // SSE transport; default true (false → one non-streamed JSON)
|
|
292
300
|
supportsSlotPinning, slotCount, // INTERNAL slot-affinity wiring (run→id_slot); never consumer-facing
|
|
293
|
-
|
|
301
|
+
topLogprobs, rawBody, // #36 opt-in data capture (PLURNK_PROVIDERS_TOP_LOGPROBS / _RAWBODY); default off
|
|
294
302
|
servedModel, // #37 backend's self-reported served id (from the probe) → Provider.servedModel
|
|
295
303
|
});
|
|
296
304
|
```
|
|
297
305
|
|
|
298
|
-
The `openai` standard provider sets `grammarStyle: "llamacpp"`, `supportsSlotPinning`, and `slotCount` from the same llama-server fingerprint (`/v1/models` `meta` block + `/props`). The
|
|
306
|
+
The `openai` standard provider sets `grammarStyle: "llamacpp"`, `supportsSlotPinning`, and `slotCount` from the same llama-server fingerprint (`/v1/models` `meta` block + `/props`). The worker→slot mapping lives inside `OpenAICompatProvider`: sticky per `workerId`, round-robin across new runs, LRU-bounded.
|
|
299
307
|
|
|
300
308
|
- **`chatCompletionStream` / `chatCompletion` / `OpenAiHttpError` / `StreamResponse`** — the shared HTTP client (`chatCompletionStream` for SSE, `chatCompletion` for the non-streamed JSON the `streaming: false` path uses). One shared copy.
|
|
301
|
-
- **`normalizeUsage(raw)` / `computeCost(usage, {input, output, cached})`** — usage normalization to the §2 invariant (handles
|
|
309
|
+
- **`normalizeUsage(raw, reasoningText?, contentText?)` / `computeCost(usage, {input, output, cached})`** — usage normalization to the §2 invariant (handles all three reasoning-reporting conventions; the optional text args feed the Fireworks re-split, #425) and the single cost formula (bills `completion + reasoning` at the output rate). `OpenAICompatProvider` applies `normalizeUsage` automatically; siblings pass their per-token rates to `computeCost` in their `costFor`.
|
|
302
310
|
- **`parseRequiredInt` / `parseOptionalInt` / `requireEnv`** — env helpers; each takes a provider `label` for error prefixing.
|
|
303
|
-
- **`effortFromBudget(budget)`** — the shared
|
|
311
|
+
- **`effortFromBudget(budget)`** — the shared reasoning-budget → `low|medium|high` breakpoints.
|
|
304
312
|
|
|
305
|
-
A **bespoke sibling** therefore reduces to a thin class whose `fromEnv` probes whatever it needs (model catalog, pricing, context window), builds the config, and returns `new OpenAICompatProvider(config)`. A **standard provider** (§5 tier 1) needs no sibling at all — it's a frozen entry in `STANDARD_PROVIDERS` describing its key var, base-URL var, reasoning style, and tokenizer; `standardProviderFromEnv(name, env, model)` (async — returns `Promise<Provider | null>`) does the rest. The endpoint
|
|
313
|
+
A **bespoke sibling** therefore reduces to a thin class whose `fromEnv` probes whatever it needs (model catalog, pricing, context window), builds the config, and returns `new OpenAICompatProvider(config)`. A **standard provider** (§5 tier 1) needs no sibling at all — it's a frozen entry in `STANDARD_PROVIDERS` describing its key var, base-URL var, reasoning style, and tokenizer; `standardProviderFromEnv(name, env, model)` (async — returns `Promise<Provider | null>`) does the rest. The endpoint's **canonical URL ships as a floored default** in `.env.defaults` (set-if-unset, overridable in the operator's env or per-alias); it is read from the base-URL var (or a `baseUrlFromEnv` deriver) with **no in-code default**, the value living in the shipped floor, never baked into the table. Only the API **key** is required operator config (a secret with no default; fail-hard when unset).
|
|
306
314
|
|
|
307
|
-
The `plurnk` entry alone sets **`firstPartyMetadata: true`** — it forwards the consumer's per-turn `generate()` `attributions` (which installed plugin packages dispatched) and `client` (the originating frontend, e.g. `plurnk.nvim/1.4.0`) as `Plurnk-Attribution` / `Plurnk-Client` headers, and the opaque `
|
|
315
|
+
The `plurnk` entry alone sets **`firstPartyMetadata: true`** — it forwards the consumer's per-turn `generate()` `attributions` (which installed plugin packages dispatched) and `client` (the originating frontend, e.g. `plurnk.nvim/1.4.0`) as `Plurnk-Attribution` / `Plurnk-Client` headers, and the opaque `workerId` as `Plurnk-Run-Id` (#26). The gate lives on the provider, not the call site, so these first-party signals are **structurally incapable** of reaching a third-party backend. Empty values emit no header. It also alone sets **`balanceMetaKey: "balance_pico"`** — the top-level response field the plurnk endpoint reports the running account balance (pico-USD) in. `OpenAICompatProvider` surfaces the backend's extra top-level fields as `ProviderResponse.meta` for **every** provider (passed through), and for a provider that set `balanceMetaKey` it additionally normalizes that field into a validated `meta.balancePico`. So `balancePico` appears only from plurnk; the raw pass-through `meta` is general (#23).
|
|
308
316
|
|
|
309
317
|
A spec may carry a **`modelPrefix`** — a constant model-id segment the backend requires but the operator's alias shouldn't repeat. `fireworks` sets `"accounts/fireworks/models/"`, so `PLURNK_MODEL_fast=fireworks/deepseek-v4-pro` carries only the distinctive tail; `standardProviderFromEnv` prepends it idempotently (an already-prefixed id is untouched) to form the wire id, which is **also** the catalog key (models.dev keys fireworks-ai on the full id). Specs without a `modelPrefix` use the model string verbatim.
|
|
310
318
|
|
|
311
|
-
`
|
|
319
|
+
`contextWindow` for a standard provider resolves (#419): `PLURNK_PROVIDERS_CONTEXT_WINDOW` -> endpoint `n_ctx` (for `probeNctx`-flagged specs like `openai`, queried from `GET /v1/models`: llama-server reports its loaded window at `data[].meta.n_ctx`, vLLM top-level; cloud endpoints don't) -> the `@plurnk/plurnk-models` catalog -> **then the hybrid: a PROBING provider degrades to `null`, a CLOUD provider (no probe) FAILS HARD** (uncataloged + unpinned = config error, the #417 kimi case, not a guessed window). The same probe fingerprints llama-server (the `meta` block) to enable grammar transport (§13), and reads the row's `id` as `servedModel` (#37) — the real served name behind a local alias — so it runs even when the env var pins the window. The probe is best-effort: any failure resolves to `null` context / no grammar capability (a legitimate "unknown"), never throws. For a PROBING provider, an underivable window (env, probe, and catalog ALL missed) is surfaced once via a **`PLURNK_CONTEXT_UNKNOWN`** warning naming the model and the remediation (`PLURNK_PROVIDERS_CONTEXT_WINDOW`, alias-scopable) -- null stays legitimate but never silent (a CLOUD provider throws here instead, above). Operator-facing warnings (`PLURNK_TOKENIZER_HEURISTIC`, `PLURNK_PROBE_FAILED`, `PLURNK_GRAMMAR_UNVERIFIABLE`, `PLURNK_CONTEXT_UNKNOWN`, `PLURNK_FINISH_REASON_UNKNOWN`) are deduplicated **once per process per (code, message)** (#40) — repeat constructions don't re-fire them, but a *different* provider/model's first surfacing is never suppressed.
|
|
312
320
|
|
|
313
321
|
## §12 Telemetry — provider failures
|
|
314
322
|
|
|
@@ -333,11 +341,13 @@ The `TelemetryEvent` shape is mirrored **locally** (`./telemetry.ts`), structura
|
|
|
333
341
|
- **This layer** owns capability detection, transport, **and enforcement verification**: `generate({ …, grammar })` attaches the string **verbatim** as the `grammar` body field when the backend supports it, sends no grammar-related field otherwise (cloud APIs reject unknown params), and — when it did transport a grammar — checks that the response actually conforms. The provider never chooses or modifies the grammar.
|
|
334
342
|
- **The consumer** owns policy: whether to constrain a given call, and which root variant to send (e.g. the `root ::= statement` single-statement substitution that forces EOS at the close tag — the shipped `statement+` root never forces EOS, so greedy generation runs to `max_tokens`).
|
|
335
343
|
|
|
336
|
-
**Sampling guard.**
|
|
344
|
+
**Sampling guard (anti-degeneration, #426).** Degeneration into repetition loops is NOT grammar-specific -- greedy/low-temperature decoding loops on the plain cloud path too -- so the penalty rides **every** request, not just grammar'd ones, and never relies on server launch flags. The wire field is keyed on the backend (`grammarStyle`): `repeat_penalty` on `llamacpp` (the managed multiplier floor, `PLURNK_PROVIDERS_REPEAT_PENALTY`, canonical `1.15`), `repetition_penalty` on `response_format`/Fireworks (same floor, the OpenAI-compat spelling, verified honored #20), and `frequency_penalty` on `none`/plain cloud (`PLURNK_PROVIDERS_FREQUENCY_PENALTY`, canonical `0.4` -- the OpenAI-standard field, since the llama.cpp `repeat_penalty` multiplier isn't accepted there). A hard grammar makes the loop WORSE (masking removes the model's natural exits), which is why the floor was born on the grammar path; #426 promoted it to the universal guard it always needed to be. (Probed live on llama.cpp b894 + gemma-4-26B; reference: plurnk-grammar `test/llama/gbnf-live.test.ts`.)
|
|
337
345
|
|
|
338
346
|
**The cap is the consumer's required guard.** The repeat-penalty floor suppresses short repetition cycles, NOT long-cycle degeneration: under the multi-op root (optional EOS) at near-greedy temperatures, a constrained emission can answer correctly in its first tokens and then loop to the **context wall** (observed live: 30,736 junk tokens to `finish_reason: length`, minutes of decode reading as a "hang", with the junk echoed into the next turn's prompt — providers#10). No layer defaults a cap: the wire default is unbounded (`n_predict: -1`) and the provider transports policy, never invents it. A consumer enabling constrained sampling MUST pass `maxTokens` (or send a root variant that forces EOS).
|
|
339
347
|
|
|
340
|
-
**Native reasoning and the grammar
|
|
348
|
+
**Native reasoning and the grammar COEXIST on llama-server; the sanctioned think block is the protection, not the hazard (#488 postmortem).** With `enable_thinking: true`, the server auto-gates the grammar around the SANCTIONED reasoning block — the model thinks up front (routed to `reasoning_content`), then content decodes constrained. Verified: a 26-run baseline green under exactly this configuration, and ZERO grammar rejects across every #488 "railless" specimen (`@plurnk/gbnf` verdicts: accept or cap-truncated incomplete — the rail never left). The #488 rails-win-the-channel clamp (grammar forces `enable_thinking:false`) is REVERTED: closing the channel starves a reasoning-tuned model of its outlet, and it ESCAPES mid-content into the raw thought channel — which the server then discards while the decode runs unconstrained and billed (measured: 12,288 completion tokens billed, 1,033 chars visible, reasoning empty). So intent maps identically under a transported grammar, and the failure mode is SURFACED instead of traded against:
|
|
349
|
+
- **Per-request rail state on `meta` (#488):** every observed-grammar response carries `meta.railsAttached` (was the grammar transported) and `meta.railsVerdict` (`accept | incomplete | reject | unverifiable` from the conformance check) — the consumer's turn row then answers "did the rail ride, and did the output conform" PER TURN from the run db, so rail presence is never again inferred from output shape.
|
|
350
|
+
- **Channel-escape telemetry (#488):** billed completion tokens vastly exceeding every visible channel (`usage.completion > countTokens(content) + countTokens(reasoning) + 64`; countTokens overcounts text, so the excess is real vanishing) attaches a `grammar_unenforced` event naming the vanished balance — the run105 class (escape into a discarded reasoning block) is loud, not invisible. With the channel closed, the model reasons **inside the DSL**: the canonical grammar's required `PLAN` statement is a free-text body the model fills with genuine step-by-step reasoning before acting (probed live, b894+gemma: correct chain-of-thought inside `<<PLAN:…:PLAN`, then a clean `SEND`, `finish_reason: stop`). The `PLAN` element belongs to the consumer's grammar contract; the provider's only job is closing the native channel deterministically.
|
|
341
351
|
|
|
342
352
|
**Grammar transport — same GBNF, different wire shapes (`grammarStyle`, default `"none"`).** Backends carry the *same* grammar string differently:
|
|
343
353
|
- **`"llamacpp"`** — top-level `grammar` field + the repeat-penalty floor (`PLURNK_PROVIDERS_REPEAT_PENALTY`, §9). Detected from the §11 probe: only llama-server rows on `GET /v1/models` carry a `meta` block. This is any local llama-server (e.g. the generic `openai` provider fronting llama.cpp). `plurnk` opts out via `detectLlamaServer: false` — it reads a window but is treated as a plain OpenAI endpoint, never fingerprinted and never sent a grammar.
|
|
@@ -361,7 +371,7 @@ Zero grammar dependency (§11) is preserved: the GBNF string arrives per call; t
|
|
|
361
371
|
|
|
362
372
|
Two OPT-IN knobs surface the full signal of a paid turn for downstream IQ scoring and model distillation. Both are **OFF by default** and **per-alias-scopable** (`PLURNK_PROVIDERS_<KNOB>_<alias>`): the flag *is* the isolation, so a serving turn requests nothing on the wire and carries nothing on the response — only a dataset-scraping alias opts in. Universal: any provider (standard or daughter), any backend that returns the data.
|
|
363
373
|
|
|
364
|
-
**`
|
|
374
|
+
**`PLURNK_PROVIDERS_TOP_LOGPROBS`** (non-negative int = the OpenAI `top_logprobs`; unset = off). When set, `generate` requests `logprobs:true, top_logprobs:<n>` and surfaces `response.assistant.logprobs: Array<{ token, logprob, top? }>` plus `assistant.meanLogprob`. These are **managed fields** — reserved from caller `sampling`, so the env flag is the single control (a proxy consumer can't forge them). A backend that returns no logprobs yields an absent field — **never synthesized**.
|
|
365
375
|
|
|
366
376
|
**The `logprob` vs `sampling_logprob` decision (the honest-confidence call).** Fireworks returns both per token: `logprob` (raw model log-probability) and `sampling_logprob` (post-sampling-transform). The structured `logprob` we surface is the **raw** value — the sampling-transform-invariant measure of the model's native belief, the correct confidence signal AND distillation target. This was settled empirically, not by assumption: under grammar the two measured **identical to full float precision**, including an *adversarial* mask (grammar forcing a token the model assigned ~8%: `logprob` −2.5229365 == `sampling_logprob` −2.5229365). A post-mask renormalization would inflate confidence toward the constraint; the raw value stays honest. Anyone wanting `sampling_logprob` reads it from `rawBody`.
|
|
367
377
|
|
package/dist/Mock.d.ts
CHANGED
|
@@ -17,17 +17,17 @@ export type MockReturnedAssistant = ProviderAssistant & {
|
|
|
17
17
|
declare const DEFAULT_USAGE: ProviderUsage;
|
|
18
18
|
export default class Mock implements Provider {
|
|
19
19
|
#private;
|
|
20
|
-
constructor({
|
|
21
|
-
|
|
20
|
+
constructor({ contextWindow, responses }: {
|
|
21
|
+
contextWindow: number | null;
|
|
22
22
|
responses: MockResponse[];
|
|
23
23
|
});
|
|
24
|
-
get
|
|
24
|
+
get contextWindow(): number | null;
|
|
25
25
|
get model(): string;
|
|
26
26
|
countTokens(text: string): number;
|
|
27
27
|
costFor(_usage: ProviderUsage): number;
|
|
28
28
|
generate({ signal }: {
|
|
29
29
|
messages: ChatMessage[];
|
|
30
|
-
|
|
30
|
+
workerId?: string;
|
|
31
31
|
signal?: AbortSignal;
|
|
32
32
|
}): Promise<{
|
|
33
33
|
assistant: MockReturnedAssistant;
|
package/dist/Mock.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"Mock.d.ts","sourceRoot":"","sources":["../src/Mock.ts"],"names":[],"mappings":"AAOA,OAAO,KAAK,EAAE,WAAW,EAAE,YAAY,EAAE,QAAQ,EAAE,iBAAiB,EAAE,aAAa,EAAE,MAAM,YAAY,CAAC;AAExG,MAAM,MAAM,aAAa,GAAG;IACxB,OAAO,EAAE,MAAM,CAAC;IAChB,SAAS,EAAE,MAAM,GAAG,IAAI,CAAC;IAEzB,KAAK,CAAC,EAAE,OAAO,CAAC,aAAa,CAAC,CAAC;IAC/B,YAAY,CAAC,EAAE,YAAY,CAAC;IAC5B,KAAK,CAAC,EAAE,MAAM,CAAC;IAKf,GAAG,CAAC,EAAE,OAAO,EAAE,CAAC;CACnB,CAAC;AAEF,MAAM,MAAM,YAAY,GAAG;IACvB,SAAS,EAAE,aAAa,CAAC;IACzB,YAAY,CAAC,EAAE,OAAO,CAAC;CAC1B,CAAC;AAGF,MAAM,MAAM,qBAAqB,GAAG,iBAAiB,GAAG;IAAE,GAAG,CAAC,EAAE,OAAO,EAAE,CAAA;CAAE,CAAC;AAE5E,QAAA,MAAM,aAAa,EAAE,aAA+E,CAAC;AAErG,MAAM,CAAC,OAAO,OAAO,IAAK,YAAW,QAAQ;;IAIzC,YAAY,EAAE,
|
|
1
|
+
{"version":3,"file":"Mock.d.ts","sourceRoot":"","sources":["../src/Mock.ts"],"names":[],"mappings":"AAOA,OAAO,KAAK,EAAE,WAAW,EAAE,YAAY,EAAE,QAAQ,EAAE,iBAAiB,EAAE,aAAa,EAAE,MAAM,YAAY,CAAC;AAExG,MAAM,MAAM,aAAa,GAAG;IACxB,OAAO,EAAE,MAAM,CAAC;IAChB,SAAS,EAAE,MAAM,GAAG,IAAI,CAAC;IAEzB,KAAK,CAAC,EAAE,OAAO,CAAC,aAAa,CAAC,CAAC;IAC/B,YAAY,CAAC,EAAE,YAAY,CAAC;IAC5B,KAAK,CAAC,EAAE,MAAM,CAAC;IAKf,GAAG,CAAC,EAAE,OAAO,EAAE,CAAC;CACnB,CAAC;AAEF,MAAM,MAAM,YAAY,GAAG;IACvB,SAAS,EAAE,aAAa,CAAC;IACzB,YAAY,CAAC,EAAE,OAAO,CAAC;CAC1B,CAAC;AAGF,MAAM,MAAM,qBAAqB,GAAG,iBAAiB,GAAG;IAAE,GAAG,CAAC,EAAE,OAAO,EAAE,CAAA;CAAE,CAAC;AAE5E,QAAA,MAAM,aAAa,EAAE,aAA+E,CAAC;AAErG,MAAM,CAAC,OAAO,OAAO,IAAK,YAAW,QAAQ;;IAIzC,YAAY,EAAE,aAAa,EAAE,SAAS,EAAE,EAAE;QAAE,aAAa,EAAE,MAAM,GAAG,IAAI,CAAC;QAAC,SAAS,EAAE,YAAY,EAAE,CAAA;KAAE,EAGpG;IAED,IAAI,aAAa,IAAI,MAAM,GAAG,IAAI,CAAgC;IAClE,IAAI,KAAK,IAAI,MAAM,CAAmB;IAItC,WAAW,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAEhC;IAGD,OAAO,CAAC,MAAM,EAAE,aAAa,GAAG,MAAM,CAAc;IAE9C,QAAQ,CAAC,EAAE,MAAM,EAAE,EAAE;QAAE,QAAQ,EAAE,WAAW,EAAE,CAAC;QAAC,QAAQ,CAAC,EAAE,MAAM,CAAC;QAAC,MAAM,CAAC,EAAE,WAAW,CAAA;KAAE,GAAG,OAAO,CAAC;QAAE,SAAS,EAAE,qBAAqB,CAAC;QAAC,YAAY,EAAE,OAAO,CAAA;KAAE,CAAC,CAgBrK;IAED,IAAI,SAAS,IAAI,MAAM,CAA+B;CACzD;AAED,OAAO,EAAE,aAAa,IAAI,gBAAgB,EAAE,CAAC"}
|
package/dist/Mock.js
CHANGED
|
@@ -6,13 +6,13 @@
|
|
|
6
6
|
// hatch — that's an intg-only convenience.
|
|
7
7
|
const DEFAULT_USAGE = { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 };
|
|
8
8
|
export default class Mock {
|
|
9
|
-
#
|
|
9
|
+
#contextWindow;
|
|
10
10
|
#queue;
|
|
11
|
-
constructor({
|
|
12
|
-
this.#
|
|
11
|
+
constructor({ contextWindow, responses }) {
|
|
12
|
+
this.#contextWindow = contextWindow;
|
|
13
13
|
this.#queue = [...responses];
|
|
14
14
|
}
|
|
15
|
-
get
|
|
15
|
+
get contextWindow() { return this.#contextWindow; }
|
|
16
16
|
get model() { return "mock"; }
|
|
17
17
|
// Heuristic tokenizer (chars/2 upper bound, matching the framework's
|
|
18
18
|
// fallback). Mock is test-only; real provider siblings ship exact counts.
|
package/dist/Mock.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"Mock.js","sourceRoot":"","sources":["../src/Mock.ts"],"names":[],"mappings":"AAAA,2DAA2D;AAC3D,EAAE;AACF,wEAAwE;AACxE,wEAAwE;AACxE,wEAAwE;AACxE,2CAA2C;AA0B3C,MAAM,aAAa,GAAkB,EAAE,MAAM,EAAE,CAAC,EAAE,UAAU,EAAE,CAAC,EAAE,SAAS,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,KAAK,EAAE,CAAC,EAAE,CAAC;AAErG,MAAM,CAAC,OAAO,OAAO,IAAI;IACrB,
|
|
1
|
+
{"version":3,"file":"Mock.js","sourceRoot":"","sources":["../src/Mock.ts"],"names":[],"mappings":"AAAA,2DAA2D;AAC3D,EAAE;AACF,wEAAwE;AACxE,wEAAwE;AACxE,wEAAwE;AACxE,2CAA2C;AA0B3C,MAAM,aAAa,GAAkB,EAAE,MAAM,EAAE,CAAC,EAAE,UAAU,EAAE,CAAC,EAAE,SAAS,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,KAAK,EAAE,CAAC,EAAE,CAAC;AAErG,MAAM,CAAC,OAAO,OAAO,IAAI;IACrB,cAAc,CAAgB;IAC9B,MAAM,CAAiB;IAEvB,YAAY,EAAE,aAAa,EAAE,SAAS,EAA+D;QACjG,IAAI,CAAC,cAAc,GAAG,aAAa,CAAC;QACpC,IAAI,CAAC,MAAM,GAAG,CAAC,GAAG,SAAS,CAAC,CAAC;IACjC,CAAC;IAED,IAAI,aAAa,KAAoB,OAAO,IAAI,CAAC,cAAc,CAAC,CAAC,CAAC;IAClE,IAAI,KAAK,KAAa,OAAO,MAAM,CAAC,CAAC,CAAC;IAEtC,qEAAqE;IACrE,0EAA0E;IAC1E,WAAW,CAAC,IAAY;QACpB,OAAO,IAAI,CAAC,MAAM,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,CAAC,MAAM,GAAG,CAAC,CAAC,CAAC;IAC9D,CAAC;IAED,gBAAgB;IAChB,OAAO,CAAC,MAAqB,IAAY,OAAO,CAAC,CAAC,CAAC,CAAC;IAEpD,KAAK,CAAC,QAAQ,CAAC,EAAE,MAAM,EAAwE;QAC3F,oEAAoE;QACpE,mEAAmE;QACnE,MAAM,EAAE,cAAc,EAAE,CAAC;QACzB,MAAM,IAAI,GAAG,IAAI,CAAC,MAAM,CAAC,KAAK,EAAE,CAAC;QACjC,IAAI,IAAI,KAAK,SAAS;YAAE,MAAM,IAAI,KAAK,CAAC,mDAAmD,CAAC,CAAC;QAC7F,MAAM,CAAC,GAAG,IAAI,CAAC,SAAS,CAAC;QACzB,MAAM,SAAS,GAA0B;YACrC,OAAO,EAAE,CAAC,CAAC,OAAO;YAClB,SAAS,EAAE,CAAC,CAAC,SAAS;YACtB,KAAK,EAAE,EAAE,GAAG,aAAa,EAAE,GAAG,CAAC,CAAC,KAAK,EAAE;YACvC,YAAY,EAAE,CAAC,CAAC,YAAY,IAAI,MAAM;YACtC,KAAK,EAAE,CAAC,CAAC,KAAK,IAAI,MAAM;YACxB,GAAG,CAAC,CAAC,CAAC,GAAG,KAAK,SAAS,CAAC,CAAC,CAAC,EAAE,GAAG,EAAE,CAAC,CAAC,GAAG,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC;SACjD,CAAC;QACF,OAAO,EAAE,SAAS,EAAE,YAAY,EAAE,IAAI,CAAC,YAAY,IAAI,IAAI,EAAE,CAAC;IAClE,CAAC;IAED,IAAI,SAAS,KAAa,OAAO,IAAI,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC,CAAC;CACzD;AAED,OAAO,EAAE,aAAa,IAAI,gBAAgB,EAAE,CAAC"}
|