@plurnk/plurnk-providers 1.3.12 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +39 -22
- package/README.md +65 -4
- package/SPEC.md +222 -56
- package/dist/AiSdkProvider.d.ts +18 -5
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +240 -153
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +14 -15
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +26 -10
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +8 -2
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +41 -8
- package/dist/Pool.js.map +1 -1
- package/dist/ProviderRegistry.d.ts +4 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +7 -3
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +4 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +18 -3
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +3 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +20 -7
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +11 -4
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +11 -0
- package/dist/cost.d.ts.map +1 -0
- package/dist/cost.js +64 -0
- package/dist/cost.js.map +1 -0
- package/dist/discover.d.ts +2 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js +15 -9
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +4 -0
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +29 -9
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +27 -0
- package/dist/errors.d.ts.map +1 -0
- package/dist/errors.js +150 -0
- package/dist/errors.js.map +1 -0
- package/dist/index.d.ts +11 -5
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +7 -4
- package/dist/index.js.map +1 -1
- package/dist/notices.d.ts +10 -0
- package/dist/notices.d.ts.map +1 -0
- package/dist/notices.js +11 -0
- package/dist/notices.js.map +1 -0
- package/dist/ollama.d.ts.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/promptTokens.d.ts +4 -0
- package/dist/promptTokens.d.ts.map +1 -0
- package/dist/promptTokens.js +32 -0
- package/dist/promptTokens.js.map +1 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +4 -3
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +43 -16
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +1 -1
- package/dist/types.js.map +1 -1
- package/dist/usage.d.ts +3 -0
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +26 -14
- package/dist/usage.js.map +1 -1
- package/dist/warnings.js +0 -0
- package/dist/warnings.js.map +1 -1
- package/package.json +13 -9
- package/src/AiSdkProvider.test.ts +480 -159
- package/src/AiSdkProvider.ts +320 -196
- package/src/Mock.test.ts +29 -14
- package/src/Mock.ts +33 -15
- package/src/Pool.test.ts +43 -6
- package/src/Pool.ts +56 -10
- package/src/ProviderRegistry.test.ts +158 -9
- package/src/ProviderRegistry.ts +19 -6
- package/src/aiSdkTransport.ts +25 -6
- package/src/boundaries.test.ts +8 -3
- package/src/catalogProvider.test.ts +17 -0
- package/src/catalogProvider.ts +25 -10
- package/src/compatibleProvider.test.ts +96 -0
- package/src/compatibleProvider.ts +15 -6
- package/src/cost.test.ts +63 -0
- package/src/cost.ts +83 -0
- package/src/defaults.test.ts +1 -0
- package/src/discover.test.ts +48 -7
- package/src/discover.ts +31 -21
- package/src/env.test.ts +38 -23
- package/src/env.ts +45 -18
- package/src/errors.test.ts +148 -0
- package/src/errors.ts +207 -0
- package/src/index.ts +29 -7
- package/src/lexicon-guard.test.ts +6 -6
- package/src/notices.ts +22 -0
- package/src/ollama.test.ts +64 -0
- package/src/ollama.ts +6 -3
- package/src/openai.ts +3 -0
- package/src/promptTokens.ts +41 -0
- package/src/sdkModels.test.ts +7 -0
- package/src/sdkModels.ts +4 -8
- package/src/types.ts +106 -64
- package/src/usage.test.ts +15 -4
- package/src/usage.ts +32 -14
- package/src/warnings.test.ts +10 -10
- package/src/warnings.ts +0 -0
- package/dist/OpenAICompat.d.ts +0 -76
- package/dist/OpenAICompat.d.ts.map +0 -1
- package/dist/OpenAICompat.js +0 -555
- package/dist/OpenAICompat.js.map +0 -1
- package/dist/openaiStream.d.ts +0 -47
- package/dist/openaiStream.d.ts.map +0 -1
- package/dist/openaiStream.js +0 -280
- package/dist/openaiStream.js.map +0 -1
- package/dist/standardProviders.d.ts +0 -31
- package/dist/standardProviders.d.ts.map +0 -1
- package/dist/standardProviders.js +0 -518
- package/dist/standardProviders.js.map +0 -1
- package/dist/telemetry.d.ts +0 -24
- package/dist/telemetry.d.ts.map +0 -1
- package/dist/telemetry.js +0 -85
- package/dist/telemetry.js.map +0 -1
- package/src/telemetry.test.ts +0 -69
- package/src/telemetry.ts +0 -116
package/src/AiSdkProvider.ts
CHANGED
|
@@ -6,28 +6,24 @@
|
|
|
6
6
|
// ordinary vendor protocol. The compatible URL path remains only for PLURNK
|
|
7
7
|
// extensions and local endpoint probes the SDK cannot represent.
|
|
8
8
|
|
|
9
|
-
import type { ChatMessage,
|
|
10
|
-
import type {
|
|
9
|
+
import type { ChatMessage, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
10
|
+
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
11
|
+
import type { Reasoning, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
|
|
11
12
|
import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.ts";
|
|
12
13
|
import type { LanguageModel } from "ai";
|
|
13
|
-
import { toProviderError, ProviderError
|
|
14
|
-
import {
|
|
14
|
+
import { toProviderError, ProviderError } from "./errors.ts";
|
|
15
|
+
import { attributeUnitemizedReasoning } from "./usage.ts";
|
|
16
|
+
import type { ProviderNotice } from "./notices.ts";
|
|
17
|
+
import { validateGbnf } from "@plurnk/gbnf";
|
|
18
|
+
import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.ts";
|
|
15
19
|
import { emitWarningOnce } from "./warnings.ts";
|
|
20
|
+
import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
16
21
|
|
|
17
22
|
export type ProviderFetch = typeof globalThis.fetch;
|
|
18
23
|
|
|
19
|
-
//
|
|
20
|
-
//
|
|
21
|
-
|
|
22
|
-
// "template" ALWAYS emits enable_thinking — the explicit false is llama-server's
|
|
23
|
-
// only working off-switch (§13); "anthropic" uses the `thinking` object and IGNORES
|
|
24
|
-
// reasoning_effort; "effort_explicit" (fireworks) sends the EXPLICIT "none" for OFF
|
|
25
|
-
// instead of omitting — reason-by-DEFAULT models (DeepSeek V4 defaults
|
|
26
|
-
// 'high') keep reasoning when the field is omitted, fatal under an active grammar
|
|
27
|
-
// (#30). Intent maps IDENTICALLY with or without a transported grammar — fireworks
|
|
28
|
-
// masks only the content channel, so reasoning and rails coexist in one call
|
|
29
|
-
// (canary-verified; the #32 clamp is lifted — it caused the plan-less service#331).
|
|
30
|
-
export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "template" | "anthropic";
|
|
24
|
+
// Backend wire spellings for the resolved reasoning intent. The switch beside each
|
|
25
|
+
// mapping retains any backend-specific omission/explicit-disable constraint.
|
|
26
|
+
export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "thinking_effort" | "template" | "anthropic";
|
|
31
27
|
|
|
32
28
|
// GBNF transport is a local llama-server capability. "none" means no
|
|
33
29
|
// service-managed constrained sampling; endpoint-owned settings are not inferred.
|
|
@@ -37,17 +33,20 @@ export type AiSdkProviderConfig = {
|
|
|
37
33
|
model: string;
|
|
38
34
|
url?: string; // OpenAI-compatible chat-completions URL
|
|
39
35
|
languageModel?: LanguageModel; // native AI SDK provider model
|
|
36
|
+
attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
40
37
|
fetchTimeoutMs: number;
|
|
41
38
|
streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
|
|
42
39
|
headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
|
|
43
40
|
fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
|
|
44
|
-
contextWindow?: number | null; // default null; caller resolves-or-fails
|
|
41
|
+
contextWindow?: number | null; // default null; caller resolves-or-fails, narrows to required with the interface
|
|
45
42
|
reasoningStyle?: ReasoningStyle; // default "none"
|
|
46
|
-
|
|
43
|
+
reasoningResponseStyle?: ReasoningResponseStyle; // {§provider-tagged-reasoning}; default "verbatim"
|
|
44
|
+
countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
|
|
47
45
|
calculateCost?: (usage: ProviderUsage) => number; // default () => 0
|
|
48
|
-
|
|
46
|
+
calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
|
|
47
|
+
source?: string; // notice/problem source, e.g. "provider:openai"; default "provider"
|
|
49
48
|
grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
|
|
50
|
-
//
|
|
49
|
+
// Send the OpenAI-standard `prompt_cache_key` set to workerId, so a
|
|
51
50
|
// serverless backend with REPLICA-LOCAL prompt caching (fireworks, verified)
|
|
52
51
|
// pins a worker's turns to one replica and claims its stable prefix. Default
|
|
53
52
|
// false -- a backend that strict-validates unknown fields 400s, so enable only
|
|
@@ -59,21 +58,24 @@ export type AiSdkProviderConfig = {
|
|
|
59
58
|
gbnfDebug?: boolean; // PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar locally + throw on invalid, but DON'T transport it (run unconstrained); default false
|
|
60
59
|
streaming?: boolean; // SSE transport (default true); false → one non-streamed JSON
|
|
61
60
|
firstPartyMetadata?: boolean; // forward per-turn attributions + client as Plurnk-* headers (plurnk only); default false
|
|
62
|
-
apiKeyRejectedMessage?: string; //
|
|
63
|
-
eosText?: string; //
|
|
64
|
-
// Slot affinity wiring
|
|
61
|
+
apiKeyRejectedMessage?: string; // friendly hint when a present key is 401/403-rejected (distinct from unset); default undefined
|
|
62
|
+
eosText?: string; // server-reported eos_token, stripped from the content tail (--special renders it as text); default undefined
|
|
63
|
+
// Slot affinity wiring is provider-internal, never consumer-facing.
|
|
65
64
|
supportsSlotPinning?: boolean; // backend accepts an `id_slot` body field (llama-server); default false
|
|
66
65
|
slotCount?: number | null; // probed slot count for pinning backends; default null
|
|
67
66
|
// Backend-served exact tokenization (llama-server /tokenize). When set, the
|
|
68
67
|
// provider exposes the optional `tokenize()` capability — the model's OWN
|
|
69
68
|
// vocab, no client-side tokenizer data needed; default unset (capability absent).
|
|
70
69
|
tokenizeUrl?: string;
|
|
71
|
-
//
|
|
70
|
+
// Provider-authoritative count of a complete chat-completions request. This
|
|
71
|
+
// is distinct from /tokenize, which sees content but not the chat template.
|
|
72
|
+
promptTokensUrl?: string;
|
|
73
|
+
// The backend's self-reported served model id (from the /v1/models probe),
|
|
72
74
|
// surfaced as Provider.servedModel. For a local llama-server the wire `model` is
|
|
73
75
|
// the alias; this is the real name (the .gguf) the tokenizer seam maps. Absent
|
|
74
76
|
// when no probe ran or it read no row.
|
|
75
77
|
servedModel?: string;
|
|
76
|
-
//
|
|
78
|
+
// Backend decodes unbounded without a caller cap (llama-server n_predict
|
|
77
79
|
// to the wall) — surfaced as Provider.requiresMaxTokens so consumers can
|
|
78
80
|
// boot-refuse an envelope-less local alias. Default unset (no claim).
|
|
79
81
|
requiresMaxTokens?: boolean;
|
|
@@ -81,36 +83,39 @@ export type AiSdkProviderConfig = {
|
|
|
81
83
|
// (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
|
|
82
84
|
// { mode: off|adaptive|on, budget: iff on }. The provider maps it to the
|
|
83
85
|
// backend's mechanism via reasoningStyle; budget is only ever a magnitude,
|
|
84
|
-
// never a hidden activation flag
|
|
86
|
+
// never a hidden activation flag.
|
|
85
87
|
reasoning: Reasoning;
|
|
86
88
|
// Decode tuning: no in-code defaults; the canonical measured values (0.2 /
|
|
87
89
|
// 1.15) ship as the floor in .env.defaults (alias-scopable). `temperature` is the
|
|
88
|
-
// DEFAULT for EVERY request, spread UNDER caller sampling
|
|
90
|
+
// DEFAULT for EVERY request, spread UNDER caller sampling.
|
|
89
91
|
// `repeatPenalty` is the FLOOR the provider manages wherever a grammar rides
|
|
90
|
-
// (greedy-under-mask loops without it
|
|
92
|
+
// (greedy-under-mask loops without it) — the VALUE is operator config;
|
|
91
93
|
// WHERE it applies stays mechanism.
|
|
92
94
|
temperature: number;
|
|
93
95
|
repeatPenalty: number;
|
|
94
|
-
//
|
|
96
|
+
// Anti-degeneration guard on the cloud path (grammarStyle "none"), where the
|
|
95
97
|
// repeat_penalty multiplier isn't available - the OpenAI-standard frequency_penalty.
|
|
96
98
|
// Optional (default 0 = off) so an out-of-date plugin that omits it just runs unguarded
|
|
97
99
|
// rather than failing construction; the standard factory always supplies it.
|
|
98
100
|
frequencyPenalty?: number;
|
|
99
|
-
//
|
|
100
|
-
//
|
|
101
|
-
//
|
|
102
|
-
//
|
|
103
|
-
// like repeatPenalty — applied for a detected llama.cpp backend, customer-overridable;
|
|
104
|
-
// absent (a plugin omitting them) = the box's own default. repeatLastN widens the window.
|
|
101
|
+
// Optional llama.cpp anti-repetition controls. DRY can suppress long
|
|
102
|
+
// repeated sequences, but it can also corrupt exact repetition required by
|
|
103
|
+
// PLURNK operations. The portable default is off; these fields ride only
|
|
104
|
+
// after an explicit operator opt-in. repeatLastN widens the repeat window.
|
|
105
105
|
dryMultiplier?: number;
|
|
106
106
|
dryBase?: number;
|
|
107
107
|
dryAllowedLength?: number;
|
|
108
108
|
repeatLastN?: number;
|
|
109
109
|
// Transient-failure retry budget — REQUIRED, no in-code default
|
|
110
110
|
// (PLURNK_PROVIDERS_RETRY_ATTEMPTS, a non-negative int): 0 = surface the
|
|
111
|
-
// first failure; N = up to N retries on a transient error
|
|
111
|
+
// first failure; N = up to N retries on a transient error
|
|
112
|
+
// ({§provider-failure-normalization}).
|
|
112
113
|
retryAttempts: number;
|
|
113
|
-
//
|
|
114
|
+
// Maximum characters retained from an upstream diagnostic in the public
|
|
115
|
+
// ProviderError Problem. Standard factories supply the env-owned value;
|
|
116
|
+
// direct construction may omit it to preserve the complete diagnostic.
|
|
117
|
+
errorDetailLimit?: number;
|
|
118
|
+
// Data-capture knobs ({§provider-evidence}), off by default — the flag is the isolation, so a
|
|
114
119
|
// serving turn requests nothing and carries nothing. `topLogprobs`: when a
|
|
115
120
|
// non-negative int, request `logprobs:true, top_logprobs:<n>` and surface the
|
|
116
121
|
// per-token confidence on assistant.logprobs (PLURNK_PROVIDERS_TOP_LOGPROBS;
|
|
@@ -119,7 +124,7 @@ export type AiSdkProviderConfig = {
|
|
|
119
124
|
// gated per-alias.
|
|
120
125
|
topLogprobs?: number | null;
|
|
121
126
|
rawBody?: boolean;
|
|
122
|
-
//
|
|
127
|
+
// {§provider-generation-envelope} The generation-envelope reserves, env-read via
|
|
123
128
|
// envelopeFromEnv — a percentage of the DETECTED window or an absolute token
|
|
124
129
|
// count. Optional so an out-of-date sibling keeps constructing (no claim);
|
|
125
130
|
// the standard factory always supplies them. Resolved against contextWindow
|
|
@@ -127,13 +132,13 @@ export type AiSdkProviderConfig = {
|
|
|
127
132
|
// derives correctly.
|
|
128
133
|
reasoningReserve?: ReserveSpec;
|
|
129
134
|
completionReserve?: ReserveSpec;
|
|
130
|
-
//
|
|
135
|
+
// The plurnk.ai router owns tuning — false suppresses the
|
|
131
136
|
// client-side temperature/penalty FLOORS on this provider (caller `sampling`
|
|
132
137
|
// still passes through verbatim). Default true (floors ride).
|
|
133
138
|
tuningFloors?: boolean;
|
|
134
139
|
};
|
|
135
140
|
|
|
136
|
-
//
|
|
141
|
+
// Drop trailing occurrences of a server-rendered EOG marker. llama-server
|
|
137
142
|
// under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
|
|
138
143
|
// trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
|
|
139
144
|
// can never eat body content (a body ending in the literal marker isn't producible
|
|
@@ -145,6 +150,44 @@ const stripTrailingSpecial = (content: string, marker: string): string => {
|
|
|
145
150
|
return out;
|
|
146
151
|
};
|
|
147
152
|
|
|
153
|
+
type TaggedReasoningProjection = {
|
|
154
|
+
readonly content: string;
|
|
155
|
+
readonly reasoning: string;
|
|
156
|
+
readonly projected: boolean;
|
|
157
|
+
readonly contentStart: number;
|
|
158
|
+
};
|
|
159
|
+
|
|
160
|
+
// {§provider-tagged-reasoning} Only the model-contract position is structural:
|
|
161
|
+
// one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
|
|
162
|
+
// on one path and leaves later literal tags in the visible suffix untouched.
|
|
163
|
+
const projectTaggedReasoning = (
|
|
164
|
+
content: string,
|
|
165
|
+
structuredReasoning: string,
|
|
166
|
+
style: ReasoningResponseStyle,
|
|
167
|
+
): TaggedReasoningProjection => {
|
|
168
|
+
const opening = "<think>";
|
|
169
|
+
if (style !== "think-tags" || structuredReasoning.length > 0 || !content.startsWith(opening)) {
|
|
170
|
+
return { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
|
|
171
|
+
}
|
|
172
|
+
const closing = "</think>";
|
|
173
|
+
const closingIndex = content.indexOf(closing, opening.length);
|
|
174
|
+
if (closingIndex === -1) {
|
|
175
|
+
return {
|
|
176
|
+
content: "",
|
|
177
|
+
reasoning: content.slice(opening.length),
|
|
178
|
+
projected: true,
|
|
179
|
+
contentStart: [...content].length,
|
|
180
|
+
};
|
|
181
|
+
}
|
|
182
|
+
const suffixStart = closingIndex + closing.length;
|
|
183
|
+
return {
|
|
184
|
+
content: content.slice(suffixStart),
|
|
185
|
+
reasoning: content.slice(opening.length, closingIndex),
|
|
186
|
+
projected: true,
|
|
187
|
+
contentStart: [...content.slice(0, suffixStart)].length,
|
|
188
|
+
};
|
|
189
|
+
};
|
|
190
|
+
|
|
148
191
|
// Shared budget→effort breakpoints (xai and google had identical copies).
|
|
149
192
|
export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
|
|
150
193
|
if (budget <= 1000) return "low";
|
|
@@ -152,44 +195,28 @@ export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
|
|
|
152
195
|
return "high";
|
|
153
196
|
};
|
|
154
197
|
|
|
155
|
-
// chars/2 upper bound (see ./tokenizers.ts) — overcounts safely, never under.
|
|
156
|
-
const heuristicTokens = (text: string): number => (text.length === 0 ? 0 : Math.ceil(text.length / 2));
|
|
157
|
-
|
|
158
198
|
// Body keys the provider owns — a caller's `sampling` passthrough may not set
|
|
159
|
-
// these. Two families
|
|
199
|
+
// these. Two families:
|
|
160
200
|
// transport/managed — grammar transport, the stream/JSON choice, slot pinning,
|
|
161
|
-
// data capture (
|
|
201
|
+
// data capture ({§provider-evidence}: backend-specific fields never cross the contract);
|
|
162
202
|
// contract invariants — `n` (atomic single completion: choices[0] is the
|
|
163
203
|
// response; n>1 = paid, dropped output), the tool-calling family (tools-in-
|
|
164
204
|
// body doctrine, §2: native tool_calls return null content = a broken turn),
|
|
165
205
|
// modalities/audio (text-only contract), prediction (decode semantics, not
|
|
166
206
|
// sampling), and the token caps (the envelope is the managed maxTokens —
|
|
167
|
-
// sampling must not bypass the consumer's
|
|
207
|
+
// sampling must not bypass the consumer's cap).
|
|
168
208
|
// Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
|
|
169
209
|
// platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
|
|
170
210
|
// metadata, store, verbosity) pass through; the managed floors spread UNDER
|
|
171
211
|
// sampling stay deliberately caller-overridable.
|
|
172
212
|
const RESERVED_BODY_KEYS: ReadonlySet<string> = new Set([
|
|
173
213
|
"model", "messages", "stream", "stream_options", "grammar", "response_format", "id_slot", "logprobs", "top_logprobs",
|
|
214
|
+
"reasoning_format", "reasoning_effort", "thinking", "think", "include_reasoning", "chat_template_kwargs", "thinking_budget_tokens", // lexicon-allow: backend wire fields
|
|
174
215
|
"n", "tools", "tool_choice", "functions", "function_call", "parallel_tool_calls",
|
|
175
216
|
"modalities", "audio", "prediction", "max_tokens", "max_completion_tokens",
|
|
176
217
|
"prompt_cache_key",
|
|
177
218
|
]);
|
|
178
219
|
|
|
179
|
-
// Render a non-accept verdict into a terse, factual grammar_unenforced message
|
|
180
|
-
// (SPEC §12 message policy: no guidance prose). `reject` names the diverging code
|
|
181
|
-
// point + what the grammar would have accepted; `incomplete` names the valid-prefix
|
|
182
|
-
// length that never reached a terminal state.
|
|
183
|
-
const describeUnenforced = (v: Exclude<Verdict, { status: "accept" }>): string => {
|
|
184
|
-
if (v.status === "reject") {
|
|
185
|
-
const expected = v.expected.length > 0
|
|
186
|
-
? v.expected.map((e) => `${e.rule} accepts ${e.accepts}`).join(", ")
|
|
187
|
-
: "end of input";
|
|
188
|
-
return `grammar not enforced: output rejected by the transported grammar at code point ${v.pos} (${JSON.stringify(v.char)}); expected ${expected}`;
|
|
189
|
-
}
|
|
190
|
-
return `grammar not enforced: output is an incomplete match of the transported grammar — a valid prefix of ${v.pos} code points that never terminated`;
|
|
191
|
-
};
|
|
192
|
-
|
|
193
220
|
export default class AiSdkProvider implements Provider {
|
|
194
221
|
#model: string;
|
|
195
222
|
#url: string | undefined;
|
|
@@ -211,8 +238,11 @@ export default class AiSdkProvider implements Provider {
|
|
|
211
238
|
#dryAllowedLength: number | undefined;
|
|
212
239
|
#repeatLastN: number | undefined;
|
|
213
240
|
#reasoningStyle: ReasoningStyle;
|
|
214
|
-
#
|
|
241
|
+
#reasoningResponseStyle: ReasoningResponseStyle;
|
|
242
|
+
#countPromptTokens: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
|
|
243
|
+
#promptTokensUrl: string | undefined;
|
|
215
244
|
#calculateCost: (usage: ProviderUsage) => number;
|
|
245
|
+
#calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
|
|
216
246
|
#source: string;
|
|
217
247
|
#grammarStyle: GrammarStyle;
|
|
218
248
|
#promptCacheKey: boolean;
|
|
@@ -223,6 +253,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
223
253
|
#supportsSlotPinning: boolean;
|
|
224
254
|
#slotCount: number | null;
|
|
225
255
|
#retryAttempts: number;
|
|
256
|
+
#errorDetailLimit: number | undefined;
|
|
226
257
|
#topLogprobs: number | null;
|
|
227
258
|
#reasoningReserve: ReserveSpec | undefined;
|
|
228
259
|
#completionReserve: ReserveSpec | undefined;
|
|
@@ -230,8 +261,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
230
261
|
#rawBody: boolean;
|
|
231
262
|
#servedModel: string | undefined;
|
|
232
263
|
#requiresMaxTokens: boolean | undefined;
|
|
264
|
+
readonly attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
233
265
|
|
|
234
|
-
// Optional capability (
|
|
266
|
+
// Optional capability ({§provider-local-capabilities}): exact tokenization served by the backend's
|
|
235
267
|
// own vocab. Assigned in the constructor ONLY when the config carries a
|
|
236
268
|
// tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
|
|
237
269
|
// the honest capability signal for every other backend.
|
|
@@ -241,6 +273,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
241
273
|
this.#model = config.model;
|
|
242
274
|
this.#url = config.url;
|
|
243
275
|
this.#languageModel = config.languageModel;
|
|
276
|
+
this.attributions = config.attributions;
|
|
244
277
|
if ((this.#url === undefined) === (this.#languageModel === undefined)) {
|
|
245
278
|
throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
|
|
246
279
|
}
|
|
@@ -264,9 +297,16 @@ export default class AiSdkProvider implements Provider {
|
|
|
264
297
|
this.#dryAllowedLength = config.dryAllowedLength;
|
|
265
298
|
this.#repeatLastN = config.repeatLastN;
|
|
266
299
|
this.#retryAttempts = config.retryAttempts;
|
|
300
|
+
this.#errorDetailLimit = config.errorDetailLimit;
|
|
267
301
|
this.#reasoningStyle = config.reasoningStyle ?? "none";
|
|
268
|
-
this.#
|
|
302
|
+
this.#reasoningResponseStyle = config.reasoningResponseStyle ?? "verbatim";
|
|
303
|
+
if (config.countPromptTokens !== undefined && config.promptTokensUrl !== undefined) {
|
|
304
|
+
throw new Error(`${config.source ?? "provider"}: configure countPromptTokens or promptTokensUrl, not both`);
|
|
305
|
+
}
|
|
306
|
+
this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
|
|
307
|
+
this.#promptTokensUrl = config.promptTokensUrl;
|
|
269
308
|
this.#calculateCost = config.calculateCost ?? (() => 0);
|
|
309
|
+
this.#calculateCharge = config.calculateCharge;
|
|
270
310
|
this.#source = config.source ?? "provider";
|
|
271
311
|
this.#grammarStyle = config.grammarStyle ?? "none";
|
|
272
312
|
this.#promptCacheKey = config.promptCacheKey ?? false;
|
|
@@ -286,6 +326,13 @@ export default class AiSdkProvider implements Provider {
|
|
|
286
326
|
this.#rawBody = config.rawBody ?? false;
|
|
287
327
|
this.#servedModel = config.servedModel;
|
|
288
328
|
this.#requiresMaxTokens = config.requiresMaxTokens;
|
|
329
|
+
const reasoningReserve = this.reasoningReserve;
|
|
330
|
+
if (this.#reasoningStyle === "template"
|
|
331
|
+
&& this.#reasoning.mode === "on"
|
|
332
|
+
&& reasoningReserve !== null
|
|
333
|
+
&& this.#reasoning.budget! > reasoningReserve) {
|
|
334
|
+
throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
|
|
335
|
+
}
|
|
289
336
|
const { tokenizeUrl } = config;
|
|
290
337
|
if (tokenizeUrl !== undefined) {
|
|
291
338
|
this.tokenize = async (text: string): Promise<number[]> => {
|
|
@@ -306,7 +353,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
306
353
|
}
|
|
307
354
|
|
|
308
355
|
get contextWindow(): number | null { return this.#contextWindow; }
|
|
309
|
-
//
|
|
356
|
+
// {§provider-generation-envelope} Envelope reserves — absolute pins stand alone; percentages need the
|
|
310
357
|
// detected window; null = underivable (no claim for core's no-cap path).
|
|
311
358
|
#resolveReserve(spec: ReserveSpec | undefined): number | null {
|
|
312
359
|
if (spec === undefined) return null;
|
|
@@ -316,63 +363,109 @@ export default class AiSdkProvider implements Provider {
|
|
|
316
363
|
get reasoningReserve(): number | null { return this.#resolveReserve(this.#reasoningReserve); }
|
|
317
364
|
get completionReserve(): number | null { return this.#resolveReserve(this.#completionReserve); }
|
|
318
365
|
get model(): string { return this.#model; }
|
|
319
|
-
//
|
|
366
|
+
// Backend's self-reported served id; undefined when unprobed/unknown.
|
|
320
367
|
get servedModel(): string | undefined { return this.#servedModel; }
|
|
321
|
-
//
|
|
368
|
+
// Resolved "decodes unbounded without a cap" fact; undefined = no claim.
|
|
322
369
|
get requiresMaxTokens(): boolean | undefined { return this.#requiresMaxTokens; }
|
|
323
|
-
// Resolved capability
|
|
370
|
+
// Resolved capability: will a transported grammar actually constrain
|
|
324
371
|
// this backend's decode? Introspectable so a consumer can verify the rails
|
|
325
372
|
// are LIVE without spending a generation on a forcing-grammar probe.
|
|
326
373
|
get constrainsOutput(): boolean { return this.#grammarStyle !== "none"; }
|
|
327
374
|
|
|
328
|
-
|
|
375
|
+
async countPromptTokens(
|
|
376
|
+
messages: readonly ChatMessage[],
|
|
377
|
+
signal?: AbortSignal,
|
|
378
|
+
): Promise<PromptTokenMeasurement> {
|
|
379
|
+
if (this.#promptTokensUrl === undefined) {
|
|
380
|
+
return assertPromptTokenMeasurement(
|
|
381
|
+
await this.#countPromptTokens(messages, signal),
|
|
382
|
+
this.#source,
|
|
383
|
+
);
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
signal?.throwIfAborted();
|
|
387
|
+
try {
|
|
388
|
+
const timeout = AbortSignal.timeout(this.#fetchTimeoutMs);
|
|
389
|
+
const response = await this.#fetch(this.#promptTokensUrl, {
|
|
390
|
+
method: "POST",
|
|
391
|
+
headers: { "Content-Type": "application/json", ...this.#headers },
|
|
392
|
+
body: JSON.stringify({
|
|
393
|
+
model: this.#model,
|
|
394
|
+
messages,
|
|
395
|
+
...this.#reasoningBody(),
|
|
396
|
+
}),
|
|
397
|
+
signal: signal === undefined ? timeout : AbortSignal.any([signal, timeout]),
|
|
398
|
+
});
|
|
399
|
+
if (!response.ok) {
|
|
400
|
+
return estimatePromptTokens(
|
|
401
|
+
messages,
|
|
402
|
+
`llama-server input-token endpoint returned HTTP ${response.status}`,
|
|
403
|
+
);
|
|
404
|
+
}
|
|
405
|
+
const body = await response.json() as { input_tokens?: unknown };
|
|
406
|
+
if (!Number.isInteger(body.input_tokens) || (body.input_tokens as number) < 0) {
|
|
407
|
+
return estimatePromptTokens(
|
|
408
|
+
messages,
|
|
409
|
+
"llama-server input-token endpoint returned no non-negative integer input_tokens",
|
|
410
|
+
);
|
|
411
|
+
}
|
|
412
|
+
return {
|
|
413
|
+
kind: "exact",
|
|
414
|
+
tokens: body.input_tokens as number,
|
|
415
|
+
source: "llama-server:/v1/chat/completions/input_tokens",
|
|
416
|
+
};
|
|
417
|
+
} catch (cause) {
|
|
418
|
+
signal?.throwIfAborted();
|
|
419
|
+
return estimatePromptTokens(
|
|
420
|
+
messages,
|
|
421
|
+
`llama-server input-token measurement failed: ${cause instanceof Error ? cause.message : String(cause)}`,
|
|
422
|
+
);
|
|
423
|
+
}
|
|
424
|
+
}
|
|
329
425
|
calculateCost(usage: ProviderUsage): number { return this.#calculateCost(usage); }
|
|
426
|
+
calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> {
|
|
427
|
+
return this.#calculateCharge?.(usage)
|
|
428
|
+
?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
|
|
429
|
+
}
|
|
330
430
|
|
|
331
|
-
//
|
|
332
|
-
//
|
|
333
|
-
// clamp (force reasoning_effort "none" under response_format) is LIFTED:
|
|
334
|
-
// canary-verified live that fireworks masks ONLY the content channel — the
|
|
335
|
-
// reasoning channel rides beside it unmasked, and the plurnk grammar's
|
|
336
|
-
// reasoning?/preplan regions absorb any in-band spillover. The old measured
|
|
337
|
-
// failures (low→cycles, high→spirals) were pre-max_tokens-cap; with a bounded
|
|
338
|
-
// cap the matrix ACCEPTs across efforts (reasoning-rails matrix
|
|
339
|
-
// F9). Clamping was the root of the plan-less regression (service#331).
|
|
431
|
+
// Reasoning intent maps independently of grammar transport. The llama-server
|
|
432
|
+
// template mapping is owned by {§llama-reasoning-request}.
|
|
340
433
|
#reasoningBody(): Record<string, unknown> {
|
|
341
434
|
const { mode, budget } = this.#reasoning;
|
|
342
435
|
const on = mode !== "off";
|
|
343
436
|
switch (this.#reasoningStyle) {
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
// constrained (26-run baseline green; zero grammar rejects across
|
|
355
|
-
// the #488 "railless" specimens). Closing the channel starved a
|
|
356
|
-
// reasoning-tuned model into ESCAPING mid-content into the raw
|
|
357
|
-
// thought channel — discarded server-side, decode unconstrained,
|
|
358
|
-
// 12,288 tokens billed for 1,033 visible chars. The escape is
|
|
359
|
-
// surfaced instead (vanished-token telemetry + meta rail state).
|
|
360
|
-
case "template": return { chat_template_kwargs: { enable_thinking: on } };
|
|
437
|
+
case "template": {
|
|
438
|
+
const allowance = mode === "off"
|
|
439
|
+
? 0
|
|
440
|
+
: mode === "on" ? budget : this.reasoningReserve;
|
|
441
|
+
return {
|
|
442
|
+
chat_template_kwargs: { enable_thinking: on },
|
|
443
|
+
reasoning_format: "auto",
|
|
444
|
+
...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
|
|
445
|
+
};
|
|
446
|
+
}
|
|
361
447
|
case "think": return on ? { think: true } : {};
|
|
362
448
|
case "include_reasoning": return on ? { include_reasoning: true } : {};
|
|
363
449
|
// effort tiers from the budget; off/adaptive omit the field (the
|
|
364
450
|
// API's default depth is its adaptive).
|
|
365
451
|
case "effort": return mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
|
|
366
452
|
// Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
|
|
367
|
-
// reason-by-default model (DeepSeek V4: default 'high') reasoning
|
|
453
|
+
// reason-by-default model (DeepSeek V4: default 'high') reasoning.
|
|
368
454
|
// ADAPTIVE omits the field: the backend's own default posture IS the
|
|
369
455
|
// adaptive semantics, and the literal "adaptive" is MiniMax-M3-only —
|
|
370
|
-
//
|
|
456
|
+
// Fireworks 400s it for every other model (wire-verified; the
|
|
371
457
|
// 1.0.2 adaptive default refused to boot on it). V4 gotcha: integer
|
|
372
458
|
// efforts 400.
|
|
373
459
|
case "effort_explicit": return mode === "off"
|
|
374
460
|
? { reasoning_effort: "none" }
|
|
375
461
|
: mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
|
|
462
|
+
// {§deepseek-reasoning-request}
|
|
463
|
+
case "thinking_effort": return mode === "off"
|
|
464
|
+
? { thinking: { type: "disabled" } }
|
|
465
|
+
: mode === "on" ? {
|
|
466
|
+
thinking: { type: "enabled" },
|
|
467
|
+
reasoning_effort: effortFromBudget(budget!),
|
|
468
|
+
} : {};
|
|
376
469
|
// Anthropic compat: explicit thinking object. off → disabled; on →
|
|
377
470
|
// enabled with budget_tokens; adaptive → omit (the API default).
|
|
378
471
|
case "anthropic": return mode === "off"
|
|
@@ -382,7 +475,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
382
475
|
}
|
|
383
476
|
}
|
|
384
477
|
|
|
385
|
-
// Per-
|
|
478
|
+
// Per-worker slot affinity: the consumer passes which worker this is; the
|
|
386
479
|
// provider owns WHICH slot serves it. Sticky per workerId, round-robin across
|
|
387
480
|
// new runs (distinct runs → distinct slots while slots last), LRU-bounded
|
|
388
481
|
// bookkeeping so a long-lived daemon never grows the map unboundedly —
|
|
@@ -405,19 +498,19 @@ export default class AiSdkProvider implements Provider {
|
|
|
405
498
|
return { id_slot: slot };
|
|
406
499
|
}
|
|
407
500
|
|
|
408
|
-
// Optional local llama-server GBNF transport (
|
|
501
|
+
// Optional local llama-server GBNF transport ({§gbnf-response-observation}). Unsupported
|
|
409
502
|
// backends receive no grammar-related field.
|
|
410
503
|
#grammarBody(grammar: string | undefined): Record<string, unknown> {
|
|
411
504
|
if (grammar === undefined) return {};
|
|
412
505
|
switch (this.#grammarStyle) {
|
|
413
506
|
// Greedy decoding under hard constraint loops without a repeat-penalty
|
|
414
|
-
// floor
|
|
507
|
+
// floor — llama.cpp spells it `repeat_penalty`.
|
|
415
508
|
case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
|
|
416
509
|
case "none": return {};
|
|
417
510
|
}
|
|
418
511
|
}
|
|
419
512
|
|
|
420
|
-
// Anti-degeneration
|
|
513
|
+
// Anti-degeneration default on every request, keyed to the backend's wire
|
|
421
514
|
// convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
|
|
422
515
|
// alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
|
|
423
516
|
// straight to the token cap on pure looped repetition (run52). Ships next to
|
|
@@ -428,7 +521,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
428
521
|
// live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
|
|
429
522
|
#repetitionPenaltyBody(): Record<string, unknown> {
|
|
430
523
|
switch (this.#grammarStyle) {
|
|
431
|
-
//
|
|
524
|
+
// repeat_penalty + optional DRY (repeated-sequence penalty) + a wider
|
|
432
525
|
// repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
|
|
433
526
|
// Each rides only when its operator knob is set; absent = the box's default.
|
|
434
527
|
case "llamacpp": return {
|
|
@@ -444,12 +537,12 @@ export default class AiSdkProvider implements Provider {
|
|
|
444
537
|
}
|
|
445
538
|
}
|
|
446
539
|
|
|
447
|
-
// First-party telemetry headers (
|
|
540
|
+
// First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
|
|
448
541
|
// opted in (the plurnk endpoint). The gate is here, not at the call site, so
|
|
449
542
|
// attributions/client/strikes can never reach a third-party backend even if
|
|
450
543
|
// the consumer passes them to the wrong provider. Empty values emit no header
|
|
451
544
|
// — EXCEPT strikes, where 0 is a real value (clean streak) distinct from
|
|
452
|
-
// absent (consumer didn't report); contract
|
|
545
|
+
// absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
|
|
453
546
|
// ride HTTP headers only — the packet never carries them (the model must
|
|
454
547
|
// never see strike state; engine accounting is not a metric to game).
|
|
455
548
|
#metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined): Record<string, string> {
|
|
@@ -458,11 +551,11 @@ export default class AiSdkProvider implements Provider {
|
|
|
458
551
|
if (attributions !== undefined && attributions.length > 0) h["Plurnk-Attribution"] = JSON.stringify(attributions);
|
|
459
552
|
if (client !== undefined && client.length > 0) h["Plurnk-Client"] = client;
|
|
460
553
|
if (strikes !== undefined && Number.isInteger(strikes) && strikes >= 0) h["Plurnk-Strikes"] = String(strikes);
|
|
461
|
-
// Worker identity
|
|
554
|
+
// Worker identity: the opaque workerId
|
|
462
555
|
// the consumer already supplies, forwarded so the endpoint can key
|
|
463
556
|
// per-worker affinity/telemetry — same gate as every first-party signal.
|
|
464
557
|
h["Plurnk-Worker-Id"] = workerId;
|
|
465
|
-
// Root worker of the lineage (
|
|
558
|
+
// Root worker of the lineage ({§worker-primary}): the no-parent ancestor of this turn's
|
|
466
559
|
// worker tree. The consumer classifies primary-vs-spawned by equality
|
|
467
560
|
// (primaryWorkerId == workerId ⇒ the primary/root worker). The provider
|
|
468
561
|
// EMITS what the consumer supplies and never invents a primary; the
|
|
@@ -470,7 +563,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
470
563
|
// own, where it equals workerId). Absence is the consumer's violation for
|
|
471
564
|
// the endpoint to surface, not a provider default.
|
|
472
565
|
if (primaryWorkerId !== undefined && primaryWorkerId.length > 0) h["Plurnk-Worker-Primary"] = primaryWorkerId;
|
|
473
|
-
// Turn coordinate (
|
|
566
|
+
// Turn coordinate ({§lifecycle-terms}): workspace/loop/turn, the
|
|
474
567
|
// daemon-side sequence the endpoint can never scrape from the wire.
|
|
475
568
|
// Coordinates are 1-based — 0 is not a real value, so no strikes-style
|
|
476
569
|
// zero exception; absent/empty/0 emits no header.
|
|
@@ -480,33 +573,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
480
573
|
return h;
|
|
481
574
|
}
|
|
482
575
|
|
|
483
|
-
//
|
|
484
|
-
// (grammarStyle !== "none"), the backend MUST have constrained the output;
|
|
485
|
-
// some silently drop the grammar field or mislabel the channel, and without
|
|
486
|
-
// this check we would return unconstrained output as if enforced. STRICT: any
|
|
487
|
-
// non-accept verdict (reject, or an incomplete/never-terminated match) is a
|
|
488
|
-
// grammar_unenforced failure. A grammar our own validator can't parse — even
|
|
489
|
-
// though the backend accepted it (a port-vs-llama.cpp gap) — is a non-fatal
|
|
490
|
-
// verify gap: warn, don't fail a transport that may have worked. This is a
|
|
491
|
-
// conformance check against the grammar we already hold, NOT a plurnk-DSL
|
|
492
|
-
// parse (§8) — it stays grammar-generic and backend-agnostic.
|
|
493
|
-
// Validate output against the grammar. Returns the verdict, or null on the
|
|
494
|
-
// verify GAP — a grammar our own validator can't parse (a port-vs-llama.cpp
|
|
495
|
-
// gap): warn, don't manufacture a conflict from a check that didn't run.
|
|
496
|
-
#grammarVerdict(grammar: string, content: string): Verdict | null {
|
|
497
|
-
try {
|
|
498
|
-
return validateGbnf(grammar, content);
|
|
499
|
-
} catch (cause) {
|
|
500
|
-
// Once per (code, message) — #40: this fires PER TURN otherwise.
|
|
501
|
-
emitWarningOnce(
|
|
502
|
-
`${this.#source}: could not verify grammar enforcement — the transported grammar did not parse in @plurnk/gbnf (${(cause as Error).message})`,
|
|
503
|
-
"PLURNK_GRAMMAR_UNVERIFIABLE",
|
|
504
|
-
);
|
|
505
|
-
return null;
|
|
506
|
-
}
|
|
507
|
-
}
|
|
508
|
-
|
|
509
|
-
// PLURNK_PROVIDERS_GBNF_DEBUG (SPEC §13): validate the supplied GBNF locally and fail
|
|
576
|
+
// PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
|
|
510
577
|
// hard if it's malformed, BEFORE any wire call — and the grammar is NOT
|
|
511
578
|
// transported, so the request runs unconstrained. A debug aid to catch invalid
|
|
512
579
|
// grammars (e.g. while editing the plurnk grammar) without a model round-trip;
|
|
@@ -521,7 +588,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
521
588
|
}
|
|
522
589
|
}
|
|
523
590
|
|
|
524
|
-
// Per-turn metadata bag
|
|
591
|
+
// Per-turn metadata bag: pass the backend's non-standard top-level fields
|
|
525
592
|
// through verbatim. Providers do not reinterpret vendor currency or account
|
|
526
593
|
// metadata; a monetary value carries its own amount and currency.
|
|
527
594
|
#buildMeta(chunkMetadata: Record<string, unknown>): Record<string, unknown> | undefined {
|
|
@@ -533,7 +600,8 @@ export default class AiSdkProvider implements Provider {
|
|
|
533
600
|
// penalties, stop, seed, …) merged UNDER the managed body: model, messages,
|
|
534
601
|
// reasoning, grammar (+ its repeat-penalty floor), max_tokens and slot always
|
|
535
602
|
// win, and reserved transport/protocol keys are stripped so the passthrough
|
|
536
|
-
// can't smuggle a grammar, a stream toggle, or a backend slot
|
|
603
|
+
// can't smuggle a grammar, a stream toggle, or a backend slot
|
|
604
|
+
// ({§provider-request-authority}).
|
|
537
605
|
#samplingBody(sampling: Record<string, unknown> | undefined): Record<string, unknown> {
|
|
538
606
|
if (sampling === undefined) return {};
|
|
539
607
|
const out: Record<string, unknown> = {};
|
|
@@ -542,27 +610,24 @@ export default class AiSdkProvider implements Provider {
|
|
|
542
610
|
}
|
|
543
611
|
|
|
544
612
|
async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse> {
|
|
545
|
-
//
|
|
613
|
+
// {§provider-interface} The worker identity is required.
|
|
546
614
|
if (workerId === undefined || workerId.length === 0) throw new Error("generate: workerId is required — the worker's stable, opaque identity");
|
|
547
|
-
// Reject before any wire call when already aborted
|
|
615
|
+
// Reject before any wire call when already aborted
|
|
616
|
+
// ({§provider-failure-normalization}).
|
|
548
617
|
signal?.throwIfAborted();
|
|
549
618
|
|
|
550
|
-
// Grammar handling (
|
|
551
|
-
// grammar
|
|
552
|
-
// model generates UNCONSTRAINED — and the free output is still verified
|
|
553
|
-
// against the grammar (below), surfacing exactly where the model's natural
|
|
554
|
-
// output and the grammar conflict. Otherwise the grammar is sent when the
|
|
555
|
-
// backend supports it (grammarStyle !== "none").
|
|
619
|
+
// Grammar handling ({§gbnf-response-observation}). Debug validates the
|
|
620
|
+
// supplied grammar before the call but withholds it from the backend.
|
|
556
621
|
const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
|
|
557
622
|
if (wantGrammar && this.#gbnfDebug) this.#assertGrammarValid(grammar!);
|
|
558
623
|
const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
|
|
559
624
|
|
|
560
625
|
// Assembly order = precedence: the family's sampling DEFAULTS
|
|
561
|
-
// (PLURNK_PROVIDERS_TEMPERATURE — universal,
|
|
626
|
+
// (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
|
|
562
627
|
// paths and the name promises every request) < the caller's `sampling`
|
|
563
628
|
// < the managed fields, which always win.
|
|
564
629
|
const body: Record<string, unknown> = {
|
|
565
|
-
//
|
|
630
|
+
// Floors are suppressed on router-owned-tuning providers (plurnk) —
|
|
566
631
|
// the router's per-model tuning must not be overridden by client floors.
|
|
567
632
|
...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
|
|
568
633
|
...this.#samplingBody(sampling),
|
|
@@ -572,11 +637,11 @@ export default class AiSdkProvider implements Provider {
|
|
|
572
637
|
...this.#reasoningBody(),
|
|
573
638
|
...this.#grammarBody(sendGrammar),
|
|
574
639
|
...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
|
|
575
|
-
//
|
|
640
|
+
// Request per-token logprobs only when enabled (managed field —
|
|
576
641
|
// reserved from caller sampling; the env flag is the single control).
|
|
577
642
|
...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
|
|
578
643
|
...this.#slotBody(workerId),
|
|
579
|
-
//
|
|
644
|
+
// Prompt-cache affinity -- workerId as the OpenAI-standard
|
|
580
645
|
// prompt_cache_key routes a worker's turns to one serverless replica so
|
|
581
646
|
// its stable prefix caches (managed; reserved from caller sampling).
|
|
582
647
|
...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
|
|
@@ -636,14 +701,14 @@ export default class AiSdkProvider implements Provider {
|
|
|
636
701
|
});
|
|
637
702
|
} catch (err) {
|
|
638
703
|
if (signal?.aborted) throw err;
|
|
639
|
-
const pe = toProviderError(err, this.#source);
|
|
704
|
+
const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
|
|
640
705
|
if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
|
|
641
706
|
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
|
|
642
707
|
}
|
|
643
708
|
throw pe;
|
|
644
709
|
}
|
|
645
710
|
|
|
646
|
-
//
|
|
711
|
+
// llama-server --special renders EOG tokens as text, so a turn ending
|
|
647
712
|
// via raw EOS carries a trailing <eos> the grammar never sanctioned - it both
|
|
648
713
|
// false-rejects the rail verdict and leaks a control token into the packet.
|
|
649
714
|
// Strip the server-reported eos_token from the tail ONCE, before the verdict
|
|
@@ -651,66 +716,125 @@ export default class AiSdkProvider implements Provider {
|
|
|
651
716
|
// wire text for forensics.
|
|
652
717
|
if (this.#eosText !== undefined) raw.content = stripTrailingSpecial(raw.content, this.#eosText);
|
|
653
718
|
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
if (
|
|
667
|
-
|
|
719
|
+
const taggedReasoning = projectTaggedReasoning(
|
|
720
|
+
raw.content,
|
|
721
|
+
raw.reasoning,
|
|
722
|
+
this.#reasoningResponseStyle,
|
|
723
|
+
);
|
|
724
|
+
|
|
725
|
+
// Preserve the exact sentence seen at the grammar boundary. llama-server's
|
|
726
|
+
// `reasoning_format: "auto"` projects one raw Harmony enclosure into the
|
|
727
|
+
// reasoning/content fields; the wire field's presence is the proof that the
|
|
728
|
+
// projection occurred. The provider represents this evidence and never grades it.
|
|
729
|
+
let grammarEvidence: GrammarEvidence | undefined;
|
|
730
|
+
if (wantGrammar) {
|
|
731
|
+
if (taggedReasoning.projected) {
|
|
732
|
+
grammarEvidence = {
|
|
733
|
+
input: raw.content,
|
|
734
|
+
contentStart: taggedReasoning.contentStart,
|
|
735
|
+
transported: sendGrammar !== undefined,
|
|
736
|
+
};
|
|
737
|
+
} else if (this.#reasoningStyle === "template" && this.#reasoning.mode !== "off") {
|
|
738
|
+
if (raw.reasoningProjected) {
|
|
739
|
+
const prefix = `<|channel>thought\n${raw.reasoning}<channel|>`;
|
|
740
|
+
grammarEvidence = {
|
|
741
|
+
input: `${prefix}${raw.content}`,
|
|
742
|
+
contentStart: [...prefix].length,
|
|
743
|
+
transported: sendGrammar !== undefined,
|
|
744
|
+
};
|
|
745
|
+
}
|
|
746
|
+
} else {
|
|
747
|
+
grammarEvidence = {
|
|
748
|
+
input: raw.content,
|
|
749
|
+
contentStart: 0,
|
|
750
|
+
transported: sendGrammar !== undefined,
|
|
751
|
+
};
|
|
668
752
|
}
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
753
|
+
}
|
|
754
|
+
|
|
755
|
+
if (taggedReasoning.projected) {
|
|
756
|
+
raw.content = taggedReasoning.content;
|
|
757
|
+
raw.reasoning = taggedReasoning.reasoning;
|
|
758
|
+
raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
|
|
759
|
+
}
|
|
760
|
+
|
|
761
|
+
let notices: ProviderNotice[] | undefined;
|
|
762
|
+
const usage = raw.usage;
|
|
763
|
+
if (sendGrammar !== undefined && this.tokenize !== undefined) {
|
|
764
|
+
// Channel-escape detector: completion tokens
|
|
675
765
|
// billed far beyond every visible channel mean the decode ESCAPED into
|
|
676
|
-
// a server-discarded reasoning block mid-emission
|
|
677
|
-
//
|
|
678
|
-
|
|
679
|
-
|
|
680
|
-
|
|
681
|
-
|
|
682
|
-
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
|
|
686
|
-
|
|
687
|
-
|
|
766
|
+
// a server-discarded reasoning block mid-emission. This diagnostic
|
|
767
|
+
// requires the serving vocabulary; an estimate cannot prove absence.
|
|
768
|
+
try {
|
|
769
|
+
const [contentTokens, reasoningTokens] = await Promise.all([
|
|
770
|
+
this.tokenize(raw.content),
|
|
771
|
+
this.tokenize(raw.reasoning),
|
|
772
|
+
]);
|
|
773
|
+
const visible = contentTokens.length + reasoningTokens.length;
|
|
774
|
+
if (usage.completion > visible + 64) {
|
|
775
|
+
(notices ??= []).push({
|
|
776
|
+
source: this.#source,
|
|
777
|
+
kind: "grammar_unenforced",
|
|
778
|
+
level: "warn",
|
|
779
|
+
message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
|
|
780
|
+
position: [...raw.content].length,
|
|
781
|
+
});
|
|
782
|
+
}
|
|
783
|
+
} catch (cause) {
|
|
784
|
+
emitWarningOnce(
|
|
785
|
+
`${this.#source}: exact visible-token diagnostic unavailable (${cause instanceof Error ? cause.message : String(cause)})`,
|
|
786
|
+
"PLURNK_VISIBLE_TOKEN_COUNT_UNAVAILABLE",
|
|
787
|
+
);
|
|
688
788
|
}
|
|
689
789
|
}
|
|
690
790
|
|
|
691
|
-
const
|
|
692
|
-
const meta = railsMeta !== undefined ? { ...builtMeta, ...railsMeta } : builtMeta;
|
|
791
|
+
const meta = this.#buildMeta(raw.metadata);
|
|
693
792
|
const logprobs = raw.logprobs.length > 0 ? raw.logprobs : undefined;
|
|
694
793
|
const meanLogprob = logprobs !== undefined
|
|
695
794
|
? logprobs.reduce((sum, token) => sum + token.logprob, 0) / logprobs.length
|
|
696
795
|
: undefined;
|
|
697
796
|
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
|
|
701
|
-
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
},
|
|
797
|
+
const assistant = {
|
|
798
|
+
content: raw.content,
|
|
799
|
+
reasoning: raw.reasoning.length > 0 ? raw.reasoning : null,
|
|
800
|
+
...(raw.reasoningEncrypted.length > 0
|
|
801
|
+
? { reasoningEncrypted: raw.reasoningEncrypted }
|
|
802
|
+
: {}),
|
|
803
|
+
usage,
|
|
804
|
+
model: raw.model,
|
|
805
|
+
...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
|
|
806
|
+
};
|
|
807
|
+
const evidence = {
|
|
710
808
|
assistantRaw: raw,
|
|
809
|
+
...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
|
|
711
810
|
...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
|
|
712
811
|
...(meta !== undefined ? { meta } : {}),
|
|
713
|
-
...(
|
|
812
|
+
...(notices !== undefined ? { notices } : {}),
|
|
813
|
+
};
|
|
814
|
+
if (raw.finishReason === "resource_interrupted") {
|
|
815
|
+
const attempt: ProviderResponse<"resource_interrupted"> = {
|
|
816
|
+
assistant: { ...assistant, finishReason: raw.finishReason },
|
|
817
|
+
...evidence,
|
|
818
|
+
};
|
|
819
|
+
throw new ProviderError(
|
|
820
|
+
this.#source,
|
|
821
|
+
"resource_interrupted",
|
|
822
|
+
"The provider interrupted generation because inference resources were unavailable.",
|
|
823
|
+
{
|
|
824
|
+
attempt,
|
|
825
|
+
extensions: {
|
|
826
|
+
stage: "provider-response",
|
|
827
|
+
finishReason: "resource_interrupted",
|
|
828
|
+
...(raw.rawFinishReason === undefined
|
|
829
|
+
? {}
|
|
830
|
+
: { rawFinishReason: raw.rawFinishReason }),
|
|
831
|
+
},
|
|
832
|
+
},
|
|
833
|
+
);
|
|
834
|
+
}
|
|
835
|
+
return {
|
|
836
|
+
assistant: { ...assistant, finishReason: raw.finishReason },
|
|
837
|
+
...evidence,
|
|
714
838
|
};
|
|
715
839
|
}
|
|
716
840
|
}
|