@plurnk/plurnk-providers 1.3.12 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +47 -38
- package/README.md +68 -4
- package/SPEC.md +245 -62
- package/dist/AiSdkProvider.d.ts +19 -5
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +264 -156
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +15 -17
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +27 -10
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +8 -2
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +41 -8
- package/dist/Pool.js.map +1 -1
- package/dist/ProviderRegistry.d.ts +4 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +7 -3
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/accounting.d.ts +3 -0
- package/dist/accounting.d.ts.map +1 -0
- package/dist/accounting.js +84 -0
- package/dist/accounting.js.map +1 -0
- package/dist/aiSdkTransport.d.ts +5 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +91 -5
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +5 -2
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +24 -11
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +11 -4
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +11 -0
- package/dist/cost.d.ts.map +1 -0
- package/dist/cost.js +61 -0
- package/dist/cost.js.map +1 -0
- package/dist/discover.d.ts +2 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js +15 -9
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +4 -6
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +27 -29
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +27 -0
- package/dist/errors.d.ts.map +1 -0
- package/dist/errors.js +152 -0
- package/dist/errors.js.map +1 -0
- package/dist/index.d.ts +12 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +8 -5
- package/dist/index.js.map +1 -1
- package/dist/notices.d.ts +10 -0
- package/dist/notices.d.ts.map +1 -0
- package/dist/notices.js +11 -0
- package/dist/notices.js.map +1 -0
- package/dist/ollama.d.ts.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/promptTokens.d.ts +4 -0
- package/dist/promptTokens.d.ts.map +1 -0
- package/dist/promptTokens.js +32 -0
- package/dist/promptTokens.js.map +1 -0
- package/dist/sdkModels.d.ts +2 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +17 -6
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +52 -16
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +1 -1
- package/dist/types.js.map +1 -1
- package/dist/usage.d.ts +4 -0
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +64 -19
- package/dist/usage.js.map +1 -1
- package/dist/warnings.js +0 -0
- package/dist/warnings.js.map +1 -1
- package/package.json +15 -10
- package/src/AiSdkProvider.test.ts +750 -169
- package/src/AiSdkProvider.ts +354 -200
- package/src/Mock.test.ts +29 -14
- package/src/Mock.ts +36 -15
- package/src/Pool.test.ts +43 -6
- package/src/Pool.ts +56 -10
- package/src/ProviderRegistry.test.ts +158 -9
- package/src/ProviderRegistry.ts +19 -6
- package/src/accounting.test.ts +58 -0
- package/src/accounting.ts +88 -0
- package/src/aiSdkTransport.ts +101 -8
- package/src/boundaries.test.ts +9 -3
- package/src/catalogProvider.test.ts +43 -15
- package/src/catalogProvider.ts +32 -16
- package/src/compatibleProvider.test.ts +96 -0
- package/src/compatibleProvider.ts +15 -6
- package/src/cost.test.ts +64 -0
- package/src/cost.ts +78 -0
- package/src/defaults.test.ts +1 -0
- package/src/discover.test.ts +48 -7
- package/src/discover.ts +31 -21
- package/src/env.test.ts +38 -48
- package/src/env.ts +43 -40
- package/src/errors.test.ts +148 -0
- package/src/errors.ts +208 -0
- package/src/index.ts +30 -8
- package/src/lexicon-guard.test.ts +6 -6
- package/src/notices.ts +22 -0
- package/src/ollama.test.ts +64 -0
- package/src/ollama.ts +6 -3
- package/src/openai.ts +3 -0
- package/src/promptTokens.ts +41 -0
- package/src/sdkModels.test.ts +29 -3
- package/src/sdkModels.ts +19 -11
- package/src/types.ts +125 -64
- package/src/usage.test.ts +24 -5
- package/src/usage.ts +72 -21
- package/src/warnings.test.ts +10 -10
- package/src/warnings.ts +0 -0
- package/dist/OpenAICompat.d.ts +0 -76
- package/dist/OpenAICompat.d.ts.map +0 -1
- package/dist/OpenAICompat.js +0 -555
- package/dist/OpenAICompat.js.map +0 -1
- package/dist/openaiStream.d.ts +0 -47
- package/dist/openaiStream.d.ts.map +0 -1
- package/dist/openaiStream.js +0 -280
- package/dist/openaiStream.js.map +0 -1
- package/dist/standardProviders.d.ts +0 -31
- package/dist/standardProviders.d.ts.map +0 -1
- package/dist/standardProviders.js +0 -518
- package/dist/standardProviders.js.map +0 -1
- package/dist/telemetry.d.ts +0 -24
- package/dist/telemetry.d.ts.map +0 -1
- package/dist/telemetry.js +0 -85
- package/dist/telemetry.js.map +0 -1
- package/src/telemetry.test.ts +0 -69
- package/src/telemetry.ts +0 -116
package/src/AiSdkProvider.ts
CHANGED
|
@@ -6,28 +6,25 @@
|
|
|
6
6
|
// ordinary vendor protocol. The compatible URL path remains only for PLURNK
|
|
7
7
|
// extensions and local endpoint probes the SDK cannot represent.
|
|
8
8
|
|
|
9
|
-
import type { ChatMessage,
|
|
10
|
-
import type {
|
|
9
|
+
import type { AuthoritativeChargeNormalizer, ChatMessage, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
10
|
+
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
11
|
+
import type { Reasoning, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
|
|
11
12
|
import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.ts";
|
|
12
13
|
import type { LanguageModel } from "ai";
|
|
13
|
-
import { toProviderError, ProviderError
|
|
14
|
-
import {
|
|
14
|
+
import { toProviderError, ProviderError } from "./errors.ts";
|
|
15
|
+
import { attributeUnitemizedReasoning } from "./usage.ts";
|
|
16
|
+
import type { ProviderNotice } from "./notices.ts";
|
|
17
|
+
import { validateGbnf } from "@plurnk/gbnf";
|
|
18
|
+
import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.ts";
|
|
15
19
|
import { emitWarningOnce } from "./warnings.ts";
|
|
20
|
+
import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
21
|
+
import { validateAuthoritativeCharge } from "./cost.ts";
|
|
16
22
|
|
|
17
23
|
export type ProviderFetch = typeof globalThis.fetch;
|
|
18
24
|
|
|
19
|
-
//
|
|
20
|
-
//
|
|
21
|
-
|
|
22
|
-
// "template" ALWAYS emits enable_thinking — the explicit false is llama-server's
|
|
23
|
-
// only working off-switch (§13); "anthropic" uses the `thinking` object and IGNORES
|
|
24
|
-
// reasoning_effort; "effort_explicit" (fireworks) sends the EXPLICIT "none" for OFF
|
|
25
|
-
// instead of omitting — reason-by-DEFAULT models (DeepSeek V4 defaults
|
|
26
|
-
// 'high') keep reasoning when the field is omitted, fatal under an active grammar
|
|
27
|
-
// (#30). Intent maps IDENTICALLY with or without a transported grammar — fireworks
|
|
28
|
-
// masks only the content channel, so reasoning and rails coexist in one call
|
|
29
|
-
// (canary-verified; the #32 clamp is lifted — it caused the plan-less service#331).
|
|
30
|
-
export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "template" | "anthropic";
|
|
25
|
+
// Backend wire spellings for the resolved reasoning intent. The switch beside each
|
|
26
|
+
// mapping retains any backend-specific omission/explicit-disable constraint.
|
|
27
|
+
export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "thinking_effort" | "template" | "anthropic";
|
|
31
28
|
|
|
32
29
|
// GBNF transport is a local llama-server capability. "none" means no
|
|
33
30
|
// service-managed constrained sampling; endpoint-owned settings are not inferred.
|
|
@@ -37,17 +34,21 @@ export type AiSdkProviderConfig = {
|
|
|
37
34
|
model: string;
|
|
38
35
|
url?: string; // OpenAI-compatible chat-completions URL
|
|
39
36
|
languageModel?: LanguageModel; // native AI SDK provider model
|
|
37
|
+
attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
40
38
|
fetchTimeoutMs: number;
|
|
41
39
|
streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
|
|
42
40
|
headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
|
|
43
41
|
fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
|
|
44
|
-
contextWindow?: number | null; // default null; caller resolves-or-fails
|
|
42
|
+
contextWindow?: number | null; // default null; caller resolves-or-fails, narrows to required with the interface
|
|
45
43
|
reasoningStyle?: ReasoningStyle; // default "none"
|
|
46
|
-
|
|
44
|
+
reasoningResponseStyle?: ReasoningResponseStyle; // {§provider-tagged-reasoning}; default "verbatim"
|
|
45
|
+
countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
|
|
47
46
|
calculateCost?: (usage: ProviderUsage) => number; // default () => 0
|
|
48
|
-
|
|
47
|
+
calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
|
|
48
|
+
normalizeCharge?: AuthoritativeChargeNormalizer;
|
|
49
|
+
source?: string; // notice/problem source, e.g. "provider:openai"; default "provider"
|
|
49
50
|
grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
|
|
50
|
-
//
|
|
51
|
+
// Send the OpenAI-standard `prompt_cache_key` set to workerId, so a
|
|
51
52
|
// serverless backend with REPLICA-LOCAL prompt caching (fireworks, verified)
|
|
52
53
|
// pins a worker's turns to one replica and claims its stable prefix. Default
|
|
53
54
|
// false -- a backend that strict-validates unknown fields 400s, so enable only
|
|
@@ -59,21 +60,24 @@ export type AiSdkProviderConfig = {
|
|
|
59
60
|
gbnfDebug?: boolean; // PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar locally + throw on invalid, but DON'T transport it (run unconstrained); default false
|
|
60
61
|
streaming?: boolean; // SSE transport (default true); false → one non-streamed JSON
|
|
61
62
|
firstPartyMetadata?: boolean; // forward per-turn attributions + client as Plurnk-* headers (plurnk only); default false
|
|
62
|
-
apiKeyRejectedMessage?: string; //
|
|
63
|
-
eosText?: string; //
|
|
64
|
-
// Slot affinity wiring
|
|
63
|
+
apiKeyRejectedMessage?: string; // friendly hint when a present key is 401/403-rejected (distinct from unset); default undefined
|
|
64
|
+
eosText?: string; // server-reported eos_token, stripped from the content tail (--special renders it as text); default undefined
|
|
65
|
+
// Slot affinity wiring is provider-internal, never consumer-facing.
|
|
65
66
|
supportsSlotPinning?: boolean; // backend accepts an `id_slot` body field (llama-server); default false
|
|
66
67
|
slotCount?: number | null; // probed slot count for pinning backends; default null
|
|
67
68
|
// Backend-served exact tokenization (llama-server /tokenize). When set, the
|
|
68
69
|
// provider exposes the optional `tokenize()` capability — the model's OWN
|
|
69
70
|
// vocab, no client-side tokenizer data needed; default unset (capability absent).
|
|
70
71
|
tokenizeUrl?: string;
|
|
71
|
-
//
|
|
72
|
+
// Provider-authoritative count of a complete chat-completions request. This
|
|
73
|
+
// is distinct from /tokenize, which sees content but not the chat template.
|
|
74
|
+
promptTokensUrl?: string;
|
|
75
|
+
// The backend's self-reported served model id (from the /v1/models probe),
|
|
72
76
|
// surfaced as Provider.servedModel. For a local llama-server the wire `model` is
|
|
73
77
|
// the alias; this is the real name (the .gguf) the tokenizer seam maps. Absent
|
|
74
78
|
// when no probe ran or it read no row.
|
|
75
79
|
servedModel?: string;
|
|
76
|
-
//
|
|
80
|
+
// Backend decodes unbounded without a caller cap (llama-server n_predict
|
|
77
81
|
// to the wall) — surfaced as Provider.requiresMaxTokens so consumers can
|
|
78
82
|
// boot-refuse an envelope-less local alias. Default unset (no claim).
|
|
79
83
|
requiresMaxTokens?: boolean;
|
|
@@ -81,36 +85,39 @@ export type AiSdkProviderConfig = {
|
|
|
81
85
|
// (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
|
|
82
86
|
// { mode: off|adaptive|on, budget: iff on }. The provider maps it to the
|
|
83
87
|
// backend's mechanism via reasoningStyle; budget is only ever a magnitude,
|
|
84
|
-
// never a hidden activation flag
|
|
88
|
+
// never a hidden activation flag.
|
|
85
89
|
reasoning: Reasoning;
|
|
86
90
|
// Decode tuning: no in-code defaults; the canonical measured values (0.2 /
|
|
87
91
|
// 1.15) ship as the floor in .env.defaults (alias-scopable). `temperature` is the
|
|
88
|
-
// DEFAULT for EVERY request, spread UNDER caller sampling
|
|
92
|
+
// DEFAULT for EVERY request, spread UNDER caller sampling.
|
|
89
93
|
// `repeatPenalty` is the FLOOR the provider manages wherever a grammar rides
|
|
90
|
-
// (greedy-under-mask loops without it
|
|
94
|
+
// (greedy-under-mask loops without it) — the VALUE is operator config;
|
|
91
95
|
// WHERE it applies stays mechanism.
|
|
92
96
|
temperature: number;
|
|
93
97
|
repeatPenalty: number;
|
|
94
|
-
//
|
|
98
|
+
// Anti-degeneration guard on the cloud path (grammarStyle "none"), where the
|
|
95
99
|
// repeat_penalty multiplier isn't available - the OpenAI-standard frequency_penalty.
|
|
96
100
|
// Optional (default 0 = off) so an out-of-date plugin that omits it just runs unguarded
|
|
97
101
|
// rather than failing construction; the standard factory always supplies it.
|
|
98
102
|
frequencyPenalty?: number;
|
|
99
|
-
//
|
|
100
|
-
//
|
|
101
|
-
//
|
|
102
|
-
//
|
|
103
|
-
// like repeatPenalty — applied for a detected llama.cpp backend, customer-overridable;
|
|
104
|
-
// absent (a plugin omitting them) = the box's own default. repeatLastN widens the window.
|
|
103
|
+
// Optional llama.cpp anti-repetition controls. DRY can suppress long
|
|
104
|
+
// repeated sequences, but it can also corrupt exact repetition required by
|
|
105
|
+
// PLURNK operations. The portable default is off; these fields ride only
|
|
106
|
+
// after an explicit operator opt-in. repeatLastN widens the repeat window.
|
|
105
107
|
dryMultiplier?: number;
|
|
106
108
|
dryBase?: number;
|
|
107
109
|
dryAllowedLength?: number;
|
|
108
110
|
repeatLastN?: number;
|
|
109
111
|
// Transient-failure retry budget — REQUIRED, no in-code default
|
|
110
112
|
// (PLURNK_PROVIDERS_RETRY_ATTEMPTS, a non-negative int): 0 = surface the
|
|
111
|
-
// first failure; N = up to N retries on a transient error
|
|
113
|
+
// first failure; N = up to N retries on a transient error
|
|
114
|
+
// ({§provider-failure-normalization}).
|
|
112
115
|
retryAttempts: number;
|
|
113
|
-
//
|
|
116
|
+
// Maximum characters retained from an upstream diagnostic in the public
|
|
117
|
+
// ProviderError Problem. Standard factories supply the env-owned value;
|
|
118
|
+
// direct construction may omit it to preserve the complete diagnostic.
|
|
119
|
+
errorDetailLimit?: number;
|
|
120
|
+
// Data-capture knobs ({§provider-evidence}), off by default — the flag is the isolation, so a
|
|
114
121
|
// serving turn requests nothing and carries nothing. `topLogprobs`: when a
|
|
115
122
|
// non-negative int, request `logprobs:true, top_logprobs:<n>` and surface the
|
|
116
123
|
// per-token confidence on assistant.logprobs (PLURNK_PROVIDERS_TOP_LOGPROBS;
|
|
@@ -119,7 +126,7 @@ export type AiSdkProviderConfig = {
|
|
|
119
126
|
// gated per-alias.
|
|
120
127
|
topLogprobs?: number | null;
|
|
121
128
|
rawBody?: boolean;
|
|
122
|
-
//
|
|
129
|
+
// {§provider-generation-envelope} The generation-envelope reserves, env-read via
|
|
123
130
|
// envelopeFromEnv — a percentage of the DETECTED window or an absolute token
|
|
124
131
|
// count. Optional so an out-of-date sibling keeps constructing (no claim);
|
|
125
132
|
// the standard factory always supplies them. Resolved against contextWindow
|
|
@@ -127,13 +134,13 @@ export type AiSdkProviderConfig = {
|
|
|
127
134
|
// derives correctly.
|
|
128
135
|
reasoningReserve?: ReserveSpec;
|
|
129
136
|
completionReserve?: ReserveSpec;
|
|
130
|
-
//
|
|
137
|
+
// The plurnk.ai router owns tuning — false suppresses the
|
|
131
138
|
// client-side temperature/penalty FLOORS on this provider (caller `sampling`
|
|
132
139
|
// still passes through verbatim). Default true (floors ride).
|
|
133
140
|
tuningFloors?: boolean;
|
|
134
141
|
};
|
|
135
142
|
|
|
136
|
-
//
|
|
143
|
+
// Drop trailing occurrences of a server-rendered EOG marker. llama-server
|
|
137
144
|
// under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
|
|
138
145
|
// trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
|
|
139
146
|
// can never eat body content (a body ending in the literal marker isn't producible
|
|
@@ -145,6 +152,58 @@ const stripTrailingSpecial = (content: string, marker: string): string => {
|
|
|
145
152
|
return out;
|
|
146
153
|
};
|
|
147
154
|
|
|
155
|
+
type TaggedReasoningProjection = {
|
|
156
|
+
readonly content: string;
|
|
157
|
+
readonly reasoning: string;
|
|
158
|
+
readonly projected: boolean;
|
|
159
|
+
readonly contentStart: number;
|
|
160
|
+
};
|
|
161
|
+
|
|
162
|
+
const projectLeadingReasoning = (
|
|
163
|
+
content: string,
|
|
164
|
+
structuredReasoning: string,
|
|
165
|
+
opening: string,
|
|
166
|
+
closing: string,
|
|
167
|
+
): TaggedReasoningProjection => {
|
|
168
|
+
if (structuredReasoning.length > 0 || !content.startsWith(opening)) {
|
|
169
|
+
return { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
|
|
170
|
+
}
|
|
171
|
+
const closingIndex = content.indexOf(closing, opening.length);
|
|
172
|
+
if (closingIndex === -1) {
|
|
173
|
+
return {
|
|
174
|
+
content: "",
|
|
175
|
+
reasoning: content.slice(opening.length),
|
|
176
|
+
projected: true,
|
|
177
|
+
contentStart: [...content].length,
|
|
178
|
+
};
|
|
179
|
+
}
|
|
180
|
+
const suffixStart = closingIndex + closing.length;
|
|
181
|
+
return {
|
|
182
|
+
content: content.slice(suffixStart),
|
|
183
|
+
reasoning: content.slice(opening.length, closingIndex),
|
|
184
|
+
projected: true,
|
|
185
|
+
contentStart: [...content.slice(0, suffixStart)].length,
|
|
186
|
+
};
|
|
187
|
+
};
|
|
188
|
+
|
|
189
|
+
// {§provider-tagged-reasoning} Only the model-contract position is structural:
|
|
190
|
+
// one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
|
|
191
|
+
// on one path and leaves later literal tags in the visible suffix untouched.
|
|
192
|
+
const projectTaggedReasoning = (
|
|
193
|
+
content: string,
|
|
194
|
+
structuredReasoning: string,
|
|
195
|
+
style: ReasoningResponseStyle,
|
|
196
|
+
): TaggedReasoningProjection => style === "think-tags"
|
|
197
|
+
? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
|
|
198
|
+
: { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
|
|
199
|
+
|
|
200
|
+
// llama-server's template reasoning parser can project this leading channel out
|
|
201
|
+
// of the OpenAI-compatible response. Grammar evidence needs the sentence before
|
|
202
|
+
// that lossy projection, so constrained template turns request it verbatim and
|
|
203
|
+
// split the observed enclosure here.
|
|
204
|
+
const projectTemplateReasoning = (content: string): TaggedReasoningProjection =>
|
|
205
|
+
projectLeadingReasoning(content, "", "<|channel>thought\n", "<channel|>");
|
|
206
|
+
|
|
148
207
|
// Shared budget→effort breakpoints (xai and google had identical copies).
|
|
149
208
|
export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
|
|
150
209
|
if (budget <= 1000) return "low";
|
|
@@ -152,44 +211,28 @@ export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
|
|
|
152
211
|
return "high";
|
|
153
212
|
};
|
|
154
213
|
|
|
155
|
-
// chars/2 upper bound (see ./tokenizers.ts) — overcounts safely, never under.
|
|
156
|
-
const heuristicTokens = (text: string): number => (text.length === 0 ? 0 : Math.ceil(text.length / 2));
|
|
157
|
-
|
|
158
214
|
// Body keys the provider owns — a caller's `sampling` passthrough may not set
|
|
159
|
-
// these. Two families
|
|
215
|
+
// these. Two families:
|
|
160
216
|
// transport/managed — grammar transport, the stream/JSON choice, slot pinning,
|
|
161
|
-
// data capture (
|
|
217
|
+
// data capture ({§provider-evidence}: backend-specific fields never cross the contract);
|
|
162
218
|
// contract invariants — `n` (atomic single completion: choices[0] is the
|
|
163
219
|
// response; n>1 = paid, dropped output), the tool-calling family (tools-in-
|
|
164
220
|
// body doctrine, §2: native tool_calls return null content = a broken turn),
|
|
165
221
|
// modalities/audio (text-only contract), prediction (decode semantics, not
|
|
166
222
|
// sampling), and the token caps (the envelope is the managed maxTokens —
|
|
167
|
-
// sampling must not bypass the consumer's
|
|
223
|
+
// sampling must not bypass the consumer's cap).
|
|
168
224
|
// Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
|
|
169
225
|
// platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
|
|
170
226
|
// metadata, store, verbosity) pass through; the managed floors spread UNDER
|
|
171
227
|
// sampling stay deliberately caller-overridable.
|
|
172
228
|
const RESERVED_BODY_KEYS: ReadonlySet<string> = new Set([
|
|
173
229
|
"model", "messages", "stream", "stream_options", "grammar", "response_format", "id_slot", "logprobs", "top_logprobs",
|
|
230
|
+
"reasoning_format", "reasoning_effort", "thinking", "think", "include_reasoning", "chat_template_kwargs", "thinking_budget_tokens", // lexicon-allow: backend wire fields
|
|
174
231
|
"n", "tools", "tool_choice", "functions", "function_call", "parallel_tool_calls",
|
|
175
232
|
"modalities", "audio", "prediction", "max_tokens", "max_completion_tokens",
|
|
176
233
|
"prompt_cache_key",
|
|
177
234
|
]);
|
|
178
235
|
|
|
179
|
-
// Render a non-accept verdict into a terse, factual grammar_unenforced message
|
|
180
|
-
// (SPEC §12 message policy: no guidance prose). `reject` names the diverging code
|
|
181
|
-
// point + what the grammar would have accepted; `incomplete` names the valid-prefix
|
|
182
|
-
// length that never reached a terminal state.
|
|
183
|
-
const describeUnenforced = (v: Exclude<Verdict, { status: "accept" }>): string => {
|
|
184
|
-
if (v.status === "reject") {
|
|
185
|
-
const expected = v.expected.length > 0
|
|
186
|
-
? v.expected.map((e) => `${e.rule} accepts ${e.accepts}`).join(", ")
|
|
187
|
-
: "end of input";
|
|
188
|
-
return `grammar not enforced: output rejected by the transported grammar at code point ${v.pos} (${JSON.stringify(v.char)}); expected ${expected}`;
|
|
189
|
-
}
|
|
190
|
-
return `grammar not enforced: output is an incomplete match of the transported grammar — a valid prefix of ${v.pos} code points that never terminated`;
|
|
191
|
-
};
|
|
192
|
-
|
|
193
236
|
export default class AiSdkProvider implements Provider {
|
|
194
237
|
#model: string;
|
|
195
238
|
#url: string | undefined;
|
|
@@ -211,8 +254,12 @@ export default class AiSdkProvider implements Provider {
|
|
|
211
254
|
#dryAllowedLength: number | undefined;
|
|
212
255
|
#repeatLastN: number | undefined;
|
|
213
256
|
#reasoningStyle: ReasoningStyle;
|
|
214
|
-
#
|
|
257
|
+
#reasoningResponseStyle: ReasoningResponseStyle;
|
|
258
|
+
#countPromptTokens: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
|
|
259
|
+
#promptTokensUrl: string | undefined;
|
|
215
260
|
#calculateCost: (usage: ProviderUsage) => number;
|
|
261
|
+
#calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
|
|
262
|
+
#normalizeCharge?: AuthoritativeChargeNormalizer;
|
|
216
263
|
#source: string;
|
|
217
264
|
#grammarStyle: GrammarStyle;
|
|
218
265
|
#promptCacheKey: boolean;
|
|
@@ -223,6 +270,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
223
270
|
#supportsSlotPinning: boolean;
|
|
224
271
|
#slotCount: number | null;
|
|
225
272
|
#retryAttempts: number;
|
|
273
|
+
#errorDetailLimit: number | undefined;
|
|
226
274
|
#topLogprobs: number | null;
|
|
227
275
|
#reasoningReserve: ReserveSpec | undefined;
|
|
228
276
|
#completionReserve: ReserveSpec | undefined;
|
|
@@ -230,17 +278,18 @@ export default class AiSdkProvider implements Provider {
|
|
|
230
278
|
#rawBody: boolean;
|
|
231
279
|
#servedModel: string | undefined;
|
|
232
280
|
#requiresMaxTokens: boolean | undefined;
|
|
281
|
+
readonly attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
233
282
|
|
|
234
|
-
// Optional capability (
|
|
283
|
+
// Optional capability ({§provider-local-capabilities}): exact tokenization served by the backend's
|
|
235
284
|
// own vocab. Assigned in the constructor ONLY when the config carries a
|
|
236
285
|
// tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
|
|
237
286
|
// the honest capability signal for every other backend.
|
|
238
287
|
tokenize?: (text: string) => Promise<number[]>;
|
|
239
|
-
|
|
240
288
|
constructor(config: AiSdkProviderConfig) {
|
|
241
289
|
this.#model = config.model;
|
|
242
290
|
this.#url = config.url;
|
|
243
291
|
this.#languageModel = config.languageModel;
|
|
292
|
+
this.attributions = config.attributions;
|
|
244
293
|
if ((this.#url === undefined) === (this.#languageModel === undefined)) {
|
|
245
294
|
throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
|
|
246
295
|
}
|
|
@@ -264,9 +313,17 @@ export default class AiSdkProvider implements Provider {
|
|
|
264
313
|
this.#dryAllowedLength = config.dryAllowedLength;
|
|
265
314
|
this.#repeatLastN = config.repeatLastN;
|
|
266
315
|
this.#retryAttempts = config.retryAttempts;
|
|
316
|
+
this.#errorDetailLimit = config.errorDetailLimit;
|
|
267
317
|
this.#reasoningStyle = config.reasoningStyle ?? "none";
|
|
268
|
-
this.#
|
|
318
|
+
this.#reasoningResponseStyle = config.reasoningResponseStyle ?? "verbatim";
|
|
319
|
+
if (config.countPromptTokens !== undefined && config.promptTokensUrl !== undefined) {
|
|
320
|
+
throw new Error(`${config.source ?? "provider"}: configure countPromptTokens or promptTokensUrl, not both`);
|
|
321
|
+
}
|
|
322
|
+
this.#countPromptTokens = config.countPromptTokens ?? ((messages) => estimatePromptTokens(messages));
|
|
323
|
+
this.#promptTokensUrl = config.promptTokensUrl;
|
|
269
324
|
this.#calculateCost = config.calculateCost ?? (() => 0);
|
|
325
|
+
this.#calculateCharge = config.calculateCharge;
|
|
326
|
+
this.#normalizeCharge = config.normalizeCharge;
|
|
270
327
|
this.#source = config.source ?? "provider";
|
|
271
328
|
this.#grammarStyle = config.grammarStyle ?? "none";
|
|
272
329
|
this.#promptCacheKey = config.promptCacheKey ?? false;
|
|
@@ -286,6 +343,13 @@ export default class AiSdkProvider implements Provider {
|
|
|
286
343
|
this.#rawBody = config.rawBody ?? false;
|
|
287
344
|
this.#servedModel = config.servedModel;
|
|
288
345
|
this.#requiresMaxTokens = config.requiresMaxTokens;
|
|
346
|
+
const reasoningReserve = this.reasoningReserve;
|
|
347
|
+
if (this.#reasoningStyle === "template"
|
|
348
|
+
&& this.#reasoning.mode === "on"
|
|
349
|
+
&& reasoningReserve !== null
|
|
350
|
+
&& this.#reasoning.budget! > reasoningReserve) {
|
|
351
|
+
throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
|
|
352
|
+
}
|
|
289
353
|
const { tokenizeUrl } = config;
|
|
290
354
|
if (tokenizeUrl !== undefined) {
|
|
291
355
|
this.tokenize = async (text: string): Promise<number[]> => {
|
|
@@ -306,7 +370,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
306
370
|
}
|
|
307
371
|
|
|
308
372
|
get contextWindow(): number | null { return this.#contextWindow; }
|
|
309
|
-
//
|
|
373
|
+
// {§provider-generation-envelope} Envelope reserves — absolute pins stand alone; percentages need the
|
|
310
374
|
// detected window; null = underivable (no claim for core's no-cap path).
|
|
311
375
|
#resolveReserve(spec: ReserveSpec | undefined): number | null {
|
|
312
376
|
if (spec === undefined) return null;
|
|
@@ -316,63 +380,110 @@ export default class AiSdkProvider implements Provider {
|
|
|
316
380
|
get reasoningReserve(): number | null { return this.#resolveReserve(this.#reasoningReserve); }
|
|
317
381
|
get completionReserve(): number | null { return this.#resolveReserve(this.#completionReserve); }
|
|
318
382
|
get model(): string { return this.#model; }
|
|
319
|
-
//
|
|
383
|
+
// Backend's self-reported served id; undefined when unprobed/unknown.
|
|
320
384
|
get servedModel(): string | undefined { return this.#servedModel; }
|
|
321
|
-
//
|
|
385
|
+
// Resolved "decodes unbounded without a cap" fact; undefined = no claim.
|
|
322
386
|
get requiresMaxTokens(): boolean | undefined { return this.#requiresMaxTokens; }
|
|
323
|
-
// Resolved capability
|
|
387
|
+
// Resolved capability: will a transported grammar actually constrain
|
|
324
388
|
// this backend's decode? Introspectable so a consumer can verify the rails
|
|
325
389
|
// are LIVE without spending a generation on a forcing-grammar probe.
|
|
326
390
|
get constrainsOutput(): boolean { return this.#grammarStyle !== "none"; }
|
|
327
391
|
|
|
328
|
-
|
|
392
|
+
async countPromptTokens(
|
|
393
|
+
messages: readonly ChatMessage[],
|
|
394
|
+
signal?: AbortSignal,
|
|
395
|
+
): Promise<PromptTokenMeasurement> {
|
|
396
|
+
if (this.#promptTokensUrl === undefined) {
|
|
397
|
+
return assertPromptTokenMeasurement(
|
|
398
|
+
await this.#countPromptTokens(messages, signal),
|
|
399
|
+
this.#source,
|
|
400
|
+
);
|
|
401
|
+
}
|
|
402
|
+
|
|
403
|
+
signal?.throwIfAborted();
|
|
404
|
+
try {
|
|
405
|
+
const timeout = AbortSignal.timeout(this.#fetchTimeoutMs);
|
|
406
|
+
const response = await this.#fetch(this.#promptTokensUrl, {
|
|
407
|
+
method: "POST",
|
|
408
|
+
headers: { "Content-Type": "application/json", ...this.#headers },
|
|
409
|
+
body: JSON.stringify({
|
|
410
|
+
model: this.#model,
|
|
411
|
+
messages,
|
|
412
|
+
...this.#reasoningBody(),
|
|
413
|
+
}),
|
|
414
|
+
signal: signal === undefined ? timeout : AbortSignal.any([signal, timeout]),
|
|
415
|
+
});
|
|
416
|
+
if (!response.ok) {
|
|
417
|
+
return estimatePromptTokens(
|
|
418
|
+
messages,
|
|
419
|
+
`llama-server input-token endpoint returned HTTP ${response.status}`,
|
|
420
|
+
);
|
|
421
|
+
}
|
|
422
|
+
const body = await response.json() as { input_tokens?: unknown };
|
|
423
|
+
if (!Number.isInteger(body.input_tokens) || (body.input_tokens as number) < 0) {
|
|
424
|
+
return estimatePromptTokens(
|
|
425
|
+
messages,
|
|
426
|
+
"llama-server input-token endpoint returned no non-negative integer input_tokens",
|
|
427
|
+
);
|
|
428
|
+
}
|
|
429
|
+
return {
|
|
430
|
+
kind: "exact",
|
|
431
|
+
tokens: body.input_tokens as number,
|
|
432
|
+
source: "llama-server:/v1/chat/completions/input_tokens",
|
|
433
|
+
};
|
|
434
|
+
} catch (cause) {
|
|
435
|
+
signal?.throwIfAborted();
|
|
436
|
+
return estimatePromptTokens(
|
|
437
|
+
messages,
|
|
438
|
+
`llama-server input-token measurement failed: ${cause instanceof Error ? cause.message : String(cause)}`,
|
|
439
|
+
);
|
|
440
|
+
}
|
|
441
|
+
}
|
|
329
442
|
calculateCost(usage: ProviderUsage): number { return this.#calculateCost(usage); }
|
|
443
|
+
calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> {
|
|
444
|
+
return this.#calculateCharge?.(usage)
|
|
445
|
+
?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
|
|
446
|
+
}
|
|
330
447
|
|
|
331
|
-
//
|
|
332
|
-
//
|
|
333
|
-
//
|
|
334
|
-
|
|
335
|
-
// reasoning channel rides beside it unmasked, and the plurnk grammar's
|
|
336
|
-
// reasoning?/preplan regions absorb any in-band spillover. The old measured
|
|
337
|
-
// failures (low→cycles, high→spirals) were pre-max_tokens-cap; with a bounded
|
|
338
|
-
// cap the matrix ACCEPTs across efforts (reasoning-rails matrix
|
|
339
|
-
// F9). Clamping was the root of the plan-less regression (service#331).
|
|
340
|
-
#reasoningBody(): Record<string, unknown> {
|
|
448
|
+
// Reasoning activation and allowance are independent of grammar transport;
|
|
449
|
+
// only the response representation becomes lossless when evidence is needed.
|
|
450
|
+
// The llama-server template mapping is owned by {§llama-reasoning-request}.
|
|
451
|
+
#reasoningBody(preserveGrammarSentence = false): Record<string, unknown> {
|
|
341
452
|
const { mode, budget } = this.#reasoning;
|
|
342
453
|
const on = mode !== "off";
|
|
343
454
|
switch (this.#reasoningStyle) {
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
// constrained (26-run baseline green; zero grammar rejects across
|
|
355
|
-
// the #488 "railless" specimens). Closing the channel starved a
|
|
356
|
-
// reasoning-tuned model into ESCAPING mid-content into the raw
|
|
357
|
-
// thought channel — discarded server-side, decode unconstrained,
|
|
358
|
-
// 12,288 tokens billed for 1,033 visible chars. The escape is
|
|
359
|
-
// surfaced instead (vanished-token telemetry + meta rail state).
|
|
360
|
-
case "template": return { chat_template_kwargs: { enable_thinking: on } };
|
|
455
|
+
case "template": {
|
|
456
|
+
const allowance = mode === "off"
|
|
457
|
+
? 0
|
|
458
|
+
: mode === "on" ? budget : this.reasoningReserve;
|
|
459
|
+
return {
|
|
460
|
+
chat_template_kwargs: { enable_thinking: on },
|
|
461
|
+
reasoning_format: preserveGrammarSentence ? "none" : "auto",
|
|
462
|
+
...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
|
|
463
|
+
};
|
|
464
|
+
}
|
|
361
465
|
case "think": return on ? { think: true } : {};
|
|
362
466
|
case "include_reasoning": return on ? { include_reasoning: true } : {};
|
|
363
467
|
// effort tiers from the budget; off/adaptive omit the field (the
|
|
364
468
|
// API's default depth is its adaptive).
|
|
365
469
|
case "effort": return mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
|
|
366
470
|
// Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
|
|
367
|
-
// reason-by-default model (DeepSeek V4: default 'high') reasoning
|
|
471
|
+
// reason-by-default model (DeepSeek V4: default 'high') reasoning.
|
|
368
472
|
// ADAPTIVE omits the field: the backend's own default posture IS the
|
|
369
473
|
// adaptive semantics, and the literal "adaptive" is MiniMax-M3-only —
|
|
370
|
-
//
|
|
474
|
+
// Fireworks 400s it for every other model (wire-verified; the
|
|
371
475
|
// 1.0.2 adaptive default refused to boot on it). V4 gotcha: integer
|
|
372
476
|
// efforts 400.
|
|
373
477
|
case "effort_explicit": return mode === "off"
|
|
374
478
|
? { reasoning_effort: "none" }
|
|
375
479
|
: mode === "on" ? { reasoning_effort: effortFromBudget(budget!) } : {};
|
|
480
|
+
// {§deepseek-reasoning-request}
|
|
481
|
+
case "thinking_effort": return mode === "off"
|
|
482
|
+
? { thinking: { type: "disabled" } }
|
|
483
|
+
: mode === "on" ? {
|
|
484
|
+
thinking: { type: "enabled" },
|
|
485
|
+
reasoning_effort: effortFromBudget(budget!),
|
|
486
|
+
} : {};
|
|
376
487
|
// Anthropic compat: explicit thinking object. off → disabled; on →
|
|
377
488
|
// enabled with budget_tokens; adaptive → omit (the API default).
|
|
378
489
|
case "anthropic": return mode === "off"
|
|
@@ -382,7 +493,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
382
493
|
}
|
|
383
494
|
}
|
|
384
495
|
|
|
385
|
-
// Per-
|
|
496
|
+
// Per-worker slot affinity: the consumer passes which worker this is; the
|
|
386
497
|
// provider owns WHICH slot serves it. Sticky per workerId, round-robin across
|
|
387
498
|
// new runs (distinct runs → distinct slots while slots last), LRU-bounded
|
|
388
499
|
// bookkeeping so a long-lived daemon never grows the map unboundedly —
|
|
@@ -405,19 +516,19 @@ export default class AiSdkProvider implements Provider {
|
|
|
405
516
|
return { id_slot: slot };
|
|
406
517
|
}
|
|
407
518
|
|
|
408
|
-
// Optional local llama-server GBNF transport (
|
|
519
|
+
// Optional local llama-server GBNF transport ({§gbnf-response-observation}). Unsupported
|
|
409
520
|
// backends receive no grammar-related field.
|
|
410
521
|
#grammarBody(grammar: string | undefined): Record<string, unknown> {
|
|
411
522
|
if (grammar === undefined) return {};
|
|
412
523
|
switch (this.#grammarStyle) {
|
|
413
524
|
// Greedy decoding under hard constraint loops without a repeat-penalty
|
|
414
|
-
// floor
|
|
525
|
+
// floor — llama.cpp spells it `repeat_penalty`.
|
|
415
526
|
case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
|
|
416
527
|
case "none": return {};
|
|
417
528
|
}
|
|
418
529
|
}
|
|
419
530
|
|
|
420
|
-
// Anti-degeneration
|
|
531
|
+
// Anti-degeneration default on every request, keyed to the backend's wire
|
|
421
532
|
// convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
|
|
422
533
|
// alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
|
|
423
534
|
// straight to the token cap on pure looped repetition (run52). Ships next to
|
|
@@ -428,7 +539,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
428
539
|
// live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
|
|
429
540
|
#repetitionPenaltyBody(): Record<string, unknown> {
|
|
430
541
|
switch (this.#grammarStyle) {
|
|
431
|
-
//
|
|
542
|
+
// repeat_penalty + optional DRY (repeated-sequence penalty) + a wider
|
|
432
543
|
// repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
|
|
433
544
|
// Each rides only when its operator knob is set; absent = the box's default.
|
|
434
545
|
case "llamacpp": return {
|
|
@@ -444,12 +555,12 @@ export default class AiSdkProvider implements Provider {
|
|
|
444
555
|
}
|
|
445
556
|
}
|
|
446
557
|
|
|
447
|
-
// First-party telemetry headers (
|
|
558
|
+
// First-party telemetry headers ({§provider-request-authority}): forwarded only when the spec
|
|
448
559
|
// opted in (the plurnk endpoint). The gate is here, not at the call site, so
|
|
449
560
|
// attributions/client/strikes can never reach a third-party backend even if
|
|
450
561
|
// the consumer passes them to the wrong provider. Empty values emit no header
|
|
451
562
|
// — EXCEPT strikes, where 0 is a real value (clean streak) distinct from
|
|
452
|
-
// absent (consumer didn't report); contract
|
|
563
|
+
// absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
|
|
453
564
|
// ride HTTP headers only — the packet never carries them (the model must
|
|
454
565
|
// never see strike state; engine accounting is not a metric to game).
|
|
455
566
|
#metadataHeaders(attributions: string[] | undefined, client: string | undefined, strikes: number | undefined, workerId: string, primaryWorkerId: string | undefined, workspaceId: string | undefined, loop: number | undefined, turn: number | undefined): Record<string, string> {
|
|
@@ -458,11 +569,11 @@ export default class AiSdkProvider implements Provider {
|
|
|
458
569
|
if (attributions !== undefined && attributions.length > 0) h["Plurnk-Attribution"] = JSON.stringify(attributions);
|
|
459
570
|
if (client !== undefined && client.length > 0) h["Plurnk-Client"] = client;
|
|
460
571
|
if (strikes !== undefined && Number.isInteger(strikes) && strikes >= 0) h["Plurnk-Strikes"] = String(strikes);
|
|
461
|
-
// Worker identity
|
|
572
|
+
// Worker identity: the opaque workerId
|
|
462
573
|
// the consumer already supplies, forwarded so the endpoint can key
|
|
463
574
|
// per-worker affinity/telemetry — same gate as every first-party signal.
|
|
464
575
|
h["Plurnk-Worker-Id"] = workerId;
|
|
465
|
-
// Root worker of the lineage (
|
|
576
|
+
// Root worker of the lineage ({§worker-primary}): the no-parent ancestor of this turn's
|
|
466
577
|
// worker tree. The consumer classifies primary-vs-spawned by equality
|
|
467
578
|
// (primaryWorkerId == workerId ⇒ the primary/root worker). The provider
|
|
468
579
|
// EMITS what the consumer supplies and never invents a primary; the
|
|
@@ -470,7 +581,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
470
581
|
// own, where it equals workerId). Absence is the consumer's violation for
|
|
471
582
|
// the endpoint to surface, not a provider default.
|
|
472
583
|
if (primaryWorkerId !== undefined && primaryWorkerId.length > 0) h["Plurnk-Worker-Primary"] = primaryWorkerId;
|
|
473
|
-
// Turn coordinate (
|
|
584
|
+
// Turn coordinate ({§lifecycle-terms}): workspace/loop/turn, the
|
|
474
585
|
// daemon-side sequence the endpoint can never scrape from the wire.
|
|
475
586
|
// Coordinates are 1-based — 0 is not a real value, so no strikes-style
|
|
476
587
|
// zero exception; absent/empty/0 emits no header.
|
|
@@ -480,33 +591,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
480
591
|
return h;
|
|
481
592
|
}
|
|
482
593
|
|
|
483
|
-
//
|
|
484
|
-
// (grammarStyle !== "none"), the backend MUST have constrained the output;
|
|
485
|
-
// some silently drop the grammar field or mislabel the channel, and without
|
|
486
|
-
// this check we would return unconstrained output as if enforced. STRICT: any
|
|
487
|
-
// non-accept verdict (reject, or an incomplete/never-terminated match) is a
|
|
488
|
-
// grammar_unenforced failure. A grammar our own validator can't parse — even
|
|
489
|
-
// though the backend accepted it (a port-vs-llama.cpp gap) — is a non-fatal
|
|
490
|
-
// verify gap: warn, don't fail a transport that may have worked. This is a
|
|
491
|
-
// conformance check against the grammar we already hold, NOT a plurnk-DSL
|
|
492
|
-
// parse (§8) — it stays grammar-generic and backend-agnostic.
|
|
493
|
-
// Validate output against the grammar. Returns the verdict, or null on the
|
|
494
|
-
// verify GAP — a grammar our own validator can't parse (a port-vs-llama.cpp
|
|
495
|
-
// gap): warn, don't manufacture a conflict from a check that didn't run.
|
|
496
|
-
#grammarVerdict(grammar: string, content: string): Verdict | null {
|
|
497
|
-
try {
|
|
498
|
-
return validateGbnf(grammar, content);
|
|
499
|
-
} catch (cause) {
|
|
500
|
-
// Once per (code, message) — #40: this fires PER TURN otherwise.
|
|
501
|
-
emitWarningOnce(
|
|
502
|
-
`${this.#source}: could not verify grammar enforcement — the transported grammar did not parse in @plurnk/gbnf (${(cause as Error).message})`,
|
|
503
|
-
"PLURNK_GRAMMAR_UNVERIFIABLE",
|
|
504
|
-
);
|
|
505
|
-
return null;
|
|
506
|
-
}
|
|
507
|
-
}
|
|
508
|
-
|
|
509
|
-
// PLURNK_PROVIDERS_GBNF_DEBUG (SPEC §13): validate the supplied GBNF locally and fail
|
|
594
|
+
// PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
|
|
510
595
|
// hard if it's malformed, BEFORE any wire call — and the grammar is NOT
|
|
511
596
|
// transported, so the request runs unconstrained. A debug aid to catch invalid
|
|
512
597
|
// grammars (e.g. while editing the plurnk grammar) without a model round-trip;
|
|
@@ -521,7 +606,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
521
606
|
}
|
|
522
607
|
}
|
|
523
608
|
|
|
524
|
-
// Per-turn metadata bag
|
|
609
|
+
// Per-turn metadata bag: pass the backend's non-standard top-level fields
|
|
525
610
|
// through verbatim. Providers do not reinterpret vendor currency or account
|
|
526
611
|
// metadata; a monetary value carries its own amount and currency.
|
|
527
612
|
#buildMeta(chunkMetadata: Record<string, unknown>): Record<string, unknown> | undefined {
|
|
@@ -533,7 +618,8 @@ export default class AiSdkProvider implements Provider {
|
|
|
533
618
|
// penalties, stop, seed, …) merged UNDER the managed body: model, messages,
|
|
534
619
|
// reasoning, grammar (+ its repeat-penalty floor), max_tokens and slot always
|
|
535
620
|
// win, and reserved transport/protocol keys are stripped so the passthrough
|
|
536
|
-
// can't smuggle a grammar, a stream toggle, or a backend slot
|
|
621
|
+
// can't smuggle a grammar, a stream toggle, or a backend slot
|
|
622
|
+
// ({§provider-request-authority}).
|
|
537
623
|
#samplingBody(sampling: Record<string, unknown> | undefined): Record<string, unknown> {
|
|
538
624
|
if (sampling === undefined) return {};
|
|
539
625
|
const out: Record<string, unknown> = {};
|
|
@@ -542,41 +628,40 @@ export default class AiSdkProvider implements Provider {
|
|
|
542
628
|
}
|
|
543
629
|
|
|
544
630
|
async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling }: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse> {
|
|
545
|
-
//
|
|
631
|
+
// {§provider-interface} The worker identity is required.
|
|
546
632
|
if (workerId === undefined || workerId.length === 0) throw new Error("generate: workerId is required — the worker's stable, opaque identity");
|
|
547
|
-
// Reject before any wire call when already aborted
|
|
633
|
+
// Reject before any wire call when already aborted
|
|
634
|
+
// ({§provider-failure-normalization}).
|
|
548
635
|
signal?.throwIfAborted();
|
|
549
636
|
|
|
550
|
-
// Grammar handling (
|
|
551
|
-
// grammar
|
|
552
|
-
// model generates UNCONSTRAINED — and the free output is still verified
|
|
553
|
-
// against the grammar (below), surfacing exactly where the model's natural
|
|
554
|
-
// output and the grammar conflict. Otherwise the grammar is sent when the
|
|
555
|
-
// backend supports it (grammarStyle !== "none").
|
|
637
|
+
// Grammar handling ({§gbnf-response-observation}). Debug validates the
|
|
638
|
+
// supplied grammar before the call but withholds it from the backend.
|
|
556
639
|
const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
|
|
557
640
|
if (wantGrammar && this.#gbnfDebug) this.#assertGrammarValid(grammar!);
|
|
558
641
|
const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
|
|
642
|
+
const preserveGrammarSentence = wantGrammar
|
|
643
|
+
&& this.#reasoningStyle === "template";
|
|
559
644
|
|
|
560
645
|
// Assembly order = precedence: the family's sampling DEFAULTS
|
|
561
|
-
// (PLURNK_PROVIDERS_TEMPERATURE — universal,
|
|
646
|
+
// (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
|
|
562
647
|
// paths and the name promises every request) < the caller's `sampling`
|
|
563
648
|
// < the managed fields, which always win.
|
|
564
649
|
const body: Record<string, unknown> = {
|
|
565
|
-
//
|
|
650
|
+
// Floors are suppressed on router-owned-tuning providers (plurnk) —
|
|
566
651
|
// the router's per-model tuning must not be overridden by client floors.
|
|
567
652
|
...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
|
|
568
653
|
...this.#samplingBody(sampling),
|
|
569
654
|
...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
|
|
570
655
|
model: this.#model,
|
|
571
656
|
messages,
|
|
572
|
-
...this.#reasoningBody(),
|
|
657
|
+
...this.#reasoningBody(preserveGrammarSentence),
|
|
573
658
|
...this.#grammarBody(sendGrammar),
|
|
574
659
|
...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
|
|
575
|
-
//
|
|
660
|
+
// Request per-token logprobs only when enabled (managed field —
|
|
576
661
|
// reserved from caller sampling; the env flag is the single control).
|
|
577
662
|
...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
|
|
578
663
|
...this.#slotBody(workerId),
|
|
579
|
-
//
|
|
664
|
+
// Prompt-cache affinity -- workerId as the OpenAI-standard
|
|
580
665
|
// prompt_cache_key routes a worker's turns to one serverless replica so
|
|
581
666
|
// its stable prefix caches (managed; reserved from caller sampling).
|
|
582
667
|
...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
|
|
@@ -584,7 +669,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
584
669
|
|
|
585
670
|
// Per-request headers = static auth/routing + any first-party telemetry.
|
|
586
671
|
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
|
|
587
|
-
const headers = Object.keys(metaHeaders).length
|
|
672
|
+
const headers = Object.keys(metaHeaders).length === 0
|
|
673
|
+
? this.#headers
|
|
674
|
+
: { ...this.#headers, ...metaHeaders };
|
|
588
675
|
let raw;
|
|
589
676
|
try {
|
|
590
677
|
raw = this.#languageModel === undefined
|
|
@@ -636,14 +723,14 @@ export default class AiSdkProvider implements Provider {
|
|
|
636
723
|
});
|
|
637
724
|
} catch (err) {
|
|
638
725
|
if (signal?.aborted) throw err;
|
|
639
|
-
const pe = toProviderError(err, this.#source);
|
|
726
|
+
const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
|
|
640
727
|
if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
|
|
641
728
|
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
|
|
642
729
|
}
|
|
643
730
|
throw pe;
|
|
644
731
|
}
|
|
645
732
|
|
|
646
|
-
//
|
|
733
|
+
// llama-server --special renders EOG tokens as text, so a turn ending
|
|
647
734
|
// via raw EOS carries a trailing <eos> the grammar never sanctioned - it both
|
|
648
735
|
// false-rejects the rail verdict and leaks a control token into the packet.
|
|
649
736
|
// Strip the server-reported eos_token from the tail ONCE, before the verdict
|
|
@@ -651,66 +738,133 @@ export default class AiSdkProvider implements Provider {
|
|
|
651
738
|
// wire text for forensics.
|
|
652
739
|
if (this.#eosText !== undefined) raw.content = stripTrailingSpecial(raw.content, this.#eosText);
|
|
653
740
|
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
741
|
+
const grammarInput = raw.content;
|
|
742
|
+
const projectedReasoning = preserveGrammarSentence && !raw.reasoningProjected
|
|
743
|
+
? projectTemplateReasoning(raw.content)
|
|
744
|
+
: projectTaggedReasoning(
|
|
745
|
+
raw.content,
|
|
746
|
+
raw.reasoning,
|
|
747
|
+
this.#reasoningResponseStyle,
|
|
748
|
+
);
|
|
749
|
+
|
|
750
|
+
// Preserve the exact sentence seen at the grammar boundary. Constrained
|
|
751
|
+
// template turns request `reasoning_format: "none"`, so even an empty
|
|
752
|
+
// channel remains observable. An unexpectedly projected response cannot
|
|
753
|
+
// supply independent pre-projection evidence.
|
|
754
|
+
let grammarEvidence: GrammarEvidence | undefined;
|
|
755
|
+
if (wantGrammar) {
|
|
756
|
+
if (preserveGrammarSentence) {
|
|
757
|
+
if (!raw.reasoningProjected) {
|
|
758
|
+
grammarEvidence = {
|
|
759
|
+
input: grammarInput,
|
|
760
|
+
contentStart: projectedReasoning.projected ? projectedReasoning.contentStart : 0,
|
|
761
|
+
transported: sendGrammar !== undefined,
|
|
762
|
+
};
|
|
763
|
+
}
|
|
764
|
+
} else if (projectedReasoning.projected) {
|
|
765
|
+
grammarEvidence = {
|
|
766
|
+
input: grammarInput,
|
|
767
|
+
contentStart: projectedReasoning.contentStart,
|
|
768
|
+
transported: sendGrammar !== undefined,
|
|
769
|
+
};
|
|
770
|
+
} else {
|
|
771
|
+
grammarEvidence = {
|
|
772
|
+
input: grammarInput,
|
|
773
|
+
contentStart: 0,
|
|
774
|
+
transported: sendGrammar !== undefined,
|
|
775
|
+
};
|
|
668
776
|
}
|
|
669
|
-
|
|
670
|
-
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
777
|
+
}
|
|
778
|
+
|
|
779
|
+
if (projectedReasoning.projected) {
|
|
780
|
+
raw.content = projectedReasoning.content;
|
|
781
|
+
raw.reasoning = projectedReasoning.reasoning;
|
|
782
|
+
raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
|
|
783
|
+
}
|
|
784
|
+
|
|
785
|
+
let notices: ProviderNotice[] | undefined;
|
|
786
|
+
const usage = raw.usage;
|
|
787
|
+
if (sendGrammar !== undefined && this.tokenize !== undefined) {
|
|
788
|
+
// Channel-escape detector: completion tokens
|
|
675
789
|
// billed far beyond every visible channel mean the decode ESCAPED into
|
|
676
|
-
// a server-discarded reasoning block mid-emission
|
|
677
|
-
//
|
|
678
|
-
|
|
679
|
-
|
|
680
|
-
|
|
681
|
-
|
|
682
|
-
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
|
|
686
|
-
|
|
687
|
-
|
|
790
|
+
// a server-discarded reasoning block mid-emission. This diagnostic
|
|
791
|
+
// requires the serving vocabulary; an estimate cannot prove absence.
|
|
792
|
+
try {
|
|
793
|
+
const [contentTokens, reasoningTokens] = await Promise.all([
|
|
794
|
+
this.tokenize(raw.content),
|
|
795
|
+
this.tokenize(raw.reasoning),
|
|
796
|
+
]);
|
|
797
|
+
const visible = contentTokens.length + reasoningTokens.length;
|
|
798
|
+
if (usage.completion > visible + 64) {
|
|
799
|
+
(notices ??= []).push({
|
|
800
|
+
source: this.#source,
|
|
801
|
+
kind: "grammar_unenforced",
|
|
802
|
+
level: "warn",
|
|
803
|
+
message: `decode escaped the grammar: ${usage.completion} completion tokens billed but only ${visible} visible across content+reasoning — the balance ran unconstrained in a discarded reasoning channel`,
|
|
804
|
+
position: [...raw.content].length,
|
|
805
|
+
});
|
|
806
|
+
}
|
|
807
|
+
} catch (cause) {
|
|
808
|
+
emitWarningOnce(
|
|
809
|
+
`${this.#source}: exact visible-token diagnostic unavailable (${cause instanceof Error ? cause.message : String(cause)})`,
|
|
810
|
+
"PLURNK_VISIBLE_TOKEN_COUNT_UNAVAILABLE",
|
|
811
|
+
);
|
|
688
812
|
}
|
|
689
813
|
}
|
|
690
814
|
|
|
691
|
-
const
|
|
692
|
-
const meta = railsMeta !== undefined ? { ...builtMeta, ...railsMeta } : builtMeta;
|
|
815
|
+
const meta = this.#buildMeta(raw.metadata);
|
|
693
816
|
const logprobs = raw.logprobs.length > 0 ? raw.logprobs : undefined;
|
|
694
817
|
const meanLogprob = logprobs !== undefined
|
|
695
818
|
? logprobs.reduce((sum, token) => sum + token.logprob, 0) / logprobs.length
|
|
696
819
|
: undefined;
|
|
697
820
|
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
|
|
701
|
-
|
|
702
|
-
|
|
703
|
-
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
821
|
+
const assistant = {
|
|
822
|
+
content: raw.content,
|
|
823
|
+
reasoning: raw.reasoning.length > 0 ? raw.reasoning : null,
|
|
824
|
+
...(raw.reasoningEncrypted.length > 0
|
|
825
|
+
? { reasoningEncrypted: raw.reasoningEncrypted }
|
|
826
|
+
: {}),
|
|
827
|
+
usage,
|
|
828
|
+
model: raw.model,
|
|
829
|
+
...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
|
|
830
|
+
};
|
|
831
|
+
const normalizedCharge = this.#normalizeCharge?.(raw.chargeEvidence);
|
|
832
|
+
const charge = normalizedCharge === undefined
|
|
833
|
+
? undefined
|
|
834
|
+
: validateAuthoritativeCharge(normalizedCharge);
|
|
835
|
+
const evidence = {
|
|
710
836
|
assistantRaw: raw,
|
|
837
|
+
...(charge === undefined ? {} : { charge }),
|
|
838
|
+
...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
|
|
711
839
|
...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
|
|
712
840
|
...(meta !== undefined ? { meta } : {}),
|
|
713
|
-
...(
|
|
841
|
+
...(notices !== undefined ? { notices } : {}),
|
|
842
|
+
};
|
|
843
|
+
if (raw.finishReason === "resource_interrupted") {
|
|
844
|
+
const attempt: ProviderResponse<"resource_interrupted"> = {
|
|
845
|
+
assistant: { ...assistant, finishReason: raw.finishReason },
|
|
846
|
+
...evidence,
|
|
847
|
+
};
|
|
848
|
+
throw new ProviderError(
|
|
849
|
+
this.#source,
|
|
850
|
+
"resource_interrupted",
|
|
851
|
+
"The provider interrupted generation because inference resources were unavailable.",
|
|
852
|
+
{
|
|
853
|
+
attempt,
|
|
854
|
+
extensions: {
|
|
855
|
+
stage: "provider-response",
|
|
856
|
+
finishReason: "resource_interrupted",
|
|
857
|
+
...(raw.rawFinishReason === undefined
|
|
858
|
+
? {}
|
|
859
|
+
: { rawFinishReason: raw.rawFinishReason }),
|
|
860
|
+
},
|
|
861
|
+
},
|
|
862
|
+
);
|
|
863
|
+
}
|
|
864
|
+
return {
|
|
865
|
+
assistant: { ...assistant, finishReason: raw.finishReason },
|
|
866
|
+
...evidence,
|
|
714
867
|
};
|
|
715
868
|
}
|
|
869
|
+
|
|
716
870
|
}
|