@plurnk/plurnk-providers 1.3.12 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +39 -22
- package/README.md +65 -4
- package/SPEC.md +222 -56
- package/dist/AiSdkProvider.d.ts +18 -5
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +240 -153
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +14 -15
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +26 -10
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +8 -2
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +41 -8
- package/dist/Pool.js.map +1 -1
- package/dist/ProviderRegistry.d.ts +4 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +7 -3
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +4 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +18 -3
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +3 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +20 -7
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +11 -4
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +11 -0
- package/dist/cost.d.ts.map +1 -0
- package/dist/cost.js +64 -0
- package/dist/cost.js.map +1 -0
- package/dist/discover.d.ts +2 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js +15 -9
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +4 -0
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +29 -9
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +27 -0
- package/dist/errors.d.ts.map +1 -0
- package/dist/errors.js +150 -0
- package/dist/errors.js.map +1 -0
- package/dist/index.d.ts +11 -5
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +7 -4
- package/dist/index.js.map +1 -1
- package/dist/notices.d.ts +10 -0
- package/dist/notices.d.ts.map +1 -0
- package/dist/notices.js +11 -0
- package/dist/notices.js.map +1 -0
- package/dist/ollama.d.ts.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/promptTokens.d.ts +4 -0
- package/dist/promptTokens.d.ts.map +1 -0
- package/dist/promptTokens.js +32 -0
- package/dist/promptTokens.js.map +1 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +4 -3
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +43 -16
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +1 -1
- package/dist/types.js.map +1 -1
- package/dist/usage.d.ts +3 -0
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +26 -14
- package/dist/usage.js.map +1 -1
- package/dist/warnings.js +0 -0
- package/dist/warnings.js.map +1 -1
- package/package.json +13 -9
- package/src/AiSdkProvider.test.ts +480 -159
- package/src/AiSdkProvider.ts +320 -196
- package/src/Mock.test.ts +29 -14
- package/src/Mock.ts +33 -15
- package/src/Pool.test.ts +43 -6
- package/src/Pool.ts +56 -10
- package/src/ProviderRegistry.test.ts +158 -9
- package/src/ProviderRegistry.ts +19 -6
- package/src/aiSdkTransport.ts +25 -6
- package/src/boundaries.test.ts +8 -3
- package/src/catalogProvider.test.ts +17 -0
- package/src/catalogProvider.ts +25 -10
- package/src/compatibleProvider.test.ts +96 -0
- package/src/compatibleProvider.ts +15 -6
- package/src/cost.test.ts +63 -0
- package/src/cost.ts +83 -0
- package/src/defaults.test.ts +1 -0
- package/src/discover.test.ts +48 -7
- package/src/discover.ts +31 -21
- package/src/env.test.ts +38 -23
- package/src/env.ts +45 -18
- package/src/errors.test.ts +148 -0
- package/src/errors.ts +207 -0
- package/src/index.ts +29 -7
- package/src/lexicon-guard.test.ts +6 -6
- package/src/notices.ts +22 -0
- package/src/ollama.test.ts +64 -0
- package/src/ollama.ts +6 -3
- package/src/openai.ts +3 -0
- package/src/promptTokens.ts +41 -0
- package/src/sdkModels.test.ts +7 -0
- package/src/sdkModels.ts +4 -8
- package/src/types.ts +106 -64
- package/src/usage.test.ts +15 -4
- package/src/usage.ts +32 -14
- package/src/warnings.test.ts +10 -10
- package/src/warnings.ts +0 -0
- package/dist/OpenAICompat.d.ts +0 -76
- package/dist/OpenAICompat.d.ts.map +0 -1
- package/dist/OpenAICompat.js +0 -555
- package/dist/OpenAICompat.js.map +0 -1
- package/dist/openaiStream.d.ts +0 -47
- package/dist/openaiStream.d.ts.map +0 -1
- package/dist/openaiStream.js +0 -280
- package/dist/openaiStream.js.map +0 -1
- package/dist/standardProviders.d.ts +0 -31
- package/dist/standardProviders.d.ts.map +0 -1
- package/dist/standardProviders.js +0 -518
- package/dist/standardProviders.js.map +0 -1
- package/dist/telemetry.d.ts +0 -24
- package/dist/telemetry.d.ts.map +0 -1
- package/dist/telemetry.js +0 -85
- package/dist/telemetry.js.map +0 -1
- package/src/telemetry.test.ts +0 -69
- package/src/telemetry.ts +0 -116
package/src/index.ts
CHANGED
|
@@ -1,17 +1,25 @@
|
|
|
1
1
|
export type {
|
|
2
2
|
ChatMessage,
|
|
3
3
|
FinishReason,
|
|
4
|
+
GrammarEvidence,
|
|
4
5
|
Provider,
|
|
5
6
|
ProviderAssistant,
|
|
7
|
+
ProviderAttempt,
|
|
8
|
+
ProviderAttemptFinishReason,
|
|
6
9
|
AiSdkProviderPlugin,
|
|
7
10
|
ProviderOptions,
|
|
8
11
|
ProviderResponse,
|
|
12
|
+
ProviderEncryptedReasoningItem,
|
|
9
13
|
ProviderUsage,
|
|
14
|
+
PromptTokenMeasurement,
|
|
10
15
|
TokenLogprob,
|
|
11
16
|
TokenAlternative,
|
|
17
|
+
AuthoritativeCharge,
|
|
12
18
|
} from "./types.ts";
|
|
19
|
+
export type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
20
|
+
export { assertPromptTokenMeasurement } from "./promptTokens.ts";
|
|
13
21
|
|
|
14
|
-
// Alias cascade — re-exported from the zero-dep @plurnk/plurnk-aliases
|
|
22
|
+
// Alias cascade — re-exported from the zero-dep @plurnk/plurnk-aliases, so
|
|
15
23
|
// the "." surface is unchanged for existing importers and there's one source of
|
|
16
24
|
// truth for the parser (thin clients depend on that package directly).
|
|
17
25
|
export type { ProviderAlias } from "@plurnk/plurnk-aliases";
|
|
@@ -23,24 +31,38 @@ export {
|
|
|
23
31
|
resetDiscoveryCache,
|
|
24
32
|
} from "./ProviderRegistry.ts";
|
|
25
33
|
|
|
26
|
-
// Scope-agnostic plugin discovery (
|
|
34
|
+
// Scope-agnostic plugin discovery ({§plugin-family-kind}).
|
|
27
35
|
export { discover } from "./discover.ts";
|
|
28
36
|
export type { DiscoverOptions, Discovery } from "./discover.ts";
|
|
29
37
|
|
|
30
38
|
// Stable PLURNK adapter over AI SDK language models and compatible local URLs.
|
|
31
39
|
export { default as AiSdkProvider, effortFromBudget } from "./AiSdkProvider.ts";
|
|
32
40
|
export type { AiSdkProviderConfig, ReasoningStyle, GrammarStyle } from "./AiSdkProvider.ts";
|
|
33
|
-
//
|
|
41
|
+
// {§provider-capacity-pool} Front N interchangeable backends as one Provider -
|
|
34
42
|
// worker-sticky for KV-cache reuse, overflow to a healthy sibling; the blend
|
|
35
43
|
// DECISION stays the consumer's, by choosing which pool to call.
|
|
36
44
|
export { default as Pool } from "./Pool.ts";
|
|
37
45
|
export type { ProviderFetch } from "./AiSdkProvider.ts";
|
|
38
|
-
export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, envelopeFromEnv, resolveReserve } from "./env.ts";
|
|
39
|
-
export type { Reasoning, ReasoningMode, ReserveSpec } from "./env.ts";
|
|
46
|
+
export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, reasoningResponseStyleFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, effectiveContextWindow, envelopeFromEnv, resolveReserve, PROVIDERS_KNOBS } from "./env.ts";
|
|
47
|
+
export type { Reasoning, ReasoningMode, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
|
|
40
48
|
export { normalizeUsage, calculateCostUsd } from "./usage.ts";
|
|
49
|
+
export {
|
|
50
|
+
providerCostFor,
|
|
51
|
+
providerCostUsd,
|
|
52
|
+
resolveProviderCost,
|
|
53
|
+
validateAuthoritativeCharge,
|
|
54
|
+
validateProviderCost,
|
|
55
|
+
} from "./cost.ts";
|
|
41
56
|
export type { RawUsage, TokenRates } from "./usage.ts";
|
|
42
|
-
export { ProviderError, classifyProviderError, toProviderError
|
|
43
|
-
export
|
|
57
|
+
export { ProviderError, classifyProviderError, toProviderError } from "./errors.ts";
|
|
58
|
+
export { providerSource } from "./notices.ts";
|
|
59
|
+
export type { ProviderErrorKind } from "./errors.ts";
|
|
60
|
+
export type { ProviderNotice, ProviderNoticeKind } from "./notices.ts";
|
|
61
|
+
export type {
|
|
62
|
+
PluginAttributionContext,
|
|
63
|
+
PluginAttributionDeclaration,
|
|
64
|
+
PluginAttributionSource,
|
|
65
|
+
} from "@plurnk/plurnk-meta";
|
|
44
66
|
|
|
45
67
|
export { default as Mock } from "./Mock.ts";
|
|
46
68
|
export type { MockAssistant, MockResponse, MockReturnedAssistant } from "./Mock.ts";
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
//
|
|
1
|
+
// {§lexicon} The providers-lane standing guard (OpenAI
|
|
2
2
|
// lexicon). Retired terms fail CI here, not at the next audit — the mirror of
|
|
3
3
|
// core's plurnk-core guard, tuned to what PROVIDERS retired. Scope: src/ non-test
|
|
4
4
|
// + SPEC.md. A `lexicon-allow` line marker exempts the shed's own call sites
|
|
@@ -33,13 +33,13 @@ const WIRE_THINKING = /enable_thinking|`thinking`|thinking\s*:|\.thinking\b|thin
|
|
|
33
33
|
|
|
34
34
|
// Each entry: the banned pattern and the canonical term the violation must become.
|
|
35
35
|
const BANNED: Array<{ label: string; re: RegExp; canon: string; exempt?: RegExp }> = [
|
|
36
|
-
{ label: "thinking (our-voice)", re: /\bthinking\b/i, canon: "reasoning (
|
|
37
|
-
{ label: "contextSize", re: /\bcontextSize\b/, canon: "contextWindow — the provider window (
|
|
38
|
-
{ label: "retired providers knob", re: /PLURNK_PROVIDERS_(THINKING|LOGPROB\b|CONTEXT_SIZE\b)/, canon: "PLURNK_PROVIDERS_{REASONING,TOP_LOGPROBS,CONTEXT_WINDOW} (
|
|
39
|
-
//
|
|
36
|
+
{ label: "thinking (our-voice)", re: /\bthinking\b/i, canon: "reasoning ({§lexicon})", exempt: WIRE_THINKING },
|
|
37
|
+
{ label: "contextSize", re: /\bcontextSize\b/, canon: "contextWindow — the provider window ({§model-fact-resolution})" },
|
|
38
|
+
{ label: "retired providers knob", re: /PLURNK_PROVIDERS_(THINKING|LOGPROB\b|CONTEXT_SIZE\b)/, canon: "PLURNK_PROVIDERS_{REASONING,TOP_LOGPROBS,CONTEXT_WINDOW} ({§provider-configuration}) — only the shed may name these" },
|
|
39
|
+
// Catch the retired run/session noun in the wire-header form too (a
|
|
40
40
|
// quoted string, not an identifier — the hole the old `Plurnk-Run-Id` hid in),
|
|
41
41
|
// alongside the coordinate identifiers.
|
|
42
|
-
{ label: "run/session (retired noun — coordinate or wire header)", re: /\b(sessionId|runId)\b|Plurnk-(Run|Session)-Id/, canon: "workerId/workspaceId, Plurnk-Worker-Id/Plurnk-Workspace-Id (
|
|
42
|
+
{ label: "run/session (retired noun — coordinate or wire header)", re: /\b(sessionId|runId)\b|Plurnk-(Run|Session)-Id/, canon: "workerId/workspaceId, Plurnk-Worker-Id/Plurnk-Workspace-Id ({§lifecycle-terms})" },
|
|
43
43
|
];
|
|
44
44
|
|
|
45
45
|
test("retired provider terms never reappear in src or SPEC — drift fails CI, not the next audit", () => {
|
package/src/notices.ts
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
export type ProviderNoticeKind = "grammar_unenforced";
|
|
2
|
+
|
|
3
|
+
// Observations about a completed model exchange. These never represent a
|
|
4
|
+
// failed provider operation; transport failures throw ProviderError with an
|
|
5
|
+
// RFC 9457 Problem Details object.
|
|
6
|
+
export interface ProviderNotice {
|
|
7
|
+
readonly source: string;
|
|
8
|
+
readonly kind: ProviderNoticeKind;
|
|
9
|
+
readonly level: "warn";
|
|
10
|
+
readonly message: string;
|
|
11
|
+
readonly position: number | null;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export const providerSource = (vendor: string): string => {
|
|
15
|
+
const raw = vendor.startsWith("provider:") ? vendor.slice("provider:".length) : vendor;
|
|
16
|
+
const normalized = raw
|
|
17
|
+
.toLowerCase()
|
|
18
|
+
.replace(/[^a-z0-9-]+/g, "-")
|
|
19
|
+
.replace(/^-+|-+$/g, "");
|
|
20
|
+
if (normalized.length === 0) throw new TypeError("provider source must name a provider");
|
|
21
|
+
return `provider:${/^[a-z]/.test(normalized) ? normalized : `p-${normalized}`}`;
|
|
22
|
+
};
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import test, { mock } from "node:test";
|
|
2
|
+
import { strict as assert } from "node:assert";
|
|
3
|
+
import { ollamaProviderFromEnv } from "./ollama.ts";
|
|
4
|
+
|
|
5
|
+
const env = Object.freeze({
|
|
6
|
+
PLURNK_PROVIDERS_FETCH_TIMEOUT: "1000",
|
|
7
|
+
PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
|
|
8
|
+
PLURNK_PROVIDERS_REASONING: "off",
|
|
9
|
+
PLURNK_PROVIDERS_TEMPERATURE: "0.2",
|
|
10
|
+
PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15",
|
|
11
|
+
PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0",
|
|
12
|
+
PLURNK_PROVIDERS_REASONING_RESERVE: "10%",
|
|
13
|
+
PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
|
|
14
|
+
PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
|
|
15
|
+
PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
|
|
16
|
+
PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "1",
|
|
17
|
+
});
|
|
18
|
+
|
|
19
|
+
test.afterEach(() => mock.restoreAll());
|
|
20
|
+
|
|
21
|
+
// {§model-fact-resolution}
|
|
22
|
+
test("#126: Ollama always probes model physics and applies an operator context-window ceiling", async () => {
|
|
23
|
+
const calls: string[] = [];
|
|
24
|
+
mock.method(globalThis, "fetch", async (input: string | URL | Request) => {
|
|
25
|
+
calls.push(String(input));
|
|
26
|
+
return new Response(JSON.stringify({
|
|
27
|
+
model_info: { "qwen.context_length": 32_768 },
|
|
28
|
+
}));
|
|
29
|
+
});
|
|
30
|
+
|
|
31
|
+
const windows: Array<number | null> = [];
|
|
32
|
+
for (const operatorCap of [undefined, "8192", "65536"] as const) {
|
|
33
|
+
const provider = await ollamaProviderFromEnv({
|
|
34
|
+
...env,
|
|
35
|
+
...(operatorCap === undefined ? {} : { PLURNK_PROVIDERS_CONTEXT_WINDOW: operatorCap }),
|
|
36
|
+
}, "qwen2.5-coder", { baseUrl: "http://ollama.test:11434/v1" });
|
|
37
|
+
windows.push(provider.contextWindow);
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
assert.deepEqual(windows, [32_768, 8_192, 32_768]);
|
|
41
|
+
assert.deepEqual(calls, Array.from({ length: 3 }, () => "http://ollama.test:11434/api/show"));
|
|
42
|
+
});
|
|
43
|
+
|
|
44
|
+
test("#126: an operator ceiling does not hide an Ollama probe HTTP failure", async () => {
|
|
45
|
+
mock.method(globalThis, "fetch", async () => new Response(null, { status: 503 }));
|
|
46
|
+
await assert.rejects(
|
|
47
|
+
() => ollamaProviderFromEnv({
|
|
48
|
+
...env,
|
|
49
|
+
PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
|
|
50
|
+
}, "qwen2.5-coder", { baseUrl: "http://ollama.test:11434" }),
|
|
51
|
+
/ollama provider: \/api\/show returned 503/,
|
|
52
|
+
);
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
test("#126: an operator ceiling does not hide a missing Ollama model fact", async () => {
|
|
56
|
+
mock.method(globalThis, "fetch", async () => new Response(JSON.stringify({ model_info: {} })));
|
|
57
|
+
await assert.rejects(
|
|
58
|
+
() => ollamaProviderFromEnv({
|
|
59
|
+
...env,
|
|
60
|
+
PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
|
|
61
|
+
}, "qwen2.5-coder", { baseUrl: "http://ollama.test:11434" }),
|
|
62
|
+
/ollama provider: \/api\/show has no \*\.context_length key for "qwen2\.5-coder"/,
|
|
63
|
+
);
|
|
64
|
+
});
|
package/src/ollama.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { createOpenAICompatible } from "@ai-sdk/openai-compatible";
|
|
2
|
-
import { contextWindowFromEnv, parseRequiredInt, requireEnv } from "./env.ts";
|
|
2
|
+
import { contextWindowFromEnv, effectiveContextWindow, parseRequiredInt, requireEnv } from "./env.ts";
|
|
3
3
|
import { providerFromSdkModel } from "./catalogProvider.ts";
|
|
4
4
|
import type { Provider, ProviderOptions } from "./types.ts";
|
|
5
5
|
|
|
@@ -46,8 +46,11 @@ export const ollamaProviderFromEnv = async (
|
|
|
46
46
|
"PLURNK_PROVIDERS_FETCH_TIMEOUT",
|
|
47
47
|
"ollama",
|
|
48
48
|
);
|
|
49
|
-
|
|
50
|
-
|
|
49
|
+
// {§model-fact-resolution}
|
|
50
|
+
const contextWindow = effectiveContextWindow(
|
|
51
|
+
contextWindowFromEnv(env, "ollama"),
|
|
52
|
+
await fetchContextWindow({ baseUrl, model, timeout }),
|
|
53
|
+
);
|
|
51
54
|
const languageModel = createOpenAICompatible({
|
|
52
55
|
name: "ollama",
|
|
53
56
|
baseURL: `${baseUrl}/v1`,
|
package/src/openai.ts
CHANGED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
import type { ChatMessage, PromptTokenMeasurement } from "./types.ts";
|
|
2
|
+
|
|
3
|
+
const KINDS = new Set<PromptTokenMeasurement["kind"]>([
|
|
4
|
+
"exact",
|
|
5
|
+
"upper_bound",
|
|
6
|
+
"estimate",
|
|
7
|
+
]);
|
|
8
|
+
|
|
9
|
+
export const assertPromptTokenMeasurement = (
|
|
10
|
+
value: unknown,
|
|
11
|
+
owner = "provider",
|
|
12
|
+
): PromptTokenMeasurement => {
|
|
13
|
+
if (typeof value !== "object" || value === null) {
|
|
14
|
+
throw new TypeError(`${owner}: prompt token measurement must be an object`);
|
|
15
|
+
}
|
|
16
|
+
const candidate = value as Partial<PromptTokenMeasurement>;
|
|
17
|
+
if (!KINDS.has(candidate.kind as PromptTokenMeasurement["kind"])) {
|
|
18
|
+
throw new TypeError(`${owner}: prompt token measurement has invalid kind ${JSON.stringify(candidate.kind)}`);
|
|
19
|
+
}
|
|
20
|
+
if (!Number.isInteger(candidate.tokens) || candidate.tokens! < 0) {
|
|
21
|
+
throw new TypeError(`${owner}: prompt token measurement tokens must be a non-negative integer`);
|
|
22
|
+
}
|
|
23
|
+
if (typeof candidate.source !== "string" || candidate.source.length === 0) {
|
|
24
|
+
throw new TypeError(`${owner}: prompt token measurement source must be a non-empty string`);
|
|
25
|
+
}
|
|
26
|
+
if (candidate.kind === "estimate"
|
|
27
|
+
&& (typeof candidate.detail !== "string" || candidate.detail.length === 0)) {
|
|
28
|
+
throw new TypeError(`${owner}: estimated prompt token measurement requires detail`);
|
|
29
|
+
}
|
|
30
|
+
return value as PromptTokenMeasurement;
|
|
31
|
+
};
|
|
32
|
+
|
|
33
|
+
export const estimatePromptTokens = (
|
|
34
|
+
messages: readonly ChatMessage[],
|
|
35
|
+
detail = "chars/2 over message content; provider request framing is unknown",
|
|
36
|
+
): PromptTokenMeasurement => ({
|
|
37
|
+
kind: "estimate",
|
|
38
|
+
tokens: Math.ceil(messages.reduce((sum, { content }) => sum + content.length, 0) / 2),
|
|
39
|
+
source: "heuristic:chars2",
|
|
40
|
+
detail,
|
|
41
|
+
});
|
package/src/sdkModels.test.ts
CHANGED
|
@@ -37,6 +37,13 @@ test("createSdkModel expands catalog endpoint variables without treating them as
|
|
|
37
37
|
});
|
|
38
38
|
});
|
|
39
39
|
|
|
40
|
+
test("#157: a cataloged compatible provider fails before transport when its declared credential is absent", () => {
|
|
41
|
+
assert.throws(
|
|
42
|
+
() => createSdkModel("deepseek", "deepseek-v4-flash", {}),
|
|
43
|
+
/deepseek provider: DEEPSEEK_API_KEY must be set/,
|
|
44
|
+
);
|
|
45
|
+
});
|
|
46
|
+
|
|
40
47
|
test("createSdkModel fails clearly for a declared but unsupported SDK package", () => {
|
|
41
48
|
assert.throws(
|
|
42
49
|
() => createSdkModel("acme", "model", {
|
package/src/sdkModels.ts
CHANGED
|
@@ -85,12 +85,6 @@ const baseUrl = (
|
|
|
85
85
|
return value === undefined ? undefined : expandEnv(value, env, provider).replace(/\/+$/, "");
|
|
86
86
|
};
|
|
87
87
|
|
|
88
|
-
const apiKey = (
|
|
89
|
-
provider: string,
|
|
90
|
-
env: NodeJS.ProcessEnv,
|
|
91
|
-
catalog: ProviderInfo,
|
|
92
|
-
): string | undefined => firstSet(env, configuredKeyNames(provider, env, catalog));
|
|
93
|
-
|
|
94
88
|
const requireApiKey = (
|
|
95
89
|
provider: string,
|
|
96
90
|
env: NodeJS.ProcessEnv,
|
|
@@ -179,12 +173,14 @@ export const createSdkModel = (
|
|
|
179
173
|
};
|
|
180
174
|
case "@ai-sdk/openai-compatible":
|
|
181
175
|
if (url === undefined) throw new Error(`${provider} provider: Models.dev supplies no API URL and no base URL was configured`);
|
|
176
|
+
const keyNames = configuredKeyNames(provider, env, catalog);
|
|
177
|
+
const key = keyNames.length === 0 ? undefined : requireApiKey(provider, env, catalog);
|
|
182
178
|
return {
|
|
183
179
|
compatible: {
|
|
184
180
|
url: `${url}/chat/completions`,
|
|
185
|
-
headers:
|
|
181
|
+
headers: key === undefined
|
|
186
182
|
? {}
|
|
187
|
-
: { Authorization: `Bearer ${
|
|
183
|
+
: { Authorization: `Bearer ${key}` },
|
|
188
184
|
},
|
|
189
185
|
catalog,
|
|
190
186
|
};
|
package/src/types.ts
CHANGED
|
@@ -1,15 +1,36 @@
|
|
|
1
1
|
// Provider transport contract. Providers return raw wire-level output —
|
|
2
|
-
// content unparsed (consumer parses via @plurnk/plurnk-
|
|
2
|
+
// content unparsed (consumer parses via @plurnk/plurnk-contracts), reasoning
|
|
3
3
|
// is the wire-reported CoT only.
|
|
4
4
|
|
|
5
|
-
import type {
|
|
5
|
+
import type { ProviderNotice } from "./notices.ts";
|
|
6
6
|
import type { LanguageModel } from "ai";
|
|
7
|
+
import type {
|
|
8
|
+
PluginAttribution,
|
|
9
|
+
PluginAttributionContext,
|
|
10
|
+
PluginAttributionSource,
|
|
11
|
+
} from "@plurnk/plurnk-meta";
|
|
12
|
+
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
7
13
|
|
|
8
14
|
export interface ChatMessage {
|
|
9
15
|
role: "system" | "user" | "assistant";
|
|
10
16
|
content: string;
|
|
11
17
|
}
|
|
12
18
|
|
|
19
|
+
// Preflight evidence for the complete provider request. An empirical estimate
|
|
20
|
+
// is useful telemetry but cannot authorize a hard physical-capacity decision.
|
|
21
|
+
export type PromptTokenMeasurement =
|
|
22
|
+
| {
|
|
23
|
+
readonly kind: "exact" | "upper_bound";
|
|
24
|
+
readonly tokens: number;
|
|
25
|
+
readonly source: string;
|
|
26
|
+
}
|
|
27
|
+
| {
|
|
28
|
+
readonly kind: "estimate";
|
|
29
|
+
readonly tokens: number;
|
|
30
|
+
readonly source: string;
|
|
31
|
+
readonly detail: string;
|
|
32
|
+
};
|
|
33
|
+
|
|
13
34
|
// Normalized token accounting. Invariant (enforced by normalizeUsage at the
|
|
14
35
|
// provider boundary): total = prompt + completion + reasoning; cached is a
|
|
15
36
|
// subset of prompt. `completion` is visible output EXCLUDING reasoning; the
|
|
@@ -23,11 +44,14 @@ export interface ProviderUsage {
|
|
|
23
44
|
readonly total: number; // prompt + completion + reasoning
|
|
24
45
|
}
|
|
25
46
|
|
|
26
|
-
|
|
27
|
-
|
|
47
|
+
export type AuthoritativeCharge = Extract<ProviderCost, { kind: "authoritative" }>;
|
|
48
|
+
|
|
49
|
+
// A successful exchange's closed finish set. ProviderAttemptFinishReason adds
|
|
50
|
+
// the failed disposition that may occur only on ProviderError attempt evidence.
|
|
28
51
|
export type FinishReason = "stop" | "length" | "tool_calls" | "content_filter" | null;
|
|
52
|
+
export type ProviderAttemptFinishReason = FinishReason | "resource_interrupted";
|
|
29
53
|
|
|
30
|
-
// A per-token logprob
|
|
54
|
+
// {§provider-evidence} A per-token logprob. `logprob` is the backend's raw model
|
|
31
55
|
// log-probability of the emitted token — the sampling-transform-invariant
|
|
32
56
|
// confidence, chosen over Fireworks' post-mask `sampling_logprob` (measured
|
|
33
57
|
// IDENTICAL under grammar, incl. an adversarial mask; the raw value is the honest
|
|
@@ -45,20 +69,24 @@ export interface TokenLogprob {
|
|
|
45
69
|
readonly top?: readonly TokenAlternative[];
|
|
46
70
|
}
|
|
47
71
|
|
|
48
|
-
|
|
72
|
+
// {§provider-encrypted-reasoning} `id` is provider detail identity; `subtype`
|
|
73
|
+
// is the provider's evidence-backed classification. Neither is a client entity
|
|
74
|
+
// correlation, so consumers must not substitute `id` for a message/tool-call ID.
|
|
75
|
+
export interface ProviderEncryptedReasoningItem {
|
|
76
|
+
readonly id: string | null;
|
|
77
|
+
readonly subtype: string;
|
|
78
|
+
readonly encrypted: ReadonlyArray<{ data: string; format: string | null }>;
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
export interface ProviderAssistant<TFinish extends ProviderAttemptFinishReason = FinishReason> {
|
|
49
82
|
readonly content: string;
|
|
50
83
|
readonly reasoning: string | null;
|
|
51
|
-
//
|
|
52
|
-
|
|
53
|
-
// items { id, subtype, encrypted: [{data, format}] } (id from the wire,
|
|
54
|
-
// subtype from wire position), never decoded, never synthesized. Readable text
|
|
55
|
-
// stays on `reasoning`. Absent when the turn produced none; consumers (agui)
|
|
56
|
-
// project it as REASONING_ENCRYPTED_VALUE.
|
|
57
|
-
readonly reasoningEncrypted?: ReadonlyArray<{ id: string | null; subtype: string; encrypted: ReadonlyArray<{ data: string; format: string | null }> }>;
|
|
84
|
+
// Encrypted reasoning remains distinct from readable `reasoning`.
|
|
85
|
+
readonly reasoningEncrypted?: ReadonlyArray<ProviderEncryptedReasoningItem>;
|
|
58
86
|
readonly usage: ProviderUsage;
|
|
59
|
-
readonly finishReason:
|
|
87
|
+
readonly finishReason: TFinish;
|
|
60
88
|
readonly model: string;
|
|
61
|
-
// Per-token logprobs
|
|
89
|
+
// Per-token logprobs, present only when PLURNK_PROVIDERS_TOP_LOGPROBS is set
|
|
62
90
|
// AND the backend returned them. Absent otherwise — NEVER synthesized. Opt-in,
|
|
63
91
|
// per-alias: a scraping alias enables it; serving turns carry nothing.
|
|
64
92
|
readonly logprobs?: readonly TokenLogprob[];
|
|
@@ -66,42 +94,59 @@ export interface ProviderAssistant {
|
|
|
66
94
|
readonly meanLogprob?: number;
|
|
67
95
|
}
|
|
68
96
|
|
|
69
|
-
export interface
|
|
70
|
-
|
|
97
|
+
export interface GrammarEvidence {
|
|
98
|
+
// Exact sentence observed at the grammar boundary before any reasoning/content
|
|
99
|
+
// projection. Offsets are Unicode code points, matching @plurnk/gbnf verdicts.
|
|
100
|
+
readonly input: string;
|
|
101
|
+
readonly contentStart: number;
|
|
102
|
+
readonly transported: boolean;
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
export interface ProviderResponse<TFinish extends ProviderAttemptFinishReason = FinishReason> {
|
|
106
|
+
readonly assistant: ProviderAssistant<TFinish>;
|
|
71
107
|
readonly assistantRaw: unknown;
|
|
108
|
+
// A settled upstream charge is a validated public fact, not opaque metadata.
|
|
109
|
+
// Non-USD settlement carries an explicit provider-owned USD equivalent for
|
|
110
|
+
// the platform's existing USD aggregate. Core never supplies an FX rate.
|
|
111
|
+
readonly charge?: AuthoritativeCharge;
|
|
112
|
+
// {§gbnf-response-observation} — evidence only; the consumer owns the verdict.
|
|
113
|
+
readonly grammarEvidence?: GrammarEvidence;
|
|
72
114
|
// Per-turn provider→client metadata bag: the backend's non-standard top-level
|
|
73
115
|
// response fields passed through verbatim. Monetary values carry their own
|
|
74
116
|
// amount and currency; the provider does not reinterpret them. The consumer
|
|
75
117
|
// (service) merges this into its Turn metadata and
|
|
76
118
|
// filters what reaches the client; it reads `meta`, never mines `assistantRaw`.
|
|
77
|
-
// Absent when the backend reported no extra fields
|
|
119
|
+
// Absent when the backend reported no extra fields.
|
|
78
120
|
readonly meta?: Record<string, unknown>;
|
|
79
|
-
// The
|
|
121
|
+
// The verbatim backend response body ({§provider-evidence}) — the full wire JSON for
|
|
80
122
|
// a non-streamed turn, or the reassembled equivalent for a streamed one.
|
|
81
123
|
// `assistantRaw` is a normalized DIGEST (it drops choices[]); this is the
|
|
82
124
|
// capture-everything record for the endpoint's fine-tune corpus. Present ONLY
|
|
83
125
|
// when PLURNK_PROVIDERS_RAWBODY is on — off by default so serving turns never
|
|
84
126
|
// carry it. Absent otherwise.
|
|
85
127
|
readonly rawBody?: unknown;
|
|
86
|
-
//
|
|
87
|
-
//
|
|
88
|
-
//
|
|
89
|
-
|
|
90
|
-
// The provider never adjudicates conformance; discard/retry/escalate/
|
|
91
|
-
// self-correct is consumer policy. Absent when the turn produced no telemetry.
|
|
92
|
-
readonly telemetry?: readonly TelemetryEvent[];
|
|
128
|
+
// Notices attached to the represented attempt. Successful
|
|
129
|
+
// returns may relay them; interrupted attempt notices remain forensic.
|
|
130
|
+
// Grammar conformance itself is consumer-owned.
|
|
131
|
+
readonly notices?: readonly ProviderNotice[];
|
|
93
132
|
}
|
|
94
133
|
|
|
134
|
+
export type ProviderAttempt = ProviderResponse<ProviderAttemptFinishReason>;
|
|
135
|
+
|
|
95
136
|
export interface Provider {
|
|
96
|
-
//
|
|
137
|
+
// Optional package-authored folksonomy evaluated by the consumer immediately
|
|
138
|
+
// before a provider emission attempt ({§plugin-attribution}).
|
|
139
|
+
attributions?(context: PluginAttributionContext): PluginAttribution;
|
|
140
|
+
// `grammar` is an optional GBNF string (canonically @plurnk/plurnk-contracts'
|
|
97
141
|
// plurnk.gbnf, possibly root-substituted by the consumer). Backends that
|
|
98
142
|
// support grammar-constrained sampling attach it verbatim; all others
|
|
99
143
|
// ignore it. The provider never chooses or modifies the grammar — whether
|
|
100
|
-
// to constrain and which root variant to send is consumer policy
|
|
144
|
+
// to constrain and which root variant to send is consumer policy
|
|
145
|
+
// ({§gbnf-response-observation}).
|
|
101
146
|
//
|
|
102
147
|
// `maxTokens` is the consumer's per-call output ceiling (wire `max_tokens`).
|
|
103
148
|
// Without it, most servers generate UNBOUNDED (llama-server n_predict -1) —
|
|
104
|
-
// under a multi-op grammar that degenerates to the context wall
|
|
149
|
+
// under a multi-op grammar that degenerates to the context wall,
|
|
105
150
|
// so a constrained consumer is expected to pass it. Policy stays the
|
|
106
151
|
// consumer's; the provider only transports.
|
|
107
152
|
//
|
|
@@ -109,70 +154,64 @@ export interface Provider {
|
|
|
109
154
|
// stream (loop/run). Providers MAY key backend affinity on it — e.g.
|
|
110
155
|
// llama-server slot pinning for KV-cache reuse — and MUST NOT interpret
|
|
111
156
|
// its content. The consumer never sees or chooses backend resources
|
|
112
|
-
// (slot integers, connections); the
|
|
157
|
+
// (slot integers, connections); the mechanism is the provider's.
|
|
113
158
|
//
|
|
114
|
-
// `attributions`
|
|
115
|
-
//
|
|
116
|
-
//
|
|
117
|
-
//
|
|
118
|
-
//
|
|
119
|
-
//
|
|
120
|
-
//
|
|
159
|
+
// `attributions` is opaque consumer-supplied creator telemetry; the consumer
|
|
160
|
+
// owns what contribution that set claims ({§attribution}). `client` is the
|
|
161
|
+
// consumer's workspace-stable, self-identified frontend. They are forwarded ONLY by a
|
|
162
|
+
// provider whose spec opts in (the first-party `plurnk` endpoint, via
|
|
163
|
+
// `Plurnk-Attribution` / `Plurnk-Client` headers); every other provider DROPS
|
|
164
|
+
// them — the gate is structural so first-party metadata can never leak to a
|
|
165
|
+
// third-party backend.
|
|
121
166
|
//
|
|
122
167
|
// `sampling` is an optional bag of standard OpenAI-compat sampling params
|
|
123
168
|
// (temperature, top_p, top_k, min_p, penalties, stop, seed, …) forwarded into
|
|
124
169
|
// the request body UNDER the provider's managed fields — model/messages/grammar/
|
|
125
170
|
// reasoning/max_tokens/slot always win, and transport/protocol keys (stream,
|
|
126
171
|
// response_format, grammar, id_slot) are stripped, so it carries sampling intent
|
|
127
|
-
// only and can't bypass grammar transport (
|
|
172
|
+
// only and can't bypass grammar transport ({§provider-request-authority}). A
|
|
173
|
+
// proxy consumer (the
|
|
128
174
|
// plurnk endpoint fronting its own backends) uses it to pass its caller's sampling
|
|
129
175
|
// knobs through; a direct consumer typically leaves it unset.
|
|
130
176
|
//
|
|
131
177
|
// `strikes` is the worker's CURRENT rail-strike streak at time-of-generate
|
|
132
|
-
// (0 = clean; a clean turn zeroes it; every loop starts at 0 — contract
|
|
133
|
-
//
|
|
178
|
+
// (0 = clean; a clean turn zeroes it; every loop starts at 0 — contract
|
|
179
|
+
// {§strikes-first-party-metadata}). Forwarded as a `Plurnk-Strikes` header ONLY under the
|
|
134
180
|
// same firstPartyMetadata gate as attributions/client; dropped everywhere
|
|
135
181
|
// else. Headers only — the packet NEVER carries strike state (the model must
|
|
136
182
|
// not see engine accounting; it would become a metric to game).
|
|
137
183
|
//
|
|
138
|
-
// `workspaceId`/`loop`/`turn`
|
|
184
|
+
// `workspaceId`/`loop`/`turn` are the turn coordinate ({§lifecycle-terms}) — the
|
|
139
185
|
// daemon-side sequence of the turn being generated, which the endpoint can
|
|
140
186
|
// never scrape from the wire. Forwarded as `Plurnk-Workspace-Id`/`Plurnk-Loop`/
|
|
141
187
|
// `Plurnk-Turn` ONLY under the same firstPartyMetadata gate; dropped
|
|
142
188
|
// everywhere else. Coordinates are 1-based: absent/0 emits no header (no
|
|
143
189
|
// strikes-style zero exception). Headers only, never the packet.
|
|
144
190
|
generate(args: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse>;
|
|
145
|
-
//
|
|
146
|
-
//
|
|
147
|
-
// FAILS AT CONSTRUCTION when it can't (#419/#417: never budget against a wrong
|
|
148
|
-
// number). A PROBING provider (openai/llama-server) instead DEGRADES to null on a
|
|
149
|
-
// probe miss - a blip must not crash it (#34) - and surfaces it once
|
|
150
|
-
// (PLURNK_CONTEXT_UNKNOWN). So null still means "window unknown -> no cap"; the
|
|
151
|
-
// consumer must NOT improvise a stand-in from it (#421). NOTE: under llama-server
|
|
152
|
-
// --parallel N, the window is PER SLOT (the server splits --ctx-size across slots
|
|
153
|
-
// and reports the divided value).
|
|
191
|
+
// {§model-fact-resolution} — effective physical context in tokens. `null`
|
|
192
|
+
// means unknown; under llama-server parallelism the probed value is per slot.
|
|
154
193
|
readonly contextWindow: number | null;
|
|
155
194
|
readonly model: string;
|
|
156
|
-
//
|
|
195
|
+
// Optional: the backend's self-reported served model id, from a
|
|
157
196
|
// /v1/models-shaped probe (llama-server today; any such backend). For a local
|
|
158
197
|
// alias, `model` is the alias but this is the real served name (the .gguf) the
|
|
159
198
|
// tokenizer seam maps exactly. Read-only, best-effort, no extra probing —
|
|
160
199
|
// absent when no probe ran. Consumers resolve `servedModel ?? model`.
|
|
161
200
|
readonly servedModel?: string;
|
|
162
|
-
//
|
|
201
|
+
// Optional resolved capability: true when a transported grammar will
|
|
163
202
|
// actually constrain the decode (rails LIVE), false/undefined otherwise —
|
|
164
203
|
// introspectable so the consumer can fail hard on a dark-rails boot instead
|
|
165
204
|
// of discovering it from unconstrained emissions.
|
|
166
205
|
readonly constrainsOutput?: boolean;
|
|
167
|
-
//
|
|
206
|
+
// Optional resolved capability: true when this backend decodes
|
|
168
207
|
// UNBOUNDED absent a caller cap — llama-server honors n_predict to the
|
|
169
|
-
// context wall (
|
|
170
|
-
// MUST bring an output envelope (
|
|
208
|
+
// context wall (observed in a 30,736-junk-token wall run), so a consumer
|
|
209
|
+
// MUST bring an output envelope ({§provider-generation-envelope}). Cloud backends that silently
|
|
171
210
|
// clamp an over-ask (fireworks/xai, verified live) never set this; undefined
|
|
172
211
|
// = no claim. Introspectable so a consumer can refuse AT BOOT a local alias
|
|
173
212
|
// with no declared envelope, instead of dying mid-turn in partition math.
|
|
174
213
|
readonly requiresMaxTokens?: boolean;
|
|
175
|
-
//
|
|
214
|
+
// Optional generation-envelope reserves ({§provider-generation-envelope}) — the amounts of
|
|
176
215
|
// the DETECTED window reserved for reasoning and completion: floor
|
|
177
216
|
// percentages of `contextWindow`, or absolute per-alias pins that win
|
|
178
217
|
// outright. The consumer's prompt budget is `contextWindow - reasoningReserve
|
|
@@ -182,22 +221,25 @@ export interface Provider {
|
|
|
182
221
|
// as null). All first-party providers claim, so null means genuinely-unknown.
|
|
183
222
|
readonly reasoningReserve?: number | null;
|
|
184
223
|
readonly completionReserve?: number | null;
|
|
185
|
-
// Provider-owned
|
|
186
|
-
//
|
|
187
|
-
//
|
|
188
|
-
|
|
224
|
+
// Provider-owned preflight measurement of the complete chat request,
|
|
225
|
+
// including provider/template framing when the adapter can know it.
|
|
226
|
+
// Estimates are explicit and MUST NOT authorize hard physical admission.
|
|
227
|
+
countPromptTokens(messages: readonly ChatMessage[], signal?: AbortSignal): Promise<PromptTokenMeasurement>;
|
|
189
228
|
// OPTIONAL capability: exact tokenization served by the backend's own vocab
|
|
190
229
|
// (llama-server /tokenize) — token ids in the model's real vocabulary.
|
|
191
230
|
// Present ONLY when the backend exposes such an endpoint (probe-gated);
|
|
192
231
|
// `tokenize === undefined` means the backend can't. Exact-counting
|
|
193
232
|
// consumers (the tokenizer seam) prefer this over any client-side data.
|
|
194
233
|
tokenize?(text: string): Promise<number[]>;
|
|
195
|
-
//
|
|
196
|
-
//
|
|
234
|
+
// {§model-fact-resolution} — frozen 1.x local USD estimate compatibility.
|
|
235
|
+
// Consumers adapt its zero to unknown; only calculateCharge can prove free.
|
|
197
236
|
calculateCost(usage: ProviderUsage): number;
|
|
237
|
+
// Current monetary result. The numeric calculateCost surface remains frozen
|
|
238
|
+
// for 1.x compatibility; zero on that legacy surface cannot prove `free`.
|
|
239
|
+
calculateCharge?(usage: ProviderUsage): Exclude<ProviderCost, AuthoritativeCharge>;
|
|
198
240
|
}
|
|
199
241
|
|
|
200
|
-
// ProviderAlias
|
|
242
|
+
// ProviderAlias lives in @plurnk/plurnk-aliases (the zero-dependency parser);
|
|
201
243
|
// index.ts re-exports it so the "." surface is unchanged.
|
|
202
244
|
|
|
203
245
|
// Per-alias instantiation overrides, threaded from the alias cascade into the
|
|
@@ -211,6 +253,6 @@ export interface ProviderOptions {
|
|
|
211
253
|
|
|
212
254
|
// A discovered provider plugin default-exports an AI SDK provider. PLURNK owns
|
|
213
255
|
// the adapter into Provider; the plugin owns only its protocol binding.
|
|
214
|
-
export interface AiSdkProviderPlugin {
|
|
256
|
+
export interface AiSdkProviderPlugin extends PluginAttributionSource {
|
|
215
257
|
languageModel(model: string): LanguageModel;
|
|
216
258
|
}
|