@plurnk/plurnk-providers 1.3.12 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +47 -38
- package/README.md +68 -4
- package/SPEC.md +245 -62
- package/dist/AiSdkProvider.d.ts +19 -5
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +264 -156
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +15 -17
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +27 -10
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +8 -2
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +41 -8
- package/dist/Pool.js.map +1 -1
- package/dist/ProviderRegistry.d.ts +4 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +7 -3
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/accounting.d.ts +3 -0
- package/dist/accounting.d.ts.map +1 -0
- package/dist/accounting.js +84 -0
- package/dist/accounting.js.map +1 -0
- package/dist/aiSdkTransport.d.ts +5 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +91 -5
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +5 -2
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +24 -11
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +11 -4
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +11 -0
- package/dist/cost.d.ts.map +1 -0
- package/dist/cost.js +61 -0
- package/dist/cost.js.map +1 -0
- package/dist/discover.d.ts +2 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js +15 -9
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +4 -6
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +27 -29
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +27 -0
- package/dist/errors.d.ts.map +1 -0
- package/dist/errors.js +152 -0
- package/dist/errors.js.map +1 -0
- package/dist/index.d.ts +12 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +8 -5
- package/dist/index.js.map +1 -1
- package/dist/notices.d.ts +10 -0
- package/dist/notices.d.ts.map +1 -0
- package/dist/notices.js +11 -0
- package/dist/notices.js.map +1 -0
- package/dist/ollama.d.ts.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/promptTokens.d.ts +4 -0
- package/dist/promptTokens.d.ts.map +1 -0
- package/dist/promptTokens.js +32 -0
- package/dist/promptTokens.js.map +1 -0
- package/dist/sdkModels.d.ts +2 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +17 -6
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +52 -16
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +1 -1
- package/dist/types.js.map +1 -1
- package/dist/usage.d.ts +4 -0
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +64 -19
- package/dist/usage.js.map +1 -1
- package/dist/warnings.js +0 -0
- package/dist/warnings.js.map +1 -1
- package/package.json +15 -10
- package/src/AiSdkProvider.test.ts +750 -169
- package/src/AiSdkProvider.ts +354 -200
- package/src/Mock.test.ts +29 -14
- package/src/Mock.ts +36 -15
- package/src/Pool.test.ts +43 -6
- package/src/Pool.ts +56 -10
- package/src/ProviderRegistry.test.ts +158 -9
- package/src/ProviderRegistry.ts +19 -6
- package/src/accounting.test.ts +58 -0
- package/src/accounting.ts +88 -0
- package/src/aiSdkTransport.ts +101 -8
- package/src/boundaries.test.ts +9 -3
- package/src/catalogProvider.test.ts +43 -15
- package/src/catalogProvider.ts +32 -16
- package/src/compatibleProvider.test.ts +96 -0
- package/src/compatibleProvider.ts +15 -6
- package/src/cost.test.ts +64 -0
- package/src/cost.ts +78 -0
- package/src/defaults.test.ts +1 -0
- package/src/discover.test.ts +48 -7
- package/src/discover.ts +31 -21
- package/src/env.test.ts +38 -48
- package/src/env.ts +43 -40
- package/src/errors.test.ts +148 -0
- package/src/errors.ts +208 -0
- package/src/index.ts +30 -8
- package/src/lexicon-guard.test.ts +6 -6
- package/src/notices.ts +22 -0
- package/src/ollama.test.ts +64 -0
- package/src/ollama.ts +6 -3
- package/src/openai.ts +3 -0
- package/src/promptTokens.ts +41 -0
- package/src/sdkModels.test.ts +29 -3
- package/src/sdkModels.ts +19 -11
- package/src/types.ts +125 -64
- package/src/usage.test.ts +24 -5
- package/src/usage.ts +72 -21
- package/src/warnings.test.ts +10 -10
- package/src/warnings.ts +0 -0
- package/dist/OpenAICompat.d.ts +0 -76
- package/dist/OpenAICompat.d.ts.map +0 -1
- package/dist/OpenAICompat.js +0 -555
- package/dist/OpenAICompat.js.map +0 -1
- package/dist/openaiStream.d.ts +0 -47
- package/dist/openaiStream.d.ts.map +0 -1
- package/dist/openaiStream.js +0 -280
- package/dist/openaiStream.js.map +0 -1
- package/dist/standardProviders.d.ts +0 -31
- package/dist/standardProviders.d.ts.map +0 -1
- package/dist/standardProviders.js +0 -518
- package/dist/standardProviders.js.map +0 -1
- package/dist/telemetry.d.ts +0 -24
- package/dist/telemetry.d.ts.map +0 -1
- package/dist/telemetry.js +0 -85
- package/dist/telemetry.js.map +0 -1
- package/src/telemetry.test.ts +0 -69
- package/src/telemetry.ts +0 -116
package/src/boundaries.test.ts
CHANGED
|
@@ -19,17 +19,23 @@ test("provider source does not import the service or database", () => {
|
|
|
19
19
|
test("provider source does not import the PLURNK parser", () => {
|
|
20
20
|
for (const file of sourceFiles()) {
|
|
21
21
|
const source = readFileSync(join(root, file), "utf8");
|
|
22
|
-
assert.ok(
|
|
22
|
+
assert.ok(
|
|
23
|
+
!/from\s+["']@plurnk\/plurnk-contracts\/grammar["']/.test(source),
|
|
24
|
+
`${file} imports the parser rather than the runtime-neutral contracts entrypoint`,
|
|
25
|
+
);
|
|
23
26
|
}
|
|
24
27
|
});
|
|
25
28
|
|
|
26
|
-
test("
|
|
29
|
+
test("the OpenAI-compatible entrypoint excludes Node-owned provider machinery", () => {
|
|
27
30
|
const allowed = new Set([
|
|
28
31
|
"AiSdkProvider.ts",
|
|
29
32
|
"aiSdkTransport.ts",
|
|
33
|
+
"cost.ts",
|
|
30
34
|
"env.ts",
|
|
35
|
+
"errors.ts",
|
|
36
|
+
"notices.ts",
|
|
31
37
|
"openai.ts",
|
|
32
|
-
"
|
|
38
|
+
"promptTokens.ts",
|
|
33
39
|
"types.ts",
|
|
34
40
|
"usage.ts",
|
|
35
41
|
"warnings.ts",
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import test, { mock } from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
3
|
import { catalogProviderFromEnv } from "./catalogProvider.ts";
|
|
4
|
+
import { providerCostFor } from "./cost.ts";
|
|
4
5
|
import { resetEmittedWarnings } from "./warnings.ts";
|
|
5
6
|
|
|
6
7
|
const env = {
|
|
@@ -15,6 +16,7 @@ const env = {
|
|
|
15
16
|
PLURNK_PROVIDERS_REASONING_RESERVE: "10%",
|
|
16
17
|
PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
|
|
17
18
|
PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
|
|
19
|
+
PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
|
|
18
20
|
PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "1",
|
|
19
21
|
};
|
|
20
22
|
|
|
@@ -39,6 +41,22 @@ test("catalog provider resolves model physics and Models.dev USD rates", () => {
|
|
|
39
41
|
}) ?? 0) > 0);
|
|
40
42
|
});
|
|
41
43
|
|
|
44
|
+
test("an operator context window caps catalog physics and percentage reserves derive from the cap", () => {
|
|
45
|
+
const provider = catalogProviderFromEnv("openai", {
|
|
46
|
+
...env,
|
|
47
|
+
PLURNK_PROVIDERS_CONTEXT_WINDOW: "128000",
|
|
48
|
+
}, "gpt-4.1-mini");
|
|
49
|
+
assert.equal(provider?.contextWindow, 128_000);
|
|
50
|
+
assert.equal(provider?.reasoningReserve, 16_000);
|
|
51
|
+
assert.equal(provider?.completionReserve, 32_000);
|
|
52
|
+
|
|
53
|
+
const oversized = catalogProviderFromEnv("openai", {
|
|
54
|
+
...env,
|
|
55
|
+
PLURNK_PROVIDERS_CONTEXT_WINDOW: "2000000",
|
|
56
|
+
}, "gpt-4.1-mini");
|
|
57
|
+
assert.equal(oversized?.contextWindow, 1_047_576, "an operator ceiling cannot enlarge model physics");
|
|
58
|
+
});
|
|
59
|
+
|
|
42
60
|
test("official AI SDK provider owns the native request while PLURNK owns call settings", async () => {
|
|
43
61
|
const calls: Array<{ url: string; body: Record<string, unknown> }> = [];
|
|
44
62
|
mock.method(globalThis, "fetch", async (input: string | URL | Request, init?: RequestInit) => {
|
|
@@ -102,22 +120,32 @@ test("cataloged unknown model fails unless its context is explicit", () => {
|
|
|
102
120
|
assert.equal(provider?.contextWindow, 8192);
|
|
103
121
|
});
|
|
104
122
|
|
|
105
|
-
test("
|
|
106
|
-
const
|
|
123
|
+
test("Models.dev is the only fallback rate table", () => {
|
|
124
|
+
const usage = {
|
|
125
|
+
prompt: 1_000,
|
|
126
|
+
cached: 400,
|
|
127
|
+
completion: 100,
|
|
128
|
+
reasoning: 50,
|
|
129
|
+
total: 1_150,
|
|
130
|
+
};
|
|
131
|
+
const cataloged = catalogProviderFromEnv("deepseek", {
|
|
132
|
+
...env,
|
|
133
|
+
DEEPSEEK_API_KEY: "test-key",
|
|
134
|
+
}, "deepseek-v4-flash");
|
|
135
|
+
assert.notEqual(cataloged, null);
|
|
136
|
+
assert.deepEqual(providerCostFor(cataloged!, usage), {
|
|
137
|
+
kind: "estimated",
|
|
138
|
+
usd: "0.00012712",
|
|
139
|
+
source: "Models.dev catalog rates",
|
|
140
|
+
});
|
|
141
|
+
|
|
142
|
+
const uncataloged = catalogProviderFromEnv("xai", {
|
|
107
143
|
...env,
|
|
108
144
|
XAI_API_KEY: "test-key",
|
|
109
145
|
PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
prompt: 1_000_000,
|
|
116
|
-
cached: 250_000,
|
|
117
|
-
completion: 500_000,
|
|
118
|
-
reasoning: 500_000,
|
|
119
|
-
total: 2_000_000,
|
|
120
|
-
};
|
|
121
|
-
assert.equal(catalogProviderFromEnv("xai", priced, "not-in-the-catalog")?.calculateCost(usage), 9.625);
|
|
122
|
-
assert.equal(catalogProviderFromEnv("openai", priced, "gpt-4.1-mini")?.calculateCost(usage), 9.625);
|
|
146
|
+
}, "not-in-the-catalog");
|
|
147
|
+
assert.deepEqual(providerCostFor(uncataloged!, usage), {
|
|
148
|
+
kind: "unknown",
|
|
149
|
+
reason: "the response reported no cost and Models.dev has no rate for this model",
|
|
150
|
+
});
|
|
123
151
|
});
|
package/src/catalogProvider.ts
CHANGED
|
@@ -1,23 +1,26 @@
|
|
|
1
1
|
import { lookupProvider, resolveModel, type ModelInfo } from "@plurnk/plurnk-models";
|
|
2
2
|
import {
|
|
3
3
|
contextWindowFromEnv,
|
|
4
|
+
effectiveContextWindow,
|
|
4
5
|
dataCaptureFromEnv,
|
|
5
6
|
envelopeFromEnv,
|
|
6
7
|
parseRequiredFloat,
|
|
7
8
|
parseRequiredInt,
|
|
8
9
|
promptCacheKeyFromEnv,
|
|
9
10
|
reasoningFromEnv,
|
|
11
|
+
reasoningResponseStyleFromEnv,
|
|
10
12
|
resolveReserve,
|
|
11
|
-
tokenRatesFromEnv,
|
|
12
13
|
type ReserveSpec,
|
|
13
14
|
} from "./env.ts";
|
|
14
15
|
import AiSdkProvider, { type ReasoningStyle } from "./AiSdkProvider.ts";
|
|
15
16
|
import { configuredProviderInfo, createSdkModel } from "./sdkModels.ts";
|
|
16
|
-
import { providerSource } from "./
|
|
17
|
-
import type { Provider, ProviderUsage } from "./types.ts";
|
|
18
|
-
import { calculateCostUsd } from "./usage.ts";
|
|
17
|
+
import { providerSource } from "./notices.ts";
|
|
18
|
+
import type { AuthoritativeChargeNormalizer, Provider, ProviderUsage } from "./types.ts";
|
|
19
|
+
import { calculateCostUsd, calculateCostUsdDecimal } from "./usage.ts";
|
|
19
20
|
import { emitWarningOnce } from "./warnings.ts";
|
|
20
21
|
import type { LanguageModel } from "ai";
|
|
22
|
+
import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
23
|
+
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
21
24
|
|
|
22
25
|
const reasoningStyleFromEnv = (
|
|
23
26
|
env: NodeJS.ProcessEnv,
|
|
@@ -28,7 +31,7 @@ const reasoningStyleFromEnv = (
|
|
|
28
31
|
if (value === undefined || value.length === 0) return undefined;
|
|
29
32
|
const styles: readonly ReasoningStyle[] = [
|
|
30
33
|
"none", "think", "include_reasoning", "effort",
|
|
31
|
-
"effort_explicit", "template", "anthropic",
|
|
34
|
+
"effort_explicit", "thinking_effort", "template", "anthropic",
|
|
32
35
|
];
|
|
33
36
|
if (!styles.includes(value as ReasoningStyle)) {
|
|
34
37
|
throw new Error(`${name} provider: PLURNK_PROVIDERS_PROVIDER_${prefix}_REASONING_STYLE has invalid value "${value}"`);
|
|
@@ -41,23 +44,27 @@ export const providerFromSdkModel = ({
|
|
|
41
44
|
env,
|
|
42
45
|
model,
|
|
43
46
|
languageModel,
|
|
47
|
+
normalizeCharge,
|
|
44
48
|
url,
|
|
45
49
|
headers,
|
|
46
50
|
contextWindow,
|
|
47
51
|
info,
|
|
52
|
+
attributions,
|
|
48
53
|
}: {
|
|
49
54
|
name: string;
|
|
50
55
|
env: NodeJS.ProcessEnv;
|
|
51
56
|
model: string;
|
|
52
57
|
languageModel?: LanguageModel;
|
|
58
|
+
normalizeCharge?: AuthoritativeChargeNormalizer;
|
|
53
59
|
url?: string;
|
|
54
60
|
headers?: Readonly<Record<string, string>>;
|
|
55
61
|
contextWindow: number;
|
|
56
62
|
info?: ModelInfo;
|
|
63
|
+
attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
57
64
|
}): Provider => {
|
|
58
65
|
emitWarningOnce(
|
|
59
|
-
`${name} provider:
|
|
60
|
-
"
|
|
66
|
+
`${name} provider: physical prompt counting is a chars/2 estimate; over-policy recovery fails closed without exact or bounded request evidence`,
|
|
67
|
+
"PLURNK_PROMPT_COUNT_ESTIMATE",
|
|
61
68
|
);
|
|
62
69
|
|
|
63
70
|
const reasoning = reasoningFromEnv(env, name);
|
|
@@ -76,36 +83,49 @@ export const providerFromSdkModel = ({
|
|
|
76
83
|
? configuredReasoning
|
|
77
84
|
: { tokens: Math.round(completionTokens / 2) };
|
|
78
85
|
|
|
79
|
-
const configuredRates = tokenRatesFromEnv(env, name);
|
|
80
86
|
const catalogCost = info?.cost;
|
|
81
|
-
const rates =
|
|
87
|
+
const rates = catalogCost === undefined ? null : {
|
|
82
88
|
input: catalogCost.inputPer1M,
|
|
83
89
|
output: catalogCost.outputPer1M,
|
|
84
90
|
cached: catalogCost.cacheReadPer1M ?? catalogCost.inputPer1M,
|
|
85
|
-
}
|
|
91
|
+
};
|
|
86
92
|
const calculateCost = rates === null
|
|
87
93
|
? undefined
|
|
88
94
|
: (usage: ProviderUsage): number => calculateCostUsd(usage, rates);
|
|
95
|
+
const calculateCharge: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }> = rates === null
|
|
96
|
+
? () => ({ kind: "unknown", reason: "the response reported no cost and Models.dev has no rate for this model" })
|
|
97
|
+
: rates.input === 0 && rates.output === 0 && rates.cached === 0
|
|
98
|
+
? () => ({ kind: "free", source: "Models.dev catalog rates" })
|
|
99
|
+
: (usage: ProviderUsage) => ({
|
|
100
|
+
kind: "estimated",
|
|
101
|
+
usd: calculateCostUsdDecimal(usage, rates),
|
|
102
|
+
source: "Models.dev catalog rates",
|
|
103
|
+
});
|
|
89
104
|
|
|
90
105
|
return new AiSdkProvider({
|
|
91
106
|
model,
|
|
107
|
+
...(attributions === undefined ? {} : { attributions }),
|
|
92
108
|
...(languageModel === undefined ? {} : { languageModel }),
|
|
109
|
+
...(normalizeCharge === undefined ? {} : { normalizeCharge }),
|
|
93
110
|
...(url === undefined ? {} : { url }),
|
|
94
111
|
...(headers === undefined ? {} : { headers: { ...headers } }),
|
|
95
112
|
contextWindow,
|
|
96
113
|
fetchTimeoutMs: parseRequiredInt(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name),
|
|
97
114
|
streamIdleTimeoutMs: parseRequiredInt(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name),
|
|
98
115
|
reasoning,
|
|
116
|
+
reasoningResponseStyle: reasoningResponseStyleFromEnv(env, name),
|
|
99
117
|
temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
|
|
100
118
|
repeatPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
|
|
101
119
|
frequencyPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_FREQUENCY_PENALTY, "PLURNK_PROVIDERS_FREQUENCY_PENALTY", name, 0),
|
|
102
120
|
reasoningReserve,
|
|
103
121
|
completionReserve,
|
|
104
122
|
retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
|
|
123
|
+
errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", name),
|
|
105
124
|
reasoningStyle: reasoningStyleFromEnv(env, name),
|
|
106
125
|
promptCacheKey: url === undefined ? false : promptCacheKeyFromEnv(env, name),
|
|
107
126
|
serviceTier: env.PLURNK_PROVIDERS_SERVICE_TIER,
|
|
108
127
|
calculateCost,
|
|
128
|
+
calculateCharge,
|
|
109
129
|
source: providerSource(name),
|
|
110
130
|
gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined
|
|
111
131
|
&& env.PLURNK_PROVIDERS_GBNF_DEBUG !== ""
|
|
@@ -133,13 +153,8 @@ export const catalogProviderFromEnv = (
|
|
|
133
153
|
const sdk = createSdkModel(name, wireModel, env, baseUrlOverride);
|
|
134
154
|
if (sdk === null) return null;
|
|
135
155
|
|
|
136
|
-
emitWarningOnce(
|
|
137
|
-
`${name} provider: countTokens is a chars/2 upper bound — exact counts come from the mimetypes tokenizer seam or tokenize()`,
|
|
138
|
-
"PLURNK_TOKENIZER_HEURISTIC",
|
|
139
|
-
);
|
|
140
|
-
|
|
141
156
|
const info = resolved?.info;
|
|
142
|
-
const contextWindow = contextOverride
|
|
157
|
+
const contextWindow = effectiveContextWindow(contextOverride, info?.contextWindow ?? null);
|
|
143
158
|
if (contextWindow === null) {
|
|
144
159
|
throw new Error(
|
|
145
160
|
`${name} provider: context window unresolved for "${wireModel}" — set PLURNK_PROVIDERS_CONTEXT_WINDOW or update the Models.dev snapshot`,
|
|
@@ -150,6 +165,7 @@ export const catalogProviderFromEnv = (
|
|
|
150
165
|
env,
|
|
151
166
|
model: wireModel,
|
|
152
167
|
languageModel: sdk.languageModel,
|
|
168
|
+
normalizeCharge: sdk.normalizeCharge,
|
|
153
169
|
url: sdk.compatible?.url,
|
|
154
170
|
headers: sdk.compatible?.headers,
|
|
155
171
|
contextWindow,
|
|
@@ -13,6 +13,7 @@ const env = {
|
|
|
13
13
|
PLURNK_PROVIDERS_REASONING_RESERVE: "10%",
|
|
14
14
|
PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
|
|
15
15
|
PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
|
|
16
|
+
PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
|
|
16
17
|
PLURNK_PROVIDERS_PROBE_ATTEMPTS: "1",
|
|
17
18
|
PLURNK_PROVIDERS_PROBE_DELAY: "0",
|
|
18
19
|
PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "1",
|
|
@@ -42,3 +43,98 @@ test("compatible endpoints preserve configured prompt-cache affinity", async ()
|
|
|
42
43
|
|
|
43
44
|
assert.equal(body?.prompt_cache_key, "worker-affinity");
|
|
44
45
|
});
|
|
46
|
+
|
|
47
|
+
test("the server-wide DRY-off floor emits no DRY request fields", async () => {
|
|
48
|
+
let body: Record<string, unknown> | undefined;
|
|
49
|
+
mock.method(globalThis, "fetch", async (input: string | URL | Request, init?: RequestInit) => {
|
|
50
|
+
if (String(input).endsWith("/models")) {
|
|
51
|
+
return new Response(JSON.stringify({
|
|
52
|
+
data: [{ id: "local", meta: { n_ctx: 8192 } }],
|
|
53
|
+
}));
|
|
54
|
+
}
|
|
55
|
+
body = JSON.parse(String(init?.body)) as Record<string, unknown>;
|
|
56
|
+
return new Response(JSON.stringify({
|
|
57
|
+
model: "local",
|
|
58
|
+
choices: [{ message: { content: "ok" }, finish_reason: "stop" }],
|
|
59
|
+
usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 },
|
|
60
|
+
}), { headers: { "content-type": "application/json" } });
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
const provider = await compatibleProviderFromEnv("openai", {
|
|
64
|
+
...env,
|
|
65
|
+
PLURNK_PROVIDERS_DRY_MULTIPLIER: "0",
|
|
66
|
+
// Stale or independently supplied shape values cannot activate DRY.
|
|
67
|
+
PLURNK_PROVIDERS_DRY_BASE: "1.75",
|
|
68
|
+
PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH: "32",
|
|
69
|
+
}, "local");
|
|
70
|
+
await provider.generate({
|
|
71
|
+
workerId: "worker-dry-off",
|
|
72
|
+
messages: [{ role: "user", content: "repeat exactly" }],
|
|
73
|
+
});
|
|
74
|
+
|
|
75
|
+
assert.equal(body?.repeat_penalty, 1.15);
|
|
76
|
+
assert.equal("dry_multiplier" in (body ?? {}), false);
|
|
77
|
+
assert.equal("dry_base" in (body ?? {}), false);
|
|
78
|
+
assert.equal("dry_allowed_length" in (body ?? {}), false);
|
|
79
|
+
});
|
|
80
|
+
|
|
81
|
+
test("detected llama-server measures the complete chat request through input_tokens", async () => {
|
|
82
|
+
let countUrl: string | undefined;
|
|
83
|
+
let countBody: Record<string, unknown> | undefined;
|
|
84
|
+
mock.method(globalThis, "fetch", async (input: string | URL | Request, init?: RequestInit) => {
|
|
85
|
+
const url = String(input);
|
|
86
|
+
if (url.endsWith("/models")) {
|
|
87
|
+
return new Response(JSON.stringify({
|
|
88
|
+
data: [{ id: "served.gguf", meta: { n_ctx: 8192 } }],
|
|
89
|
+
}));
|
|
90
|
+
}
|
|
91
|
+
if (url.endsWith("/props")) {
|
|
92
|
+
return new Response(JSON.stringify({ total_slots: 1 }));
|
|
93
|
+
}
|
|
94
|
+
if (url.endsWith("/chat/completions/input_tokens")) {
|
|
95
|
+
countUrl = url;
|
|
96
|
+
countBody = JSON.parse(String(init?.body)) as Record<string, unknown>;
|
|
97
|
+
return new Response(JSON.stringify({ input_tokens: 37 }), {
|
|
98
|
+
headers: { "content-type": "application/json" },
|
|
99
|
+
});
|
|
100
|
+
}
|
|
101
|
+
throw new Error(`unexpected request ${url}`);
|
|
102
|
+
});
|
|
103
|
+
|
|
104
|
+
const provider = await compatibleProviderFromEnv("openai", env, "local");
|
|
105
|
+
const messages = [
|
|
106
|
+
{ role: "system" as const, content: "system slot" },
|
|
107
|
+
{ role: "user" as const, content: "漢漢漢" },
|
|
108
|
+
];
|
|
109
|
+
assert.deepEqual(await provider.countPromptTokens(messages), {
|
|
110
|
+
kind: "exact",
|
|
111
|
+
tokens: 37,
|
|
112
|
+
source: "llama-server:/v1/chat/completions/input_tokens",
|
|
113
|
+
});
|
|
114
|
+
assert.equal(countUrl, "http://local.test/v1/chat/completions/input_tokens");
|
|
115
|
+
assert.deepEqual(countBody?.messages, messages, "measurement receives the exact dispatched message slots");
|
|
116
|
+
assert.equal(countBody?.model, "local");
|
|
117
|
+
assert.deepEqual(countBody?.chat_template_kwargs, { enable_thinking: false });
|
|
118
|
+
});
|
|
119
|
+
|
|
120
|
+
test("a missing llama-server input-token endpoint degrades explicitly, never to a claimed bound", async () => {
|
|
121
|
+
mock.method(globalThis, "fetch", async (input: string | URL | Request) => {
|
|
122
|
+
const url = String(input);
|
|
123
|
+
if (url.endsWith("/models")) {
|
|
124
|
+
return new Response(JSON.stringify({
|
|
125
|
+
data: [{ id: "served.gguf", meta: { n_ctx: 8192 } }],
|
|
126
|
+
}));
|
|
127
|
+
}
|
|
128
|
+
if (url.endsWith("/props")) return new Response(JSON.stringify({ total_slots: 1 }));
|
|
129
|
+
if (url.endsWith("/chat/completions/input_tokens")) return new Response("missing", { status: 404 });
|
|
130
|
+
throw new Error(`unexpected request ${url}`);
|
|
131
|
+
});
|
|
132
|
+
|
|
133
|
+
const provider = await compatibleProviderFromEnv("openai", env, "local");
|
|
134
|
+
assert.deepEqual(await provider.countPromptTokens([{ role: "user", content: "漢漢漢" }]), {
|
|
135
|
+
kind: "estimate",
|
|
136
|
+
tokens: 2,
|
|
137
|
+
source: "heuristic:chars2",
|
|
138
|
+
detail: "llama-server input-token endpoint returned HTTP 404",
|
|
139
|
+
});
|
|
140
|
+
});
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import AiSdkProvider, { type GrammarStyle, type ReasoningStyle } from "./AiSdkProvider.ts";
|
|
2
2
|
import {
|
|
3
3
|
contextWindowFromEnv,
|
|
4
|
+
effectiveContextWindow,
|
|
4
5
|
dataCaptureFromEnv,
|
|
5
6
|
envelopeFromEnv,
|
|
6
7
|
parseOptionalFloat,
|
|
@@ -9,8 +10,9 @@ import {
|
|
|
9
10
|
parseRequiredInt,
|
|
10
11
|
promptCacheKeyFromEnv,
|
|
11
12
|
reasoningFromEnv,
|
|
13
|
+
reasoningResponseStyleFromEnv,
|
|
12
14
|
} from "./env.ts";
|
|
13
|
-
import { providerSource } from "./
|
|
15
|
+
import { providerSource } from "./notices.ts";
|
|
14
16
|
import type { Provider } from "./types.ts";
|
|
15
17
|
import { emitWarningOnce } from "./warnings.ts";
|
|
16
18
|
|
|
@@ -125,7 +127,7 @@ export const compatibleProviderFromEnv = async (
|
|
|
125
127
|
attempts,
|
|
126
128
|
parseRequiredInt(env.PLURNK_PROVIDERS_PROBE_DELAY, "PLURNK_PROVIDERS_PROBE_DELAY", provider),
|
|
127
129
|
);
|
|
128
|
-
|
|
130
|
+
const contextWindow = effectiveContextWindow(contextWindowFromEnv(env, provider), probe.nCtx);
|
|
129
131
|
const pinRaw = env.PLURNK_PROVIDERS_LLAMA_SERVER;
|
|
130
132
|
if (pinRaw !== undefined && pinRaw !== "" && pinRaw !== "0" && pinRaw !== "1") {
|
|
131
133
|
throw new Error(`${provider} provider: PLURNK_PROVIDERS_LLAMA_SERVER must be "1", "0", or unset`);
|
|
@@ -150,6 +152,7 @@ export const compatibleProviderFromEnv = async (
|
|
|
150
152
|
let slotCount: number | null = null;
|
|
151
153
|
let eosText: string | undefined;
|
|
152
154
|
let tokenizeUrl: string | undefined;
|
|
155
|
+
let promptTokensUrl: string | undefined;
|
|
153
156
|
if (llamaServer) {
|
|
154
157
|
grammarStyle = "llamacpp";
|
|
155
158
|
reasoningStyle = "template";
|
|
@@ -157,12 +160,15 @@ export const compatibleProviderFromEnv = async (
|
|
|
157
160
|
slotCount = props.slotCount;
|
|
158
161
|
eosText = props.eosText ?? undefined;
|
|
159
162
|
tokenizeUrl = url.replace(/\/v1\/chat\/completions$/, "/tokenize");
|
|
163
|
+
promptTokensUrl = url.replace(/\/v1\/chat\/completions$/, "/v1/chat/completions/input_tokens");
|
|
160
164
|
}
|
|
161
165
|
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
+
if (!llamaServer) {
|
|
167
|
+
emitWarningOnce(
|
|
168
|
+
`${provider} provider: physical prompt counting is a chars/2 estimate; over-policy recovery fails closed without exact or bounded request evidence`,
|
|
169
|
+
"PLURNK_PROMPT_COUNT_ESTIMATE",
|
|
170
|
+
);
|
|
171
|
+
}
|
|
166
172
|
const { reasoningReserve, completionReserve } = envelopeFromEnv(env, provider);
|
|
167
173
|
return new AiSdkProvider({
|
|
168
174
|
model,
|
|
@@ -172,6 +178,7 @@ export const compatibleProviderFromEnv = async (
|
|
|
172
178
|
fetchTimeoutMs: timeout,
|
|
173
179
|
streamIdleTimeoutMs: parseRequiredInt(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", provider),
|
|
174
180
|
reasoning: reasoningFromEnv(env, provider),
|
|
181
|
+
reasoningResponseStyle: reasoningResponseStyleFromEnv(env, provider),
|
|
175
182
|
reasoningStyle,
|
|
176
183
|
temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", provider, 0),
|
|
177
184
|
repeatPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", provider, 0),
|
|
@@ -184,6 +191,7 @@ export const compatibleProviderFromEnv = async (
|
|
|
184
191
|
completionReserve,
|
|
185
192
|
tuningFloors: provider !== "plurnk",
|
|
186
193
|
retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", provider),
|
|
194
|
+
errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", provider),
|
|
187
195
|
promptCacheKey: promptCacheKeyFromEnv(env, provider),
|
|
188
196
|
source: providerSource(provider),
|
|
189
197
|
grammarStyle,
|
|
@@ -199,6 +207,7 @@ export const compatibleProviderFromEnv = async (
|
|
|
199
207
|
slotCount,
|
|
200
208
|
eosText,
|
|
201
209
|
tokenizeUrl,
|
|
210
|
+
promptTokensUrl,
|
|
202
211
|
servedModel: probe.servedModel ?? undefined,
|
|
203
212
|
requiresMaxTokens: llamaServer || undefined,
|
|
204
213
|
});
|
package/src/cost.test.ts
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
import assert from "node:assert/strict";
|
|
2
|
+
import test from "node:test";
|
|
3
|
+
import {
|
|
4
|
+
providerCostFor,
|
|
5
|
+
providerCostUsd,
|
|
6
|
+
validateAuthoritativeCharge,
|
|
7
|
+
} from "./cost.ts";
|
|
8
|
+
import type { ProviderUsage } from "./types.ts";
|
|
9
|
+
|
|
10
|
+
const usage: ProviderUsage = {
|
|
11
|
+
prompt: 1,
|
|
12
|
+
completion: 1,
|
|
13
|
+
reasoning: 0,
|
|
14
|
+
cached: 0,
|
|
15
|
+
total: 2,
|
|
16
|
+
};
|
|
17
|
+
|
|
18
|
+
test("authoritative provider charge wins over a local estimate", () => {
|
|
19
|
+
const provider = {
|
|
20
|
+
calculateCost: () => 12,
|
|
21
|
+
calculateCharge: () => ({ kind: "estimated", usd: "12", source: "catalog" } as const),
|
|
22
|
+
};
|
|
23
|
+
const charge = {
|
|
24
|
+
kind: "authoritative",
|
|
25
|
+
amount: { amount: "0.0000042", currency: "XMR" },
|
|
26
|
+
usdEquivalent: "0.73",
|
|
27
|
+
source: "settled upstream turn charge",
|
|
28
|
+
} as const;
|
|
29
|
+
assert.deepEqual(providerCostFor(provider, usage, charge), charge);
|
|
30
|
+
assert.equal(providerCostUsd(charge), 0.73);
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
test("explicit free remains distinguishable from unknown", () => {
|
|
34
|
+
const free = providerCostFor({
|
|
35
|
+
calculateCost: () => 0,
|
|
36
|
+
calculateCharge: () => ({ kind: "free", source: "local model" }),
|
|
37
|
+
}, usage);
|
|
38
|
+
const unknown = providerCostFor({ calculateCost: () => 0 }, usage);
|
|
39
|
+
assert.deepEqual(free, { kind: "free", source: "local model" });
|
|
40
|
+
assert.deepEqual(unknown, {
|
|
41
|
+
kind: "unknown",
|
|
42
|
+
reason: "the response reported no cost and Models.dev has no rate for this model",
|
|
43
|
+
});
|
|
44
|
+
assert.equal(providerCostUsd(free), 0);
|
|
45
|
+
assert.equal(providerCostUsd(unknown), null);
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
test("legacy calculateCost is not a monetary-reporting authority", () => {
|
|
49
|
+
const cost = providerCostFor({ calculateCost: () => 0.25 }, usage);
|
|
50
|
+
assert.deepEqual(cost, {
|
|
51
|
+
kind: "unknown",
|
|
52
|
+
reason: "the response reported no cost and Models.dev has no rate for this model",
|
|
53
|
+
});
|
|
54
|
+
assert.equal(providerCostUsd(cost), null);
|
|
55
|
+
});
|
|
56
|
+
|
|
57
|
+
test("rejects malformed money instead of mining or coercing it", () => {
|
|
58
|
+
assert.throws(() => validateAuthoritativeCharge({
|
|
59
|
+
kind: "authoritative",
|
|
60
|
+
amount: { amount: "1e3", currency: "usd" },
|
|
61
|
+
usdEquivalent: "1000",
|
|
62
|
+
source: "wire",
|
|
63
|
+
}), /canonical non-negative decimal string/);
|
|
64
|
+
});
|
package/src/cost.ts
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
import type {
|
|
2
|
+
AuthoritativeCharge,
|
|
3
|
+
ProviderUsage,
|
|
4
|
+
} from "./types.ts";
|
|
5
|
+
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
6
|
+
|
|
7
|
+
const DECIMAL = /^(?:0|[1-9]\d*)(?:\.\d+)?$/;
|
|
8
|
+
const CURRENCY = /^[A-Z][A-Z0-9]{2,11}$/;
|
|
9
|
+
|
|
10
|
+
const nonEmpty = (value: string, name: string): string => {
|
|
11
|
+
if (value.trim() === "") throw new TypeError(`${name} must be non-empty`);
|
|
12
|
+
return value;
|
|
13
|
+
};
|
|
14
|
+
|
|
15
|
+
const decimal = (value: string, name: string): string => {
|
|
16
|
+
if (!DECIMAL.test(value)) throw new TypeError(`${name} must be a canonical non-negative decimal string`);
|
|
17
|
+
return value;
|
|
18
|
+
};
|
|
19
|
+
|
|
20
|
+
export const validateAuthoritativeCharge = (charge: AuthoritativeCharge): AuthoritativeCharge => {
|
|
21
|
+
if (charge.kind !== "authoritative") throw new TypeError("provider charge must be authoritative");
|
|
22
|
+
decimal(charge.amount.amount, "provider charge amount");
|
|
23
|
+
if (!CURRENCY.test(charge.amount.currency)) {
|
|
24
|
+
throw new TypeError("provider charge currency must be an uppercase currency code");
|
|
25
|
+
}
|
|
26
|
+
decimal(charge.usdEquivalent, "provider charge USD equivalent");
|
|
27
|
+
nonEmpty(charge.source, "provider charge source");
|
|
28
|
+
return charge;
|
|
29
|
+
};
|
|
30
|
+
|
|
31
|
+
export const validateProviderCost = (cost: ProviderCost): ProviderCost => {
|
|
32
|
+
switch (cost.kind) {
|
|
33
|
+
case "authoritative":
|
|
34
|
+
return validateAuthoritativeCharge(cost);
|
|
35
|
+
case "estimated":
|
|
36
|
+
decimal(cost.usd, "provider cost estimate");
|
|
37
|
+
nonEmpty(cost.source, "provider cost estimate source");
|
|
38
|
+
return cost;
|
|
39
|
+
case "free":
|
|
40
|
+
nonEmpty(cost.source, "provider free source");
|
|
41
|
+
return cost;
|
|
42
|
+
case "unknown":
|
|
43
|
+
nonEmpty(cost.reason, "provider unknown-cost reason");
|
|
44
|
+
return cost;
|
|
45
|
+
}
|
|
46
|
+
};
|
|
47
|
+
|
|
48
|
+
export const resolveProviderCost = (
|
|
49
|
+
charge: AuthoritativeCharge | undefined,
|
|
50
|
+
current: ProviderCost | undefined,
|
|
51
|
+
): ProviderCost => {
|
|
52
|
+
if (charge !== undefined) return validateAuthoritativeCharge(charge);
|
|
53
|
+
if (current !== undefined) return validateProviderCost(current);
|
|
54
|
+
return {
|
|
55
|
+
kind: "unknown",
|
|
56
|
+
reason: "the response reported no cost and Models.dev has no rate for this model",
|
|
57
|
+
};
|
|
58
|
+
};
|
|
59
|
+
|
|
60
|
+
export const providerCostUsd = (cost: ProviderCost): number | null => {
|
|
61
|
+
const value = cost.kind === "authoritative"
|
|
62
|
+
? cost.usdEquivalent
|
|
63
|
+
: cost.kind === "estimated"
|
|
64
|
+
? cost.usd
|
|
65
|
+
: cost.kind === "free"
|
|
66
|
+
? "0"
|
|
67
|
+
: null;
|
|
68
|
+
return value === null ? null : Number(value);
|
|
69
|
+
};
|
|
70
|
+
|
|
71
|
+
export const providerCostFor = (
|
|
72
|
+
provider: { calculateCharge?(usage: ProviderUsage): Exclude<ProviderCost, AuthoritativeCharge>; calculateCost(usage: ProviderUsage): number },
|
|
73
|
+
usage: ProviderUsage,
|
|
74
|
+
charge?: AuthoritativeCharge,
|
|
75
|
+
): ProviderCost => resolveProviderCost(
|
|
76
|
+
charge,
|
|
77
|
+
provider.calculateCharge?.(usage),
|
|
78
|
+
);
|
package/src/defaults.test.ts
CHANGED
|
@@ -7,6 +7,7 @@ test("withProviderDefaults supplies the package-owned operational floor", () =>
|
|
|
7
7
|
assert.equal(env.PLURNK_PROVIDERS_PROMPT_CACHE_KEY, "1");
|
|
8
8
|
assert.equal(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "600000");
|
|
9
9
|
assert.equal(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "3");
|
|
10
|
+
assert.equal(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "512");
|
|
10
11
|
});
|
|
11
12
|
|
|
12
13
|
test("withProviderDefaults preserves every explicit operator value", () => {
|