@plurnk/plurnk-providers 1.3.3 → 1.3.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +20 -21
- package/SPEC.md +65 -44
- package/dist/Mock.d.ts +1 -1
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +1 -1
- package/dist/Mock.js.map +1 -1
- package/dist/OpenAICompat.d.ts +5 -4
- package/dist/OpenAICompat.d.ts.map +1 -1
- package/dist/OpenAICompat.js +38 -38
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/Pool.d.ts +1 -1
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +1 -1
- package/dist/Pool.js.map +1 -1
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +2 -0
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +2 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -2
- package/dist/index.js.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -1
- package/dist/openai.js.map +1 -1
- package/dist/openaiStream.d.ts +6 -1
- package/dist/openaiStream.d.ts.map +1 -1
- package/dist/openaiStream.js +30 -2
- package/dist/openaiStream.js.map +1 -1
- package/dist/standardProviders.d.ts +3 -3
- package/dist/standardProviders.d.ts.map +1 -1
- package/dist/standardProviders.js +36 -17
- package/dist/standardProviders.js.map +1 -1
- package/dist/types.d.ts +1 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +1 -1
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +4 -2
- package/dist/usage.js.map +1 -1
- package/package.json +10 -7
- package/src/Mock.test.ts +2 -2
- package/src/Mock.ts +1 -1
- package/src/OpenAICompat.test.ts +86 -86
- package/src/OpenAICompat.ts +46 -45
- package/src/Pool.test.ts +3 -3
- package/src/Pool.ts +1 -1
- package/src/ProviderRegistry.test.ts +30 -1
- package/src/aiSdkAdapter.spike.test.ts +242 -0
- package/src/env.test.ts +8 -0
- package/src/env.ts +2 -0
- package/src/index.ts +2 -2
- package/src/openai.ts +1 -1
- package/src/openaiStream.ts +30 -2
- package/src/standardProviders.test.ts +45 -31
- package/src/standardProviders.ts +42 -29
- package/src/types.ts +5 -5
- package/src/usage.test.ts +8 -10
- package/src/usage.ts +7 -3
package/src/standardProviders.ts
CHANGED
|
@@ -14,7 +14,7 @@ import OpenAICompatProvider, { type ReasoningStyle, type GrammarStyle } from "./
|
|
|
14
14
|
import { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, reasoningFromEnv, dataCaptureFromEnv, contextWindowFromEnv, envelopeFromEnv, resolveReserve, type ReserveSpec } from "./env.ts";
|
|
15
15
|
import { emitWarningOnce } from "./warnings.ts";
|
|
16
16
|
import { providerSource } from "./telemetry.ts";
|
|
17
|
-
import {
|
|
17
|
+
import { calculateCostUsd } from "./usage.ts";
|
|
18
18
|
import { lookup } from "@plurnk/plurnk-models";
|
|
19
19
|
|
|
20
20
|
type StandardProviderSpec = {
|
|
@@ -59,21 +59,19 @@ type StandardProviderSpec = {
|
|
|
59
59
|
// not already include it.
|
|
60
60
|
flexBaseStrip?: boolean;
|
|
61
61
|
reasoningStyle: ReasoningStyle;
|
|
62
|
-
//
|
|
63
|
-
// probeNctx entry is upgraded to "llamacpp" when the probe sees a
|
|
64
|
-
// llama-server; cloud backends that support GBNF set their shape statically
|
|
65
|
-
// (fireworks → "response_format", verified live).
|
|
66
|
-
grammarStyle?: GrammarStyle;
|
|
67
|
-
// SSE streaming (default true). The streaming transport is dropped
|
|
68
|
-
// per-request only when it would break a feature (a response_format grammar
|
|
69
|
-
// arrives mislabeled as reasoning_content under fireworks' stream); leave
|
|
70
|
-
// unset to keep streaming on for every other call. See OpenAICompat.generate.
|
|
62
|
+
// SSE streaming (default true).
|
|
71
63
|
streaming?: boolean;
|
|
72
64
|
// Constant model-id prefix the backend requires but the alias shouldn't
|
|
73
65
|
// repeat (fireworks → "accounts/fireworks/models/", so the alias is just
|
|
74
66
|
// `fireworks/deepseek-v4-pro`). Prepended idempotently to form the wire id,
|
|
75
67
|
// which is ALSO the catalog key (models.dev keys fireworks-ai on the full id).
|
|
76
68
|
modelPrefix?: string;
|
|
69
|
+
// A fully qualified model namespace that bypasses modelPrefix. Fireworks
|
|
70
|
+
// uses sibling models/, routers/, and deployments/ resource collections.
|
|
71
|
+
qualifiedModelPrefix?: string;
|
|
72
|
+
// Fixed request tiers supported by this provider. An unset knob means the
|
|
73
|
+
// provider default; configured values are validated and sent every call.
|
|
74
|
+
serviceTiers?: readonly string[];
|
|
77
75
|
// First-party telemetry forwarding. ONLY the plurnk hosted endpoint sets
|
|
78
76
|
// this — it forwards the consumer's per-turn `attributions` (contributor
|
|
79
77
|
// credit) and `client` (originating frontend) as `Plurnk-Attribution` /
|
|
@@ -93,9 +91,6 @@ type StandardProviderSpec = {
|
|
|
93
91
|
// Default ON for standard providers (OpenAI-standard field, broadly accepted);
|
|
94
92
|
// set false to opt a backend out (e.g. anthropic's cache_control mechanism).
|
|
95
93
|
promptCacheKey?: boolean;
|
|
96
|
-
// Top-level response field the endpoint reports account balance (pico-USD) in,
|
|
97
|
-
// surfaced as ProviderResponse.balancePico (plurnk only, #23). Absent elsewhere.
|
|
98
|
-
balanceMetaKey?: string;
|
|
99
94
|
// RETIRED knob (#27→mimetypes#44): exact client-side tokenizer families were
|
|
100
95
|
// removed with the tokenizer shed — the var is kept ONLY to fail hard with a
|
|
101
96
|
// migration pointer when an operator still sets it.
|
|
@@ -155,7 +150,11 @@ export const STANDARD_PROVIDERS: Readonly<Record<string, StandardProviderSpec>>
|
|
|
155
150
|
fireworks: {
|
|
156
151
|
apiKeyVar: "FIREWORKS_API_KEY", apiKeyRequired: true,
|
|
157
152
|
baseUrlVar: "FIREWORKS_BASE_URL", chatPath: "/chat/completions",
|
|
158
|
-
reasoningStyle: "effort_explicit",
|
|
153
|
+
reasoningStyle: "effort_explicit",
|
|
154
|
+
modelPrefix: "accounts/fireworks/models/",
|
|
155
|
+
qualifiedModelPrefix: "accounts/fireworks/",
|
|
156
|
+
serviceTiers: ["auto", "default", "flex", "priority"],
|
|
157
|
+
tokenizerEnvVar: "FIREWORKS_TOKENIZER",
|
|
159
158
|
},
|
|
160
159
|
deepinfra: {
|
|
161
160
|
apiKeyVar: ["DEEPINFRA_API_KEY", "DEEPINFRA_API_TOKEN", "DEEPINFRA_TOKEN"], apiKeyRequired: true,
|
|
@@ -289,8 +288,8 @@ export const STANDARD_PROVIDERS: Readonly<Record<string, StandardProviderSpec>>
|
|
|
289
288
|
apiKeyVar: "PLURNK_API_KEY", apiKeyRequired: true,
|
|
290
289
|
apiKeyMessage: "PLURNK_API_KEY not found. Acquire one at https://plurnk.ai . Plurnk also supports local models and alternative cloud provider configurations.",
|
|
291
290
|
apiKeyRejectedMessage: "PLURNK_API_KEY was rejected by plurnk.ai (invalid or expired). Verify it at https://plurnk.ai .",
|
|
292
|
-
reasoningStyle: "none",
|
|
293
|
-
probeNctx: true, detectLlamaServer: false, firstPartyMetadata: true,
|
|
291
|
+
reasoningStyle: "none", tokenizerEnvVar: "PLURNK_TOKENIZER",
|
|
292
|
+
probeNctx: true, detectLlamaServer: false, firstPartyMetadata: true, suppressTuningFloors: true,
|
|
294
293
|
},
|
|
295
294
|
});
|
|
296
295
|
|
|
@@ -417,7 +416,8 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
417
416
|
// The on-the-wire model id: a backend-required constant prefix (fireworks)
|
|
418
417
|
// prepended idempotently, so the operator's alias carries only the distinctive
|
|
419
418
|
// tail. This id is what the backend, the probe, AND the catalog key on.
|
|
420
|
-
const
|
|
419
|
+
const qualified = spec.qualifiedModelPrefix !== undefined && model.startsWith(spec.qualifiedModelPrefix);
|
|
420
|
+
const wireModel = spec.modelPrefix !== undefined && !qualified && !model.startsWith(spec.modelPrefix)
|
|
421
421
|
? `${spec.modelPrefix}${model}`
|
|
422
422
|
: model;
|
|
423
423
|
|
|
@@ -444,17 +444,29 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
444
444
|
);
|
|
445
445
|
const url = resolveUrl(spec, env, name, baseUrlOverride);
|
|
446
446
|
const fetchTimeoutMs = parseRequiredInt(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name);
|
|
447
|
+
const streamIdleTimeoutMs = parseRequiredInt(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name);
|
|
448
|
+
const serviceTier = (() => {
|
|
449
|
+
const raw = env.PLURNK_PROVIDERS_SERVICE_TIER;
|
|
450
|
+
if (raw === undefined || raw.length === 0) return undefined;
|
|
451
|
+
if (spec.serviceTiers === undefined) {
|
|
452
|
+
throw new Error(`${name} provider: PLURNK_PROVIDERS_SERVICE_TIER is not supported`);
|
|
453
|
+
}
|
|
454
|
+
if (!spec.serviceTiers.includes(raw)) {
|
|
455
|
+
throw new Error(`${name} provider: PLURNK_PROVIDERS_SERVICE_TIER must be one of ${spec.serviceTiers.map((tier) => JSON.stringify(tier)).join(", ")} (got "${raw}")`);
|
|
456
|
+
}
|
|
457
|
+
return raw;
|
|
458
|
+
})();
|
|
447
459
|
|
|
448
460
|
// The probe always runs for probeNctx specs — grammar capability must not
|
|
449
461
|
// hinge on whether the operator pinned PLURNK_PROVIDERS_CONTEXT_WINDOW. For
|
|
450
462
|
// contextWindow itself, explicit env still wins over the probed n_ctx.
|
|
451
463
|
let contextWindow = contextWindowFromEnv(env, name);
|
|
452
|
-
//
|
|
453
|
-
//
|
|
464
|
+
// GBNF transport is upgraded to "llamacpp" only when the probe fingerprints
|
|
465
|
+
// a llama-server. Slot
|
|
454
466
|
// pinning is llama-server-only, so it keys on that same fingerprint. A spec
|
|
455
467
|
// can opt out of the fingerprint entirely (detectLlamaServer: false → plurnk)
|
|
456
468
|
// to read the window but stay a plain remote OpenAI server.
|
|
457
|
-
let grammarStyle: GrammarStyle =
|
|
469
|
+
let grammarStyle: GrammarStyle = "none";
|
|
458
470
|
let supportsSlotPinning = false;
|
|
459
471
|
let slotCount: number | null = null;
|
|
460
472
|
let eosText: string | undefined;
|
|
@@ -500,7 +512,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
500
512
|
// un-upgraded, but NEVER silently (#34) — rails going dark without a
|
|
501
513
|
// signal cost the consumer weeks of misattributed rambles.
|
|
502
514
|
emitWarningOnce(
|
|
503
|
-
`${name} provider: llama-server detection failed after ${probeAttempts} attempts —
|
|
515
|
+
`${name} provider: llama-server detection failed after ${probeAttempts} attempts — local capabilities are unknown. If this endpoint is a llama-server, pin PLURNK_PROVIDERS_LLAMA_SERVER=1 (or its _<alias> form).`,
|
|
504
516
|
"PLURNK_PROBE_FAILED",
|
|
505
517
|
);
|
|
506
518
|
}
|
|
@@ -511,7 +523,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
511
523
|
// window for a known cloud model (groq/deepseek/mistral/…, which don't
|
|
512
524
|
// probe). A local llama-server model misses the catalog and keeps its
|
|
513
525
|
// probed n_ctx. Standard providers carry NO live pricing, so the catalog is
|
|
514
|
-
// the sole — never shadowing — cost source
|
|
526
|
+
// the sole — never shadowing — cost source, expressed in USD per 1M tokens.
|
|
515
527
|
// A relay with a catalogContextLookup (bedrock) resolves its window via the
|
|
516
528
|
// underlying model's publisher and carries NO catalog cost (native rate ≠ relay
|
|
517
529
|
// rate, #22); everyone else keys the catalog directly on (name, wireModel).
|
|
@@ -536,12 +548,12 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
536
548
|
);
|
|
537
549
|
}
|
|
538
550
|
const cost = fallback?.cost;
|
|
539
|
-
const
|
|
551
|
+
const calculateCost = cost === undefined
|
|
540
552
|
? undefined
|
|
541
|
-
: (usage: ProviderUsage): number =>
|
|
542
|
-
input: cost.inputPer1M
|
|
543
|
-
output: cost.outputPer1M
|
|
544
|
-
cached:
|
|
553
|
+
: (usage: ProviderUsage): number => calculateCostUsd(usage, {
|
|
554
|
+
input: cost.inputPer1M,
|
|
555
|
+
output: cost.outputPer1M,
|
|
556
|
+
cached: cost.cacheReadPer1M ?? cost.inputPer1M,
|
|
545
557
|
});
|
|
546
558
|
|
|
547
559
|
// #507: completion cap — when the catalog reports a maxOutput, use
|
|
@@ -577,6 +589,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
577
589
|
headers,
|
|
578
590
|
contextWindow,
|
|
579
591
|
fetchTimeoutMs,
|
|
592
|
+
streamIdleTimeoutMs,
|
|
580
593
|
reasoning,
|
|
581
594
|
temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
|
|
582
595
|
repeatPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
|
|
@@ -594,7 +607,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
594
607
|
retryDelayMs: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_DELAY, "PLURNK_PROVIDERS_RETRY_DELAY", name),
|
|
595
608
|
retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
|
|
596
609
|
reasoningStyle,
|
|
597
|
-
|
|
610
|
+
calculateCost,
|
|
598
611
|
source: providerSource(name),
|
|
599
612
|
grammarStyle,
|
|
600
613
|
// Optional debug toggle (off by default): validate a transported grammar
|
|
@@ -607,7 +620,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
607
620
|
firstPartyMetadata: spec.firstPartyMetadata,
|
|
608
621
|
apiKeyRejectedMessage: spec.apiKeyRejectedMessage,
|
|
609
622
|
promptCacheKey: spec.promptCacheKey ?? true, // #518: default-on for standard providers (OpenAI-standard field, 6/6 backends verified accept it); per-spec opt-out below
|
|
610
|
-
|
|
623
|
+
serviceTier,
|
|
611
624
|
supportsSlotPinning,
|
|
612
625
|
slotCount,
|
|
613
626
|
eosText,
|
package/src/types.ts
CHANGED
|
@@ -69,9 +69,9 @@ export interface ProviderResponse {
|
|
|
69
69
|
readonly assistant: ProviderAssistant;
|
|
70
70
|
readonly assistantRaw: unknown;
|
|
71
71
|
// Per-turn provider→client metadata bag: the backend's non-standard top-level
|
|
72
|
-
// response fields
|
|
73
|
-
//
|
|
74
|
-
//
|
|
72
|
+
// response fields passed through verbatim. Monetary values carry their own
|
|
73
|
+
// amount and currency; the provider does not reinterpret them. The consumer
|
|
74
|
+
// (service) merges this into its Turn metadata and
|
|
75
75
|
// filters what reaches the client; it reads `meta`, never mines `assistantRaw`.
|
|
76
76
|
// Absent when the backend reported no extra fields (#23, generalized).
|
|
77
77
|
readonly meta?: Record<string, unknown>;
|
|
@@ -191,9 +191,9 @@ export interface Provider {
|
|
|
191
191
|
// `tokenize === undefined` means the backend can't. Exact-counting
|
|
192
192
|
// consumers (the tokenizer seam) prefer this over any client-side data.
|
|
193
193
|
tokenize?(text: string): Promise<number[]>;
|
|
194
|
-
// Provider-owned cost calculation. Returns
|
|
194
|
+
// Provider-owned estimated cost calculation. Returns USD.
|
|
195
195
|
// Returns 0 for siblings/models with no known rates.
|
|
196
|
-
|
|
196
|
+
calculateCost(usage: ProviderUsage): number;
|
|
197
197
|
}
|
|
198
198
|
|
|
199
199
|
// ProviderAlias moved to @plurnk/plurnk-aliases (the zero-dep parser, #27);
|
package/src/usage.test.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import test from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
|
-
import { normalizeUsage,
|
|
3
|
+
import { normalizeUsage, calculateCostUsd } from "./usage.ts";
|
|
4
4
|
|
|
5
5
|
// — normalizeUsage —
|
|
6
6
|
|
|
@@ -115,22 +115,20 @@ test("normalizeUsage: no total reported -> re-split skipped, reasoning stays 0 (
|
|
|
115
115
|
assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
|
|
116
116
|
});
|
|
117
117
|
|
|
118
|
-
// —
|
|
118
|
+
// — calculateCostUsd —
|
|
119
119
|
|
|
120
|
-
test("
|
|
120
|
+
test("calculateCostUsd: bills reasoning at the USD-per-million output rate", () => {
|
|
121
121
|
// 100 input, 0 cached, 50 completion + 200 reasoning = 250 output.
|
|
122
122
|
const usage = { prompt: 100, completion: 50, reasoning: 200, cached: 0, total: 350 };
|
|
123
|
-
|
|
124
|
-
assert.equal(computeCost(usage, { input: 1, output: 10, cached: 0 }), 2600);
|
|
123
|
+
assert.equal(calculateCostUsd(usage, { input: 1, output: 10, cached: 0 }), 0.0026);
|
|
125
124
|
});
|
|
126
125
|
|
|
127
|
-
test("
|
|
126
|
+
test("calculateCostUsd: cached prompt billed at the cache rate, remainder at input", () => {
|
|
128
127
|
const usage = { prompt: 1000, completion: 0, reasoning: 0, cached: 400, total: 1000 };
|
|
129
|
-
|
|
130
|
-
assert.equal(computeCost(usage, { input: 5, output: 99, cached: 1 }), 3400);
|
|
128
|
+
assert.equal(calculateCostUsd(usage, { input: 5, output: 99, cached: 1 }), 0.0034);
|
|
131
129
|
});
|
|
132
130
|
|
|
133
|
-
test("
|
|
131
|
+
test("calculateCostUsd: zero rates → 0", () => {
|
|
134
132
|
const usage = { prompt: 9, completion: 9, reasoning: 9, cached: 9, total: 27 };
|
|
135
|
-
assert.equal(
|
|
133
|
+
assert.equal(calculateCostUsd(usage, { input: 0, output: 0, cached: 0 }), 0);
|
|
136
134
|
});
|
package/src/usage.ts
CHANGED
|
@@ -69,14 +69,18 @@ export const normalizeUsage = (raw: RawUsage | null | undefined, reasoningText =
|
|
|
69
69
|
return { prompt, completion, reasoning, cached, total };
|
|
70
70
|
};
|
|
71
71
|
|
|
72
|
-
//
|
|
72
|
+
// Conventional provider pricing: USD per million tokens, matching Models.dev.
|
|
73
73
|
export type TokenRates = { input: number; output: number; cached: number };
|
|
74
74
|
|
|
75
75
|
// The one cost formula every provider uses: non-cached prompt at the input
|
|
76
76
|
// rate, cached prompt at the cache rate, and billable output (completion +
|
|
77
77
|
// reasoning) at the output rate.
|
|
78
|
-
export const
|
|
78
|
+
export const calculateCostUsd = (usage: ProviderUsage, rates: TokenRates): number => {
|
|
79
79
|
const nonCachedPrompt = Math.max(0, usage.prompt - usage.cached);
|
|
80
80
|
const output = usage.completion + usage.reasoning;
|
|
81
|
-
return
|
|
81
|
+
return (
|
|
82
|
+
nonCachedPrompt * rates.input
|
|
83
|
+
+ usage.cached * rates.cached
|
|
84
|
+
+ output * rates.output
|
|
85
|
+
) / 1_000_000;
|
|
82
86
|
};
|