@plurnk/plurnk-providers 1.5.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +36 -22
- package/SPEC.md +133 -59
- package/dist/AiSdkProvider.d.ts +19 -26
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +318 -106
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +4 -9
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +36 -9
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -21
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +19 -14
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts +5 -2
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +100 -16
- package/dist/accounting.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +9 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +160 -62
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +7 -3
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +30 -24
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +18 -7
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +10 -10
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +90 -42
- package/dist/cost.js.map +1 -1
- package/dist/env.d.ts +5 -1
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +30 -10
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +14 -2
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +58 -2
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -2
- package/dist/index.js.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/sdkModels.d.ts +6 -2
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +38 -5
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +33 -31
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +21 -5
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +164 -83
- package/dist/usage.js.map +1 -1
- package/package.json +7 -6
- package/src/AiSdkProvider.test.ts +788 -191
- package/src/AiSdkProvider.ts +381 -124
- package/src/Mock.test.ts +37 -12
- package/src/Mock.ts +45 -14
- package/src/Pool.test.ts +19 -6
- package/src/Pool.ts +20 -16
- package/src/ProviderRegistry.test.ts +16 -11
- package/src/accounting.test.ts +58 -22
- package/src/accounting.ts +120 -18
- package/src/aiSdkTransport.test.ts +42 -49
- package/src/aiSdkTransport.ts +174 -62
- package/src/boundaries.test.ts +1 -0
- package/src/catalogProvider.test.ts +258 -22
- package/src/catalogProvider.ts +42 -27
- package/src/compatibleProvider.test.ts +6 -3
- package/src/compatibleProvider.ts +20 -7
- package/src/cost.test.ts +55 -36
- package/src/cost.ts +111 -50
- package/src/defaults.test.ts +13 -3
- package/src/env.test.ts +54 -5
- package/src/env.ts +43 -18
- package/src/errors.test.ts +47 -2
- package/src/errors.ts +67 -3
- package/src/index.ts +21 -5
- package/src/ollama.test.ts +4 -1
- package/src/ollama.ts +3 -3
- package/src/sdkModels.test.ts +76 -4
- package/src/sdkModels.ts +45 -7
- package/src/types.ts +77 -38
- package/src/usage.test.ts +112 -116
- package/src/usage.ts +209 -93
package/src/types.ts
CHANGED
|
@@ -9,15 +9,32 @@ import type {
|
|
|
9
9
|
PluginAttributionContext,
|
|
10
10
|
PluginAttributionSource,
|
|
11
11
|
} from "@plurnk/plurnk-meta";
|
|
12
|
-
import type {
|
|
12
|
+
import type {
|
|
13
|
+
ProviderAccounting,
|
|
14
|
+
ProviderCost,
|
|
15
|
+
ProviderRequestAccounting,
|
|
16
|
+
ProviderUsage,
|
|
17
|
+
} from "@plurnk/plurnk-contracts";
|
|
18
|
+
|
|
19
|
+
export type {
|
|
20
|
+
ProviderAccounting,
|
|
21
|
+
ProviderCost,
|
|
22
|
+
ProviderRequestAccounting,
|
|
23
|
+
ProviderUsage,
|
|
24
|
+
} from "@plurnk/plurnk-contracts";
|
|
13
25
|
|
|
14
26
|
export interface ChatMessage {
|
|
15
27
|
role: "system" | "user" | "assistant";
|
|
16
28
|
content: string;
|
|
17
29
|
}
|
|
18
30
|
|
|
31
|
+
// {§provider-call-kind} The caller-owned semantic contract for one logical model call. Providers do
|
|
32
|
+
// not infer this from messages or grammar presence: an emission expects PLURNK
|
|
33
|
+
// output, while a bare call expects unconstrained response text.
|
|
34
|
+
export type ProviderCallKind = "emission" | "bare";
|
|
35
|
+
|
|
19
36
|
// Preflight evidence for the complete provider request. An empirical estimate
|
|
20
|
-
// is useful telemetry but cannot authorize
|
|
37
|
+
// is useful telemetry but cannot authorize hard context-envelope admission.
|
|
21
38
|
export type PromptTokenMeasurement =
|
|
22
39
|
| {
|
|
23
40
|
readonly kind: "exact" | "upper_bound";
|
|
@@ -31,39 +48,44 @@ export type PromptTokenMeasurement =
|
|
|
31
48
|
readonly detail: string;
|
|
32
49
|
};
|
|
33
50
|
|
|
34
|
-
|
|
35
|
-
// provider boundary): total = prompt + completion + reasoning; cached is a
|
|
36
|
-
// subset of prompt. `completion` is visible output EXCLUDING reasoning; the
|
|
37
|
-
// billable output is `completion + reasoning` (frontier providers bill reasoning
|
|
38
|
-
// tokens at the output rate).
|
|
39
|
-
export interface ProviderUsage {
|
|
40
|
-
readonly prompt: number; // input tokens (cached ones included)
|
|
41
|
-
readonly completion: number; // visible output tokens, excluding reasoning
|
|
42
|
-
readonly reasoning: number; // reasoning tokens, billed as output
|
|
43
|
-
readonly cached: number; // subset of prompt served from cache
|
|
44
|
-
readonly total: number; // prompt + completion + reasoning
|
|
45
|
-
}
|
|
46
|
-
|
|
47
|
-
export type AuthoritativeCharge = Extract<ProviderCost, { kind: "authoritative" }>;
|
|
51
|
+
export type ChargedCost = Extract<ProviderCost, { kind: "charged" }>;
|
|
48
52
|
|
|
49
53
|
// Evidence exposed by the transport to the provider adapter that owns its
|
|
50
54
|
// vendor protocol. Core and downstream consumers receive only the normalized
|
|
51
55
|
// charge, never a requirement to understand provider metadata fields.
|
|
52
56
|
export interface ProviderChargeEvidence {
|
|
53
57
|
readonly providerMetadata?: unknown;
|
|
58
|
+
// A protocol-owned direct monetary field. It remains unknown until the
|
|
59
|
+
// selected adapter explicitly validates and normalizes it.
|
|
60
|
+
readonly charge?: unknown;
|
|
54
61
|
// Provider-owned raw usage projection retained independently of optional
|
|
55
62
|
// full-body capture. Accounting fields cannot disappear merely because
|
|
56
63
|
// forensic raw-body capture is disabled.
|
|
57
64
|
readonly usage?: unknown;
|
|
58
65
|
readonly response: {
|
|
59
|
-
readonly id
|
|
66
|
+
readonly id?: string;
|
|
60
67
|
readonly headers?: Readonly<Record<string, string>>;
|
|
61
68
|
};
|
|
62
69
|
}
|
|
63
70
|
|
|
64
|
-
export type
|
|
71
|
+
export type ProviderCostNormalizer = (
|
|
65
72
|
evidence: ProviderChargeEvidence,
|
|
66
|
-
) =>
|
|
73
|
+
) => ProviderCost | undefined;
|
|
74
|
+
|
|
75
|
+
export interface ProviderRequestIdentity {
|
|
76
|
+
readonly provider: string;
|
|
77
|
+
readonly model: string;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
export type ProviderRequestSettlement = (
|
|
81
|
+
accounting: ProviderRequestAccounting,
|
|
82
|
+
) => Promise<void>;
|
|
83
|
+
|
|
84
|
+
// Core opens durable physical-request identity through this observer before
|
|
85
|
+
// provider I/O. The returned settlement closes that exact identity.
|
|
86
|
+
export type ProviderRequestObserver = (
|
|
87
|
+
identity: ProviderRequestIdentity,
|
|
88
|
+
) => Promise<ProviderRequestSettlement>;
|
|
67
89
|
|
|
68
90
|
// A successful exchange's closed finish set. ProviderAttemptFinishReason adds
|
|
69
91
|
// the failed disposition that may occur only on ProviderError attempt evidence.
|
|
@@ -102,7 +124,6 @@ export interface ProviderAssistant<TFinish extends ProviderAttemptFinishReason =
|
|
|
102
124
|
readonly reasoning: string | null;
|
|
103
125
|
// Encrypted reasoning remains distinct from readable `reasoning`.
|
|
104
126
|
readonly reasoningEncrypted?: ReadonlyArray<ProviderEncryptedReasoningItem>;
|
|
105
|
-
readonly usage: ProviderUsage;
|
|
106
127
|
readonly finishReason: TFinish;
|
|
107
128
|
readonly model: string;
|
|
108
129
|
// Per-token logprobs, present only when PLURNK_PROVIDERS_TOP_LOGPROBS is set
|
|
@@ -114,8 +135,9 @@ export interface ProviderAssistant<TFinish extends ProviderAttemptFinishReason =
|
|
|
114
135
|
}
|
|
115
136
|
|
|
116
137
|
export interface GrammarEvidence {
|
|
117
|
-
// Exact
|
|
118
|
-
//
|
|
138
|
+
// Exact pre-projection response represented by the provider. A generated
|
|
139
|
+
// rail's response root composes any template prefix before @plurnk/gbnf
|
|
140
|
+
// grades it. Offsets are Unicode code points, matching validator verdicts.
|
|
119
141
|
readonly input: string;
|
|
120
142
|
readonly contentStart: number;
|
|
121
143
|
readonly transported: boolean;
|
|
@@ -124,10 +146,9 @@ export interface GrammarEvidence {
|
|
|
124
146
|
export interface ProviderResponse<TFinish extends ProviderAttemptFinishReason = FinishReason> {
|
|
125
147
|
readonly assistant: ProviderAssistant<TFinish>;
|
|
126
148
|
readonly assistantRaw: unknown;
|
|
127
|
-
//
|
|
128
|
-
//
|
|
129
|
-
|
|
130
|
-
readonly charge?: AuthoritativeCharge;
|
|
149
|
+
// Ordered physical request evidence, including automatic retries and pool
|
|
150
|
+
// failover that preceded this response. {§provider-request-accounting}
|
|
151
|
+
readonly accounting: readonly ProviderRequestAccounting[];
|
|
131
152
|
// {§gbnf-response-observation} — evidence only; the consumer owns the verdict.
|
|
132
153
|
readonly grammarEvidence?: GrammarEvidence;
|
|
133
154
|
// Per-turn provider→client metadata bag: the backend's non-standard top-level
|
|
@@ -152,12 +173,30 @@ export interface ProviderResponse<TFinish extends ProviderAttemptFinishReason =
|
|
|
152
173
|
|
|
153
174
|
export type ProviderAttempt = ProviderResponse<ProviderAttemptFinishReason>;
|
|
154
175
|
|
|
176
|
+
export interface ProviderGenerateArgs {
|
|
177
|
+
readonly messages: ChatMessage[];
|
|
178
|
+
readonly workerId: string;
|
|
179
|
+
readonly primaryWorkerId?: string;
|
|
180
|
+
readonly signal?: AbortSignal;
|
|
181
|
+
readonly grammar?: string;
|
|
182
|
+
readonly maxTokens?: number;
|
|
183
|
+
readonly attributions?: string[];
|
|
184
|
+
readonly client?: string;
|
|
185
|
+
readonly strikes?: number;
|
|
186
|
+
readonly workspaceId?: string;
|
|
187
|
+
readonly loop?: number;
|
|
188
|
+
readonly turn?: number;
|
|
189
|
+
readonly sampling?: Record<string, unknown>;
|
|
190
|
+
readonly observeRequest?: ProviderRequestObserver;
|
|
191
|
+
readonly callKind?: ProviderCallKind;
|
|
192
|
+
}
|
|
193
|
+
|
|
155
194
|
export interface Provider {
|
|
156
195
|
// Optional package-authored folksonomy evaluated by the consumer immediately
|
|
157
196
|
// before a provider emission attempt ({§plugin-attribution}).
|
|
158
197
|
attributions?(context: PluginAttributionContext): PluginAttribution;
|
|
159
198
|
// `grammar` is an optional GBNF string (canonically @plurnk/plurnk-contracts'
|
|
160
|
-
// plurnk.gbnf, possibly root-substituted by the consumer). Backends that
|
|
199
|
+
// plurnk.gemma.gbnf or plurnk.qwen.gbnf, possibly root-substituted by the consumer). Backends that
|
|
161
200
|
// support grammar-constrained sampling attach it verbatim; all others
|
|
162
201
|
// ignore it. The provider never chooses or modifies the grammar — whether
|
|
163
202
|
// to constrain and which root variant to send is consumer policy
|
|
@@ -206,9 +245,15 @@ export interface Provider {
|
|
|
206
245
|
// `Plurnk-Turn` ONLY under the same firstPartyMetadata gate; dropped
|
|
207
246
|
// everywhere else. Coordinates are 1-based: absent/0 emits no header (no
|
|
208
247
|
// strikes-style zero exception). Headers only, never the packet.
|
|
209
|
-
|
|
210
|
-
//
|
|
211
|
-
//
|
|
248
|
+
//
|
|
249
|
+
// `callKind` is the caller's explicit output contract. It is transported as
|
|
250
|
+
// `Plurnk-Call-Kind` only under the first-party metadata gate and never
|
|
251
|
+
// inferred from the request shape. Generic callers may omit it; Core always
|
|
252
|
+
// supplies `emission` or `bare`.
|
|
253
|
+
generate(args: ProviderGenerateArgs): Promise<ProviderResponse>;
|
|
254
|
+
// {§model-fact-resolution} — effective total context envelope in tokens,
|
|
255
|
+
// including any stricter operator cap. `null` means unknown; under
|
|
256
|
+
// llama-server parallelism the probed natural value is per slot.
|
|
212
257
|
readonly contextWindow: number | null;
|
|
213
258
|
readonly model: string;
|
|
214
259
|
// Optional: the backend's self-reported served model id, from a
|
|
@@ -231,7 +276,7 @@ export interface Provider {
|
|
|
231
276
|
// with no declared envelope, instead of dying mid-turn in partition math.
|
|
232
277
|
readonly requiresMaxTokens?: boolean;
|
|
233
278
|
// Optional generation-envelope reserves ({§provider-generation-envelope}) — the amounts of
|
|
234
|
-
// the
|
|
279
|
+
// the effective window reserved for reasoning and completion: floor
|
|
235
280
|
// percentages of `contextWindow`, or absolute per-alias pins that win
|
|
236
281
|
// outright. The consumer's prompt budget is `contextWindow - reasoningReserve
|
|
237
282
|
// - completionReserve - <its own packing-safety margin>`; the generation cap
|
|
@@ -242,7 +287,7 @@ export interface Provider {
|
|
|
242
287
|
readonly completionReserve?: number | null;
|
|
243
288
|
// Provider-owned preflight measurement of the complete chat request,
|
|
244
289
|
// including provider/template framing when the adapter can know it.
|
|
245
|
-
// Estimates are explicit and MUST NOT authorize hard
|
|
290
|
+
// Estimates are explicit and MUST NOT authorize hard context-envelope admission.
|
|
246
291
|
countPromptTokens(messages: readonly ChatMessage[], signal?: AbortSignal): Promise<PromptTokenMeasurement>;
|
|
247
292
|
// OPTIONAL capability: exact tokenization served by the backend's own vocab
|
|
248
293
|
// (llama-server /tokenize) — token ids in the model's real vocabulary.
|
|
@@ -250,12 +295,6 @@ export interface Provider {
|
|
|
250
295
|
// `tokenize === undefined` means the backend can't. Exact-counting
|
|
251
296
|
// consumers (the tokenizer seam) prefer this over any client-side data.
|
|
252
297
|
tokenize?(text: string): Promise<number[]>;
|
|
253
|
-
// {§model-fact-resolution} — frozen 1.x local USD estimate compatibility.
|
|
254
|
-
// This is not a monetary-reporting authority.
|
|
255
|
-
calculateCost(usage: ProviderUsage): number;
|
|
256
|
-
// Models.dev-derived monetary fallback. Direct response charges travel on
|
|
257
|
-
// ProviderResponse and take precedence at the consuming attempt boundary.
|
|
258
|
-
calculateCharge?(usage: ProviderUsage): Exclude<ProviderCost, AuthoritativeCharge>;
|
|
259
298
|
}
|
|
260
299
|
|
|
261
300
|
// ProviderAlias lives in @plurnk/plurnk-aliases (the zero-dependency parser);
|
package/src/usage.test.ts
CHANGED
|
@@ -1,153 +1,149 @@
|
|
|
1
1
|
import test from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
|
-
import {
|
|
3
|
+
import { calculateCostUsdDecimal, normalizeUsage } from "./usage.ts";
|
|
4
4
|
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
assert.
|
|
5
|
+
test("normalizeUsage recovers additive hidden reasoning from an exact total", () => {
|
|
6
|
+
const usage = normalizeUsage({
|
|
7
|
+
prompt_tokens: 19,
|
|
8
|
+
completion_tokens: 285,
|
|
9
|
+
total_tokens: 1165,
|
|
10
|
+
});
|
|
11
|
+
assert.deepEqual(usage, {
|
|
12
|
+
inputTokens: 19,
|
|
13
|
+
outputTokens: 1146,
|
|
14
|
+
totalTokens: 1165,
|
|
15
|
+
outputTokenDetails: { textTokens: 285, reasoningTokens: 861 },
|
|
16
|
+
});
|
|
12
17
|
});
|
|
13
18
|
|
|
14
|
-
test("normalizeUsage
|
|
15
|
-
|
|
16
|
-
const u = normalizeUsage({
|
|
19
|
+
test("normalizeUsage preserves OpenAI output totals while itemizing reasoning", () => {
|
|
20
|
+
const usage = normalizeUsage({
|
|
17
21
|
prompt_tokens: 10,
|
|
18
22
|
completion_tokens: 100,
|
|
19
23
|
total_tokens: 110,
|
|
20
24
|
completion_tokens_details: { reasoning_tokens: 40 },
|
|
21
25
|
});
|
|
22
|
-
assert.deepEqual(
|
|
23
|
-
|
|
26
|
+
assert.deepEqual(usage, {
|
|
27
|
+
inputTokens: 10,
|
|
28
|
+
outputTokens: 100,
|
|
29
|
+
totalTokens: 110,
|
|
30
|
+
outputTokenDetails: { textTokens: 60, reasoningTokens: 40 },
|
|
31
|
+
});
|
|
24
32
|
});
|
|
25
33
|
|
|
26
|
-
test("normalizeUsage
|
|
27
|
-
|
|
28
|
-
// detailed but ADDITIVE — total = prompt + completion + reasoning.
|
|
29
|
-
const u = normalizeUsage({
|
|
34
|
+
test("normalizeUsage recognizes xAI reasoning as additive from the total identity", () => {
|
|
35
|
+
const usage = normalizeUsage({
|
|
30
36
|
prompt_tokens: 143,
|
|
31
37
|
completion_tokens: 1,
|
|
32
38
|
total_tokens: 441,
|
|
33
39
|
prompt_tokens_details: { cached_tokens: 128 },
|
|
34
40
|
completion_tokens_details: { reasoning_tokens: 297 },
|
|
35
41
|
});
|
|
36
|
-
assert.deepEqual(
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
assert.equal(u.cached, 30);
|
|
42
|
+
assert.deepEqual(usage, {
|
|
43
|
+
inputTokens: 143,
|
|
44
|
+
outputTokens: 298,
|
|
45
|
+
totalTokens: 441,
|
|
46
|
+
inputTokenDetails: { cacheReadTokens: 128 },
|
|
47
|
+
outputTokenDetails: { textTokens: 1, reasoningTokens: 297 },
|
|
48
|
+
});
|
|
44
49
|
});
|
|
45
50
|
|
|
46
|
-
test("normalizeUsage
|
|
47
|
-
|
|
48
|
-
|
|
51
|
+
test("normalizeUsage maps cache-read spellings without inventing uncached tokens", () => {
|
|
52
|
+
assert.deepEqual(normalizeUsage({
|
|
53
|
+
prompt_tokens: 50,
|
|
54
|
+
completion_tokens: 10,
|
|
55
|
+
total_tokens: 60,
|
|
56
|
+
prompt_tokens_details: { cached_tokens: 30 },
|
|
57
|
+
})?.inputTokenDetails, { cacheReadTokens: 30 });
|
|
58
|
+
assert.deepEqual(normalizeUsage({
|
|
59
|
+
prompt_tokens: 50,
|
|
60
|
+
completion_tokens: 10,
|
|
61
|
+
total_tokens: 60,
|
|
62
|
+
cached_tokens: 12,
|
|
63
|
+
})?.inputTokenDetails, { cacheReadTokens: 12 });
|
|
49
64
|
});
|
|
50
65
|
|
|
51
|
-
test("#157: normalizeUsage maps DeepSeek
|
|
52
|
-
|
|
66
|
+
test("#157: normalizeUsage maps DeepSeek cache hit and miss counts", () => {
|
|
67
|
+
assert.deepEqual(normalizeUsage({
|
|
53
68
|
prompt_tokens: 50,
|
|
54
69
|
prompt_cache_hit_tokens: 30,
|
|
55
70
|
prompt_cache_miss_tokens: 20,
|
|
56
71
|
completion_tokens: 10,
|
|
57
72
|
total_tokens: 60,
|
|
73
|
+
}), {
|
|
74
|
+
inputTokens: 50,
|
|
75
|
+
outputTokens: 10,
|
|
76
|
+
totalTokens: 60,
|
|
77
|
+
inputTokenDetails: { noCacheTokens: 20, cacheReadTokens: 30 },
|
|
58
78
|
});
|
|
59
|
-
assert.deepEqual(u, { prompt: 50, completion: 10, reasoning: 0, cached: 30, total: 60 });
|
|
60
|
-
});
|
|
61
|
-
|
|
62
|
-
test("normalizeUsage: no reasoning — plain prompt+completion", () => {
|
|
63
|
-
const u = normalizeUsage({ prompt_tokens: 10, completion_tokens: 20, total_tokens: 30 });
|
|
64
|
-
assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
|
|
65
|
-
});
|
|
66
|
-
|
|
67
|
-
test("normalizeUsage: missing total is reconstructed, never negative reasoning", () => {
|
|
68
|
-
const u = normalizeUsage({ prompt_tokens: 10, completion_tokens: 20 });
|
|
69
|
-
assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
|
|
70
79
|
});
|
|
71
80
|
|
|
72
|
-
test("normalizeUsage
|
|
73
|
-
assert.deepEqual(normalizeUsage(
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
{ prompt_tokens: 100, completion_tokens: 1000, total_tokens: 1100 },
|
|
84
|
-
"r".repeat(750),
|
|
85
|
-
"c".repeat(250),
|
|
86
|
-
);
|
|
87
|
-
assert.deepEqual(u, { prompt: 100, completion: 250, reasoning: 750, cached: 0, total: 1100 });
|
|
88
|
-
assert.equal(u.completion + u.reasoning, 1000); // billable output byte-identical
|
|
89
|
-
assert.equal(u.prompt + u.completion + u.reasoning, u.total); // invariant
|
|
90
|
-
});
|
|
91
|
-
|
|
92
|
-
test("normalizeUsage: pure-reasoning turn (empty content) attributes all completion to reasoning", () => {
|
|
93
|
-
// The run52 runaway shape: 0 visible content, the whole budget spent reasoning.
|
|
94
|
-
const u = normalizeUsage(
|
|
95
|
-
{ prompt_tokens: 100, completion_tokens: 500, total_tokens: 600 },
|
|
96
|
-
"t".repeat(9000),
|
|
97
|
-
"",
|
|
98
|
-
);
|
|
99
|
-
assert.deepEqual(u, { prompt: 100, completion: 0, reasoning: 500, cached: 0, total: 600 });
|
|
100
|
-
});
|
|
101
|
-
|
|
102
|
-
test("normalizeUsage: text args never perturb the itemized (reasoning_tokens) path", () => {
|
|
103
|
-
// OpenAI o-series reports reasoning_tokens -> that split wins, text is ignored.
|
|
104
|
-
const u = normalizeUsage(
|
|
105
|
-
{ prompt_tokens: 10, completion_tokens: 100, total_tokens: 110, completion_tokens_details: { reasoning_tokens: 40 } },
|
|
106
|
-
"r".repeat(999), "c".repeat(1),
|
|
107
|
-
);
|
|
108
|
-
assert.deepEqual(u, { prompt: 10, completion: 60, reasoning: 40, cached: 0, total: 110 });
|
|
109
|
-
});
|
|
110
|
-
|
|
111
|
-
test("normalizeUsage: text args never perturb the Gemini gap path (gap already yields reasoning)", () => {
|
|
112
|
-
// A real total gap means reasoning is itemized-by-subtraction; do not re-split.
|
|
113
|
-
const u = normalizeUsage(
|
|
114
|
-
{ prompt_tokens: 19, completion_tokens: 285, total_tokens: 1165 },
|
|
115
|
-
"r".repeat(500), "c".repeat(500),
|
|
116
|
-
);
|
|
117
|
-
assert.equal(u.reasoning, 861); // from the gap, NOT a text re-split
|
|
118
|
-
assert.equal(u.completion, 285);
|
|
119
|
-
});
|
|
120
|
-
|
|
121
|
-
test("normalizeUsage: no total reported -> re-split skipped, reasoning stays 0 (cannot split an unknown base)", () => {
|
|
122
|
-
const u = normalizeUsage(
|
|
123
|
-
{ prompt_tokens: 10, completion_tokens: 20 },
|
|
124
|
-
"r".repeat(500), "c".repeat(500),
|
|
125
|
-
);
|
|
126
|
-
assert.deepEqual(u, { prompt: 10, completion: 20, reasoning: 0, cached: 0, total: 30 });
|
|
127
|
-
});
|
|
128
|
-
|
|
129
|
-
// — calculateCostUsd —
|
|
130
|
-
|
|
131
|
-
test("calculateCostUsd: bills reasoning at the USD-per-million output rate", () => {
|
|
132
|
-
// 100 input, 0 cached, 50 completion + 200 reasoning = 250 output.
|
|
133
|
-
const usage = { prompt: 100, completion: 50, reasoning: 200, cached: 0, total: 350 };
|
|
134
|
-
assert.equal(calculateCostUsd(usage, { input: 1, output: 10, cached: 0 }), 0.0026);
|
|
81
|
+
test("normalizeUsage derives only exact totals", () => {
|
|
82
|
+
assert.deepEqual(normalizeUsage({ prompt_tokens: 10, completion_tokens: 20 }), {
|
|
83
|
+
inputTokens: 10,
|
|
84
|
+
outputTokens: 20,
|
|
85
|
+
totalTokens: 30,
|
|
86
|
+
});
|
|
87
|
+
assert.deepEqual(normalizeUsage({ prompt_tokens: 10, total_tokens: 30 }), {
|
|
88
|
+
inputTokens: 10,
|
|
89
|
+
outputTokens: 20,
|
|
90
|
+
totalTokens: 30,
|
|
91
|
+
});
|
|
135
92
|
});
|
|
136
93
|
|
|
137
|
-
test("
|
|
138
|
-
|
|
139
|
-
assert.equal(
|
|
94
|
+
test("normalizeUsage preserves unknown usage as absence", () => {
|
|
95
|
+
assert.equal(normalizeUsage(null), undefined);
|
|
96
|
+
assert.equal(normalizeUsage(undefined), undefined);
|
|
140
97
|
});
|
|
141
98
|
|
|
142
|
-
test("
|
|
143
|
-
|
|
144
|
-
|
|
99
|
+
test("normalizeUsage never apportions tokens from reasoning or content length", () => {
|
|
100
|
+
assert.deepEqual(normalizeUsage({
|
|
101
|
+
prompt_tokens: 100,
|
|
102
|
+
completion_tokens: 1000,
|
|
103
|
+
total_tokens: 1100,
|
|
104
|
+
}), {
|
|
105
|
+
inputTokens: 100,
|
|
106
|
+
outputTokens: 1000,
|
|
107
|
+
totalTokens: 1100,
|
|
108
|
+
});
|
|
145
109
|
});
|
|
146
110
|
|
|
147
|
-
test("calculateCostUsdDecimal
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
111
|
+
test("calculateCostUsdDecimal bills all output, including reasoning, at the output rate", () => {
|
|
112
|
+
assert.equal(calculateCostUsdDecimal({
|
|
113
|
+
inputTokens: 100,
|
|
114
|
+
outputTokens: 250,
|
|
115
|
+
totalTokens: 350,
|
|
116
|
+
outputTokenDetails: { textTokens: 50, reasoningTokens: 200 },
|
|
117
|
+
}, { input: 1, output: 10 }), "0.0026");
|
|
118
|
+
});
|
|
119
|
+
|
|
120
|
+
test("calculateCostUsdDecimal applies distinct cache-read and cache-write rates", () => {
|
|
121
|
+
assert.equal(calculateCostUsdDecimal({
|
|
122
|
+
inputTokens: 1000,
|
|
123
|
+
outputTokens: 0,
|
|
124
|
+
totalTokens: 1000,
|
|
125
|
+
inputTokenDetails: {
|
|
126
|
+
noCacheTokens: 500,
|
|
127
|
+
cacheReadTokens: 400,
|
|
128
|
+
cacheWriteTokens: 100,
|
|
129
|
+
},
|
|
130
|
+
}, { input: 5, output: 99, cacheRead: 1, cacheWrite: 8 }), "0.0037");
|
|
131
|
+
});
|
|
132
|
+
|
|
133
|
+
test("calculateCostUsdDecimal returns unknown when a differently-priced category is absent", () => {
|
|
134
|
+
assert.equal(calculateCostUsdDecimal({
|
|
135
|
+
inputTokens: 1000,
|
|
136
|
+
outputTokens: 0,
|
|
137
|
+
totalTokens: 1000,
|
|
138
|
+
}, { input: 5, output: 99, cacheRead: 1 }), null);
|
|
139
|
+
});
|
|
140
|
+
|
|
141
|
+
test("calculateCostUsdDecimal preserves Models.dev decimals without floating-point artifacts", () => {
|
|
142
|
+
assert.equal(calculateCostUsdDecimal({
|
|
143
|
+
inputTokens: 1000,
|
|
144
|
+
outputTokens: 150,
|
|
145
|
+
totalTokens: 1150,
|
|
146
|
+
inputTokenDetails: { cacheReadTokens: 400 },
|
|
147
|
+
outputTokenDetails: { textTokens: 100, reasoningTokens: 50 },
|
|
148
|
+
}, { input: 0.14, output: 0.28, cacheRead: 0.0028 }), "0.00012712");
|
|
153
149
|
});
|