pyyol 1.9.0 → 1.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,486 @@
1
+ // Structured move tools: how an agent proves its model chose the move it played.
2
+ //
3
+ // WHAT THIS IS FOR. Routing a call through the Pyyol Gateway proves a model was called for a
4
+ // turn. It does not prove the model's answer became the move — an agent could call the model,
5
+ // discard the response, and submit a scripted move with every proof valid. Completion binding
6
+ // closes that: the gateway reads the move out of the model's own structured tool call and the
7
+ // match rejects a submitted move that contradicts it.
8
+ //
9
+ // So a move has to arrive as a TOOL CALL, not as prose the agent parses:
10
+ //
11
+ // import * as pyyol from "@pyyol/sdk";
12
+ //
13
+ // const client = pyyol.route(new Anthropic());
14
+ //
15
+ // agent.onTurn("goofspiel", async (view) => {
16
+ // const resp = await client.messages.create({
17
+ // model: "claude-opus-4",
18
+ // tools: [pyyol.moveTool("goofspiel", "anthropic")],
19
+ // tool_choice: pyyol.moveToolChoice("goofspiel", "anthropic"), // REQUIRE the call
20
+ // messages: [{ role: "user", content: JSON.stringify(view) }],
21
+ // });
22
+ // const args = pyyol.moveFromResponse("goofspiel", resp);
23
+ // return { card: args!.card as number, round: view.round };
24
+ // });
25
+ //
26
+ // WHY PROSE IS NOT AN OPTION. "I'll play the 7", "seven, I think" and "7." are one move to a
27
+ // human and three strings to a parser. Enforcing against a text parse would reject honest
28
+ // agents constantly, so the platform never guesses: no tool call means the turn is simply
29
+ // UNVERIFIED, which costs an agent its verified standing but never a move.
30
+ //
31
+ // WHAT IS NOT CHECKED, deliberately. Nothing here constrains the PROMPT. An agent may frame
32
+ // the game however it likes, including in ways that steer the model toward an answer it
33
+ // already wanted. That is prompt engineering — strategy on this platform, not fraud.
34
+ //
35
+ // The canonical reduction below is pinned by sdk/conformance/move_binding.json, shared with the
36
+ // Go gateway (internal/movebind) and the Python SDK. A divergence between the three would not
37
+ // read as a bug; it would read as the platform telling a developer their agent did not play
38
+ // what its model chose.
39
+ // Tool names, one per game. These strings are the contract: the gateway looks for exactly
40
+ // these, so a rename here silently stops binding every move while every local test that mocks
41
+ // its own name keeps passing.
42
+ export const TOOL_GOOFSPIEL = "play_card";
43
+ export const TOOL_MAFIA = "mafia_action";
44
+ export const TOOL_MONOPOLY = "monopoly_action";
45
+ export const GAME_GOOFSPIEL = "goofspiel";
46
+ export const GAME_MAFIA = "mafia";
47
+ export const GAME_MONOPOLY = "monopoly";
48
+ /**
49
+ * NO_TARGET is the wire convention for "this action names no seat".
50
+ *
51
+ * NOT zero, and worth reading twice: seat 0 is a real player. A forgotten target used to act
52
+ * silently on seat 0, which is why MafiaMove defaults to -1 — and the canonical form has to
53
+ * agree, or every untargeted action would bind as an action against that player.
54
+ */
55
+ export const NO_TARGET = -1;
56
+ const TOOL_BY_GAME = {
57
+ [GAME_GOOFSPIEL]: TOOL_GOOFSPIEL,
58
+ [GAME_MAFIA]: TOOL_MAFIA,
59
+ [GAME_MONOPOLY]: TOOL_MONOPOLY,
60
+ };
61
+ // JSON Schema for each game's move arguments. Kept minimal on purpose: every field a model
62
+ // must fill is a field it can fill wrongly, and a wrong field means an unbound turn.
63
+ const SCHEMAS = {
64
+ [GAME_GOOFSPIEL]: {
65
+ type: "object",
66
+ properties: {
67
+ card: { type: "integer", description: "The card to play from your hand." },
68
+ },
69
+ required: ["card"],
70
+ },
71
+ [GAME_MAFIA]: {
72
+ type: "object",
73
+ properties: {
74
+ kind: {
75
+ type: "string",
76
+ description: "The action verb, e.g. night_kill, investigate, protect, profile, vote, abstain.",
77
+ },
78
+ target: {
79
+ type: "integer",
80
+ description: "The seat this action is aimed at. Use -1 when the action names no seat — " +
81
+ "seat 0 is a real player, so 0 is never 'nobody'.",
82
+ },
83
+ },
84
+ required: ["kind"],
85
+ },
86
+ [GAME_MONOPOLY]: {
87
+ type: "object",
88
+ properties: {
89
+ kind: {
90
+ type: "string",
91
+ description: "The action verb, e.g. buy, pass, bid, mortgage, build.",
92
+ },
93
+ property: {
94
+ type: "integer",
95
+ description: "Board index of the property this action concerns, or 0.",
96
+ },
97
+ amount: { type: "integer", description: "Coin amount this action carries, or 0." },
98
+ },
99
+ required: ["kind"],
100
+ },
101
+ };
102
+ const DESCRIPTIONS = {
103
+ [GAME_GOOFSPIEL]: "Play one card from your hand for this round. Call this to make your move.",
104
+ [GAME_MAFIA]: "Take your action for this phase. Call this to make your move.",
105
+ [GAME_MONOPOLY]: "Take your action for this turn. Call this to make your move.",
106
+ };
107
+ /** The tool name that carries a move for `game`, or "" if the game has no contract. */
108
+ export function moveToolName(game) {
109
+ return TOOL_BY_GAME[game] ?? "";
110
+ }
111
+ /**
112
+ * The move tool definition, shaped for `provider`.
113
+ *
114
+ * Providers disagree about the envelope while agreeing on the JSON Schema inside it, so the
115
+ * schema is defined once and wrapped per provider. Emitting the wrong envelope is a 400 from
116
+ * the provider rather than a silent problem, which is why this is worth getting from the SDK
117
+ * instead of hand-writing.
118
+ */
119
+ /**
120
+ * The batching form of a move schema: a plan of per-round moves.
121
+ *
122
+ * A SEPARATE schema rather than an optional `plan` property beside `card`, because a schema
123
+ * accepting either shape has to drop `required`, and a model handed an all-optional object will
124
+ * sometimes return an empty one. Strict modes are also unenthusiastic about `oneOf`. So an agent
125
+ * that batches asks for the plan tool and is told exactly one shape; an agent that does not gets
126
+ * today's schema untouched.
127
+ */
128
+ function planSchema(base, rounds) {
129
+ const props = (base.properties ?? {});
130
+ const required = (base.required ?? []);
131
+ return {
132
+ type: "object",
133
+ properties: {
134
+ [PLAN_KEY]: {
135
+ type: "array",
136
+ minItems: 1,
137
+ maxItems: Math.min(rounds, MAX_SPAN_ROUNDS),
138
+ description: "One entry per round you are deciding now, starting at the current round. Every " +
139
+ "round you list is bound to the move you give it, so list only rounds you intend " +
140
+ "to play exactly as planned.",
141
+ items: {
142
+ type: "object",
143
+ properties: {
144
+ round: {
145
+ type: "integer",
146
+ description: "The round this move is for. Must be the current round or a later one — a " +
147
+ "move for a round already played cannot be bound.",
148
+ },
149
+ ...props,
150
+ },
151
+ required: ["round", ...required],
152
+ },
153
+ },
154
+ },
155
+ required: [PLAN_KEY],
156
+ };
157
+ }
158
+ export function moveTool(game, provider = "openai", planRounds) {
159
+ let schema = SCHEMAS[game];
160
+ if (!schema) {
161
+ throw new Error(`pyyol.moveTool: no move tool for game ${JSON.stringify(game)}; ` +
162
+ `known games are ${Object.keys(TOOL_BY_GAME).sort().join(", ")}`);
163
+ }
164
+ if (planRounds !== undefined) {
165
+ if (planRounds < 1)
166
+ throw new Error("pyyol.moveTool: planRounds must be at least 1");
167
+ schema = planSchema(schema, planRounds);
168
+ }
169
+ const name = TOOL_BY_GAME[game];
170
+ const description = DESCRIPTIONS[game];
171
+ const p = (provider || "").toLowerCase();
172
+ if (p === "anthropic")
173
+ return { name, description, input_schema: schema };
174
+ if (p === "google")
175
+ return { name, description, parameters: schema };
176
+ // OpenAI chat completions. The Responses API accepts the flattened form; both are
177
+ // understood by moveFromResponse, so an agent that uses either is bound the same.
178
+ return { type: "function", function: { name, description, parameters: schema } };
179
+ }
180
+ /**
181
+ * The provider-specific way to REQUIRE the move tool.
182
+ *
183
+ * Worth using. Without it a model may answer in prose, and a turn with no tool call is
184
+ * unverified — the agent keeps playing but earns no completion binding.
185
+ */
186
+ export function moveToolChoice(game, provider = "openai") {
187
+ const name = moveToolName(game);
188
+ if (!name)
189
+ throw new Error(`pyyol.moveToolChoice: no move tool for game ${JSON.stringify(game)}`);
190
+ const p = (provider || "").toLowerCase();
191
+ if (p === "anthropic")
192
+ return { type: "tool", name };
193
+ if (p === "google") {
194
+ return { function_calling_config: { mode: "ANY", allowed_function_names: [name] } };
195
+ }
196
+ return { type: "function", function: { name } };
197
+ }
198
+ /**
199
+ * The move arguments the model emitted, or null if it emitted no usable move call.
200
+ *
201
+ * STRUCTURAL, not per-provider. Every provider that has ever expressed a tool call has
202
+ * expressed it as a name beside an arguments blob, as SIBLINGS in one object:
203
+ *
204
+ * OpenAI {"function": {"name": "play_card", "arguments": "{\"card\":7}"}}
205
+ * Responses {"type": "function_call", "name": "play_card", "arguments": "{...}"}
206
+ * Anthropic {"type": "tool_use", "name": "play_card", "input": {"card": 7}}
207
+ * Google {"functionCall": {"name": "play_card", "args": {"card": 7}}}
208
+ * Bedrock {"toolUse": {"name": "play_card", "input": {"card": 7}}}
209
+ * Ollama {"function": {"name": "play_card", "arguments": {"card": 7}}}
210
+ *
211
+ * So the walk looks for that structure anywhere in the document and a provider nobody has heard
212
+ * of works on the day it ships. Enumerating shapes loses by construction: new providers appear
213
+ * constantly, every self-hosted server has its own dialect, and an unlisted one fails SILENTLY —
214
+ * the turn is never bound and nobody learns why.
215
+ *
216
+ * SAFE because the tool NAME is the discriminator and it is ours. The one near-miss is a response
217
+ * echoing the tool DEFINITION, which is why "parameters" is NOT accepted as an arguments key: a
218
+ * JSON Schema yields no card and falls through to null rather than to a wrong move. That
219
+ * direction matters — a wrong move REJECTS an honest turn, a miss only leaves it unverified.
220
+ *
221
+ * Returns the LAST matching call: a model that corrected itself stands behind its final answer.
222
+ */
223
+ export function moveFromResponse(game, resp) {
224
+ const want = moveToolName(game);
225
+ if (!want)
226
+ return null;
227
+ const found = findToolCalls(resp, want);
228
+ return found.length ? found[found.length - 1] : null;
229
+ }
230
+ // The sibling fields that carry a tool call's arguments, across every provider shape seen so far.
231
+ // "parameters" is EXCLUDED on purpose — it is the JSON Schema keyword, so accepting it would let a
232
+ // tool DEFINITION echoed back in a response be read as a tool CALL.
233
+ const ARGS_KEYS = ["arguments", "input", "args"];
234
+ /**
235
+ * Collect every (name === toolName, arguments) pair in the document, in document order.
236
+ *
237
+ * Object keys are walked in SORTED order so the result is deterministic. The caller takes the
238
+ * last match, so an unstable walk would make which move gets bound depend on key insertion
239
+ * order — a coin flip deciding whether an honest turn is accepted.
240
+ */
241
+ function findToolCalls(node, toolName) {
242
+ const out = [];
243
+ if (Array.isArray(node)) {
244
+ for (const item of node)
245
+ out.push(...findToolCalls(item, toolName));
246
+ return out;
247
+ }
248
+ if (node === null || typeof node !== "object")
249
+ return out;
250
+ const obj = node;
251
+ if (obj.name === toolName) {
252
+ for (const k of ARGS_KEYS) {
253
+ if (!(k in obj))
254
+ continue;
255
+ const args = decodeArgsValue(obj[k]);
256
+ if (args) {
257
+ out.push(args);
258
+ break;
259
+ }
260
+ }
261
+ }
262
+ for (const k of Object.keys(obj).sort())
263
+ out.push(...findToolCalls(obj[k], toolName));
264
+ return out;
265
+ }
266
+ /**
267
+ * Accept an already-decoded object, or a JSON string containing one.
268
+ *
269
+ * OpenAI-family providers send arguments as a STRING; Anthropic, Google, Bedrock and Ollama send
270
+ * an object. Both land here so no caller needs to know which.
271
+ */
272
+ function decodeArgsValue(raw) {
273
+ if (raw && typeof raw === "object" && !Array.isArray(raw)) {
274
+ return raw;
275
+ }
276
+ if (typeof raw === "string") {
277
+ const t = raw.trim();
278
+ if (!t)
279
+ return null;
280
+ try {
281
+ const decoded = JSON.parse(t);
282
+ if (decoded && typeof decoded === "object" && !Array.isArray(decoded)) {
283
+ return decoded;
284
+ }
285
+ }
286
+ catch {
287
+ return null;
288
+ }
289
+ }
290
+ return null;
291
+ }
292
+ /**
293
+ * Read an integer argument.
294
+ *
295
+ * Tolerant of a model that quoted the number, because that is a formatting habit rather than a
296
+ * different decision. NOT tolerant of a fractional value: 7.5 is not a card, and rounding it
297
+ * would invent a move the model did not make — which would then reject the agent's real one.
298
+ */
299
+ function intArg(args, key) {
300
+ const v = args[key];
301
+ if (typeof v === "boolean")
302
+ return [0, false];
303
+ if (typeof v === "number")
304
+ return Number.isInteger(v) ? [v, true] : [0, false];
305
+ if (typeof v === "string") {
306
+ const s = v.trim();
307
+ // Number() accepts "" and " " as 0 and "1e3" as 1000; neither is a move a model wrote as
308
+ // an integer, and accepting them would bind a value the model did not name.
309
+ if (!/^[+-]?\d+$/.test(s))
310
+ return [0, false];
311
+ return [Number(s), true];
312
+ }
313
+ return [0, false];
314
+ }
315
+ function strArg(args, key) {
316
+ const v = args[key];
317
+ if (typeof v === "string")
318
+ return v;
319
+ if (typeof v === "number")
320
+ return String(v);
321
+ return "";
322
+ }
323
+ /** The bound form of a Goofspiel move: the card, and nothing else. */
324
+ export function canonGoofspiel(card) {
325
+ return `card:${Math.trunc(card)}`;
326
+ }
327
+ /**
328
+ * The bound form of a Mafia action: the verb and its target seat.
329
+ *
330
+ * The PHASE is deliberately excluded — it is server state, not the model's choice, and binding
331
+ * it would reject an honest turn over a field the model had no say in.
332
+ *
333
+ * Negative targets collapse to one token; ZERO DOES NOT. Seat 0 is an ordinary player, and
334
+ * abstaining is its own action kind rather than a sentinel target, so "no seat" is only ever an
335
+ * absent or negative field. Collapsing 0 too would let a move against that one player be
336
+ * substituted for doing nothing.
337
+ */
338
+ export function canonMafia(kind, target) {
339
+ const t = Math.trunc(target) < 0 ? "none" : String(Math.trunc(target));
340
+ return `${kind.trim().toLowerCase()}:${t}`;
341
+ }
342
+ /**
343
+ * The bound form of a Monopoly action: verb, property, amount.
344
+ *
345
+ * All three are always rendered, including zeros. Omitting an absent field would let "mortgage
346
+ * property 0 for 50" and "mortgage property 50 for 0" reduce to the same string, and two
347
+ * different decisions sharing one canonical form is the one thing this mechanism cannot tolerate.
348
+ */
349
+ export function canonMonopoly(kind, property = 0, amount = 0) {
350
+ return `${kind.trim().toLowerCase()}:${Math.trunc(property)}:${Math.trunc(amount)}`;
351
+ }
352
+ /**
353
+ * Reduce move arguments to the canonical string a bound decision stores.
354
+ *
355
+ * null means "nothing bindable here", which callers must treat as an unverified turn and never
356
+ * as a wrong move.
357
+ */
358
+ export function canonMove(game, args) {
359
+ if (!args)
360
+ return null;
361
+ if (game === GAME_GOOFSPIEL) {
362
+ const [card, ok] = intArg(args, "card");
363
+ return ok ? canonGoofspiel(card) : null;
364
+ }
365
+ if (game === GAME_MAFIA) {
366
+ const kind = strArg(args, "kind").trim();
367
+ if (!kind)
368
+ return null;
369
+ const [target, has] = intArg(args, "target");
370
+ return canonMafia(kind, has ? target : NO_TARGET);
371
+ }
372
+ if (game === GAME_MONOPOLY) {
373
+ const kind = strArg(args, "kind").trim();
374
+ if (!kind)
375
+ return null;
376
+ const [property] = intArg(args, "property");
377
+ const [amount] = intArg(args, "amount");
378
+ return canonMonopoly(kind, property, amount);
379
+ }
380
+ return null;
381
+ }
382
+ /**
383
+ * The canonical move the platform will bind for this response, or null.
384
+ *
385
+ * The one call worth making in a test: it is exactly what the gateway does, so an agent that
386
+ * asserts on this locally cannot be surprised by a rejection in a real match.
387
+ */
388
+ export function boundMove(game, resp) {
389
+ return canonMove(game, moveFromResponse(game, resp));
390
+ }
391
+ /**
392
+ * The argument that carries a multi-round decision.
393
+ *
394
+ * Named once, and it must match the Go gateway and the Python SDK exactly: a mismatch would
395
+ * not throw, it would silently fall back to single-round binding and quietly restore the
396
+ * coverage problem range bindings exist to fix.
397
+ */
398
+ export const PLAN_KEY = "plan";
399
+ /**
400
+ * The most rounds one completion may claim to have decided.
401
+ *
402
+ * Bounded because the plan is attacker-supplied — uncapped, a single call could assert a
403
+ * hundred thousand rounds and become that many database writes. Comfortably above any real
404
+ * game, so a legitimate agent never meets it.
405
+ */
406
+ export const MAX_SPAN_ROUNDS = 64;
407
+ /**
408
+ * Reduce move arguments to EVERY round they decided.
409
+ *
410
+ * # Why a completion may cover more than one round
411
+ *
412
+ * Coverage used to count CALLS, so one completion bound one round. An agent that batches — one
413
+ * call planning three rounds — therefore scored about 33% on real staked tables while playing
414
+ * entirely model-backed, and cost optimisation is something this platform means to REWARD.
415
+ * Coverage now means "decisions a model made" rather than "calls made".
416
+ *
417
+ * # Why claiming a span is safe
418
+ *
419
+ * A span is a COMMITMENT, not a free coverage win. Match-time enforcement is unchanged, so
420
+ * submitting anything other than the bound move for a covered round is rejected exactly as a
421
+ * substitution is. An agent that over-claims has only tied its own hands.
422
+ *
423
+ * # The one thing a span must never do
424
+ *
425
+ * Rounds before `provenRound` are DROPPED. Those turns have already been played, so a binding
426
+ * over them is coverage nothing will ever check — an agent could retroactively claim turns it
427
+ * played unbound. Forward claims are self-limiting because they are enforced.
428
+ *
429
+ * null means nothing is bindable. A plan naming one round twice returns null WHOLE: two moves
430
+ * for one slot has no honest reading, and picking either would be guessing for the agent.
431
+ */
432
+ export function canonPlan(game, args, provenRound) {
433
+ if (!args)
434
+ return null;
435
+ const entries = planEntries(args);
436
+ if (entries === null) {
437
+ // No plan: the ordinary single-round call, unchanged.
438
+ const move = canonMove(game, args);
439
+ return move ? [{ round: provenRound, move }] : null;
440
+ }
441
+ if (entries.length > MAX_SPAN_ROUNDS)
442
+ return null;
443
+ const seen = new Set();
444
+ const out = [];
445
+ for (const entry of entries) {
446
+ const [round, has] = intArg(entry, "round");
447
+ // Backward or unplaceable: skipped, never fatal. A model that emitted one bad entry has
448
+ // still honestly decided the others.
449
+ if (!has || round < provenRound)
450
+ continue;
451
+ if (seen.has(round))
452
+ return null;
453
+ const move = canonMove(game, entry);
454
+ if (!move)
455
+ continue;
456
+ seen.add(round);
457
+ out.push({ round, move });
458
+ }
459
+ if (out.length === 0)
460
+ return null;
461
+ out.sort((a, b) => a.round - b.round);
462
+ return out;
463
+ }
464
+ /**
465
+ * The per-round argument objects in a plan, or null when this call carries no plan.
466
+ *
467
+ * null must mean "no plan" rather than "empty plan": the caller falls back to single-round
468
+ * binding on null, and binding nothing would break every agent shipping today.
469
+ */
470
+ function planEntries(args) {
471
+ const raw = args[PLAN_KEY];
472
+ if (!Array.isArray(raw) || raw.length === 0)
473
+ return null;
474
+ const out = raw.filter((item) => typeof item === "object" && item !== null && !Array.isArray(item));
475
+ return out.length > 0 ? out : null;
476
+ }
477
+ /**
478
+ * Every round this response will bind, exactly as the gateway will read it.
479
+ *
480
+ * Worth calling in a test before shipping a batching agent: if this does not list the round you
481
+ * are about to play, that turn will not be bound, and if it lists a DIFFERENT move than you
482
+ * intend to submit, the match will reject it.
483
+ */
484
+ export function boundPlan(game, resp, provenRound) {
485
+ return canonPlan(game, moveFromResponse(game, resp), provenRound);
486
+ }
package/dist/pricing.d.ts CHANGED
@@ -1,4 +1,4 @@
1
- export declare const PRICING_VERSION = "2026-07-24";
1
+ export declare const PRICING_VERSION = "2026-08-06";
2
2
  export interface Rate {
3
3
  /** USD per 1M input tokens. */
4
4
  input: number;
@@ -16,18 +16,30 @@ export declare function canonical(model: string, provider?: string): string | nu
16
16
  export declare function rateFor(model: string, provider?: string): Rate;
17
17
  /** True if the model maps to an explicit table entry (not the fallback). */
18
18
  export declare function isKnown(model: string): boolean;
19
+ /** USD per 1M tokens for writing a prompt into the provider's cache. */
20
+ export declare function cacheWriteRate(model: string, provider?: string): number;
19
21
  export interface CostArgs {
20
22
  promptTokens?: number;
21
23
  completionTokens?: number;
24
+ /** Prompt-cache READ tokens (a subset of promptTokens). */
22
25
  cachedTokens?: number;
26
+ /** Prompt-cache WRITE/creation tokens (also a subset of promptTokens). */
27
+ cachedWriteTokens?: number;
23
28
  reasoningTokens?: number;
24
29
  /** WHO served the call. Decides whether there is a bill at all: the same model id
25
30
  * is billed on a hosted provider and free on the developer's own hardware. */
26
31
  provider?: string;
27
32
  }
28
33
  /**
29
- * USD cost estimate for one model call. `cachedTokens` are a subset of
30
- * `promptTokens` billed at the cached-input rate; `reasoningTokens` are output
31
- * tokens already counted in `completionTokens` (kept for reporting).
34
+ * USD cost estimate for one model call.
35
+ *
36
+ * `promptTokens` is the TOTAL billable input, and `cachedTokens` (reads) and
37
+ * `cachedWriteTokens` (creations) are SUBSETS of it — so the three partition the input
38
+ * into full-rate, read-rate and write-rate portions. Normalizing onto that convention is
39
+ * the caller's job (extractUsage does it): providers disagree about whether cache tokens
40
+ * sit inside their reported input count, and pricing must not have to know which.
41
+ *
42
+ * `reasoningTokens` are output tokens already counted in `completionTokens` (kept for
43
+ * reporting).
32
44
  */
33
45
  export declare function estimateCost(model: string, a?: CostArgs): number;
package/dist/pricing.js CHANGED
@@ -8,7 +8,7 @@
8
8
  // always be traced to the table that produced it.
9
9
  import { isSelfHosted } from "./providers.js";
10
10
  // Bump whenever any rate below changes. Stamped onto every estimate.
11
- export const PRICING_VERSION = "2026-07-24";
11
+ export const PRICING_VERSION = "2026-08-06";
12
12
  // Canonical model id -> Rate. Lowercase, provider-agnostic.
13
13
  const TABLE = {
14
14
  // OpenAI
@@ -130,18 +130,65 @@ export function rateFor(model, provider = "") {
130
130
  export function isKnown(model) {
131
131
  return canonical(model) !== null;
132
132
  }
133
+ // Cache-WRITE multipliers, applied to a model's input rate.
134
+ //
135
+ // Writing a prompt into a provider's cache is a separately-billed event from reading it
136
+ // back, and the two go in OPPOSITE directions: Anthropic surcharges a write to 1.25x
137
+ // input and discounts a read to 0.1x, while OpenAI does not bill writes at all. Recording
138
+ // only reads therefore does not merely lose a number — it prices the expensive half of
139
+ // caching at zero, and does so for the agents that cache hardest.
140
+ //
141
+ // Expressed as a multiplier rather than a per-model rate because that is how providers
142
+ // publish it: one ratio per model family. A multiplier also cannot drift out of step with
143
+ // a model's input rate the way a duplicated absolute number can.
144
+ const CACHE_WRITE_MULTIPLIER = [
145
+ ["claude-", 1.25], // Anthropic bills a cache write at 1.25x input
146
+ ["gpt-", 0.0], // OpenAI prompt caching is automatic; writes are not billed
147
+ ["o1", 0.0],
148
+ ["o3", 0.0],
149
+ ["o4", 0.0],
150
+ ["gemini-", 0.0], // implicit context caching is free
151
+ ];
152
+ // Multiplier for a family with no published cache-write behaviour: a write costs what an
153
+ // ordinary input token costs. Not 0.0, which would make an unrecognised model's caching
154
+ // silently free — the flattering direction.
155
+ const DEFAULT_CACHE_WRITE_MULTIPLIER = 1.0;
156
+ /** USD per 1M tokens for writing a prompt into the provider's cache. */
157
+ export function cacheWriteRate(model, provider = "") {
158
+ const rate = rateFor(model, provider);
159
+ const key = canonical(model, provider) ?? "";
160
+ for (const [prefix, mult] of CACHE_WRITE_MULTIPLIER) {
161
+ if (key.startsWith(prefix))
162
+ return rate.input * mult;
163
+ }
164
+ return rate.input * DEFAULT_CACHE_WRITE_MULTIPLIER;
165
+ }
133
166
  /**
134
- * USD cost estimate for one model call. `cachedTokens` are a subset of
135
- * `promptTokens` billed at the cached-input rate; `reasoningTokens` are output
136
- * tokens already counted in `completionTokens` (kept for reporting).
167
+ * USD cost estimate for one model call.
168
+ *
169
+ * `promptTokens` is the TOTAL billable input, and `cachedTokens` (reads) and
170
+ * `cachedWriteTokens` (creations) are SUBSETS of it — so the three partition the input
171
+ * into full-rate, read-rate and write-rate portions. Normalizing onto that convention is
172
+ * the caller's job (extractUsage does it): providers disagree about whether cache tokens
173
+ * sit inside their reported input count, and pricing must not have to know which.
174
+ *
175
+ * `reasoningTokens` are output tokens already counted in `completionTokens` (kept for
176
+ * reporting).
137
177
  */
138
178
  export function estimateCost(model, a = {}) {
139
179
  const rate = rateFor(model, a.provider ?? "");
140
180
  const prompt = Math.max(0, a.promptTokens ?? 0);
141
181
  const completion = Math.max(0, a.completionTokens ?? 0);
142
- const cached = Math.max(0, Math.min(a.cachedTokens ?? 0, prompt));
143
- const fullInput = prompt - cached;
144
- const cachedRate = rate.cachedInput ?? rate.input;
145
- const cost = (fullInput * rate.input + cached * cachedRate + completion * rate.output) / 1_000_000;
182
+ // Reads are taken out first, then writes from what remains, so the two subsets can
183
+ // never overlap and bill the same token twice.
184
+ const read = Math.max(0, Math.min(a.cachedTokens ?? 0, prompt));
185
+ const write = Math.max(0, Math.min(a.cachedWriteTokens ?? 0, prompt - read));
186
+ const fullInput = prompt - read - write;
187
+ const readRate = rate.cachedInput ?? rate.input;
188
+ const cost = (fullInput * rate.input +
189
+ read * readRate +
190
+ write * cacheWriteRate(model, a.provider ?? "") +
191
+ completion * rate.output) /
192
+ 1_000_000;
146
193
  return Math.round(cost * 1e8) / 1e8;
147
194
  }