pyyol 1.9.0 → 1.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.d.ts +24 -0
- package/dist/cli.js +77 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.js +11 -0
- package/dist/instrument.d.ts +29 -3
- package/dist/instrument.js +307 -102
- package/dist/movetools.d.ts +141 -0
- package/dist/movetools.js +486 -0
- package/dist/pricing.d.ts +16 -4
- package/dist/pricing.js +55 -8
- package/dist/scaffold.d.ts +74 -0
- package/dist/scaffold.js +276 -0
- package/dist/telemetry.d.ts +24 -0
- package/dist/telemetry.js +62 -0
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +6 -2
- package/rules/llms-full.txt +687 -26
- package/skill/SKILL.md +1 -0
- package/skill/references/telemetry.md +55 -0
|
@@ -0,0 +1,486 @@
|
|
|
1
|
+
// Structured move tools: how an agent proves its model chose the move it played.
|
|
2
|
+
//
|
|
3
|
+
// WHAT THIS IS FOR. Routing a call through the Pyyol Gateway proves a model was called for a
|
|
4
|
+
// turn. It does not prove the model's answer became the move — an agent could call the model,
|
|
5
|
+
// discard the response, and submit a scripted move with every proof valid. Completion binding
|
|
6
|
+
// closes that: the gateway reads the move out of the model's own structured tool call and the
|
|
7
|
+
// match rejects a submitted move that contradicts it.
|
|
8
|
+
//
|
|
9
|
+
// So a move has to arrive as a TOOL CALL, not as prose the agent parses:
|
|
10
|
+
//
|
|
11
|
+
// import * as pyyol from "@pyyol/sdk";
|
|
12
|
+
//
|
|
13
|
+
// const client = pyyol.route(new Anthropic());
|
|
14
|
+
//
|
|
15
|
+
// agent.onTurn("goofspiel", async (view) => {
|
|
16
|
+
// const resp = await client.messages.create({
|
|
17
|
+
// model: "claude-opus-4",
|
|
18
|
+
// tools: [pyyol.moveTool("goofspiel", "anthropic")],
|
|
19
|
+
// tool_choice: pyyol.moveToolChoice("goofspiel", "anthropic"), // REQUIRE the call
|
|
20
|
+
// messages: [{ role: "user", content: JSON.stringify(view) }],
|
|
21
|
+
// });
|
|
22
|
+
// const args = pyyol.moveFromResponse("goofspiel", resp);
|
|
23
|
+
// return { card: args!.card as number, round: view.round };
|
|
24
|
+
// });
|
|
25
|
+
//
|
|
26
|
+
// WHY PROSE IS NOT AN OPTION. "I'll play the 7", "seven, I think" and "7." are one move to a
|
|
27
|
+
// human and three strings to a parser. Enforcing against a text parse would reject honest
|
|
28
|
+
// agents constantly, so the platform never guesses: no tool call means the turn is simply
|
|
29
|
+
// UNVERIFIED, which costs an agent its verified standing but never a move.
|
|
30
|
+
//
|
|
31
|
+
// WHAT IS NOT CHECKED, deliberately. Nothing here constrains the PROMPT. An agent may frame
|
|
32
|
+
// the game however it likes, including in ways that steer the model toward an answer it
|
|
33
|
+
// already wanted. That is prompt engineering — strategy on this platform, not fraud.
|
|
34
|
+
//
|
|
35
|
+
// The canonical reduction below is pinned by sdk/conformance/move_binding.json, shared with the
|
|
36
|
+
// Go gateway (internal/movebind) and the Python SDK. A divergence between the three would not
|
|
37
|
+
// read as a bug; it would read as the platform telling a developer their agent did not play
|
|
38
|
+
// what its model chose.
|
|
39
|
+
// Tool names, one per game. These strings are the contract: the gateway looks for exactly
|
|
40
|
+
// these, so a rename here silently stops binding every move while every local test that mocks
|
|
41
|
+
// its own name keeps passing.
|
|
42
|
+
export const TOOL_GOOFSPIEL = "play_card";
|
|
43
|
+
export const TOOL_MAFIA = "mafia_action";
|
|
44
|
+
export const TOOL_MONOPOLY = "monopoly_action";
|
|
45
|
+
export const GAME_GOOFSPIEL = "goofspiel";
|
|
46
|
+
export const GAME_MAFIA = "mafia";
|
|
47
|
+
export const GAME_MONOPOLY = "monopoly";
|
|
48
|
+
/**
|
|
49
|
+
* NO_TARGET is the wire convention for "this action names no seat".
|
|
50
|
+
*
|
|
51
|
+
* NOT zero, and worth reading twice: seat 0 is a real player. A forgotten target used to act
|
|
52
|
+
* silently on seat 0, which is why MafiaMove defaults to -1 — and the canonical form has to
|
|
53
|
+
* agree, or every untargeted action would bind as an action against that player.
|
|
54
|
+
*/
|
|
55
|
+
export const NO_TARGET = -1;
|
|
56
|
+
const TOOL_BY_GAME = {
|
|
57
|
+
[GAME_GOOFSPIEL]: TOOL_GOOFSPIEL,
|
|
58
|
+
[GAME_MAFIA]: TOOL_MAFIA,
|
|
59
|
+
[GAME_MONOPOLY]: TOOL_MONOPOLY,
|
|
60
|
+
};
|
|
61
|
+
// JSON Schema for each game's move arguments. Kept minimal on purpose: every field a model
|
|
62
|
+
// must fill is a field it can fill wrongly, and a wrong field means an unbound turn.
|
|
63
|
+
const SCHEMAS = {
|
|
64
|
+
[GAME_GOOFSPIEL]: {
|
|
65
|
+
type: "object",
|
|
66
|
+
properties: {
|
|
67
|
+
card: { type: "integer", description: "The card to play from your hand." },
|
|
68
|
+
},
|
|
69
|
+
required: ["card"],
|
|
70
|
+
},
|
|
71
|
+
[GAME_MAFIA]: {
|
|
72
|
+
type: "object",
|
|
73
|
+
properties: {
|
|
74
|
+
kind: {
|
|
75
|
+
type: "string",
|
|
76
|
+
description: "The action verb, e.g. night_kill, investigate, protect, profile, vote, abstain.",
|
|
77
|
+
},
|
|
78
|
+
target: {
|
|
79
|
+
type: "integer",
|
|
80
|
+
description: "The seat this action is aimed at. Use -1 when the action names no seat — " +
|
|
81
|
+
"seat 0 is a real player, so 0 is never 'nobody'.",
|
|
82
|
+
},
|
|
83
|
+
},
|
|
84
|
+
required: ["kind"],
|
|
85
|
+
},
|
|
86
|
+
[GAME_MONOPOLY]: {
|
|
87
|
+
type: "object",
|
|
88
|
+
properties: {
|
|
89
|
+
kind: {
|
|
90
|
+
type: "string",
|
|
91
|
+
description: "The action verb, e.g. buy, pass, bid, mortgage, build.",
|
|
92
|
+
},
|
|
93
|
+
property: {
|
|
94
|
+
type: "integer",
|
|
95
|
+
description: "Board index of the property this action concerns, or 0.",
|
|
96
|
+
},
|
|
97
|
+
amount: { type: "integer", description: "Coin amount this action carries, or 0." },
|
|
98
|
+
},
|
|
99
|
+
required: ["kind"],
|
|
100
|
+
},
|
|
101
|
+
};
|
|
102
|
+
const DESCRIPTIONS = {
|
|
103
|
+
[GAME_GOOFSPIEL]: "Play one card from your hand for this round. Call this to make your move.",
|
|
104
|
+
[GAME_MAFIA]: "Take your action for this phase. Call this to make your move.",
|
|
105
|
+
[GAME_MONOPOLY]: "Take your action for this turn. Call this to make your move.",
|
|
106
|
+
};
|
|
107
|
+
/** The tool name that carries a move for `game`, or "" if the game has no contract. */
|
|
108
|
+
export function moveToolName(game) {
|
|
109
|
+
return TOOL_BY_GAME[game] ?? "";
|
|
110
|
+
}
|
|
111
|
+
/**
|
|
112
|
+
* The move tool definition, shaped for `provider`.
|
|
113
|
+
*
|
|
114
|
+
* Providers disagree about the envelope while agreeing on the JSON Schema inside it, so the
|
|
115
|
+
* schema is defined once and wrapped per provider. Emitting the wrong envelope is a 400 from
|
|
116
|
+
* the provider rather than a silent problem, which is why this is worth getting from the SDK
|
|
117
|
+
* instead of hand-writing.
|
|
118
|
+
*/
|
|
119
|
+
/**
|
|
120
|
+
* The batching form of a move schema: a plan of per-round moves.
|
|
121
|
+
*
|
|
122
|
+
* A SEPARATE schema rather than an optional `plan` property beside `card`, because a schema
|
|
123
|
+
* accepting either shape has to drop `required`, and a model handed an all-optional object will
|
|
124
|
+
* sometimes return an empty one. Strict modes are also unenthusiastic about `oneOf`. So an agent
|
|
125
|
+
* that batches asks for the plan tool and is told exactly one shape; an agent that does not gets
|
|
126
|
+
* today's schema untouched.
|
|
127
|
+
*/
|
|
128
|
+
function planSchema(base, rounds) {
|
|
129
|
+
const props = (base.properties ?? {});
|
|
130
|
+
const required = (base.required ?? []);
|
|
131
|
+
return {
|
|
132
|
+
type: "object",
|
|
133
|
+
properties: {
|
|
134
|
+
[PLAN_KEY]: {
|
|
135
|
+
type: "array",
|
|
136
|
+
minItems: 1,
|
|
137
|
+
maxItems: Math.min(rounds, MAX_SPAN_ROUNDS),
|
|
138
|
+
description: "One entry per round you are deciding now, starting at the current round. Every " +
|
|
139
|
+
"round you list is bound to the move you give it, so list only rounds you intend " +
|
|
140
|
+
"to play exactly as planned.",
|
|
141
|
+
items: {
|
|
142
|
+
type: "object",
|
|
143
|
+
properties: {
|
|
144
|
+
round: {
|
|
145
|
+
type: "integer",
|
|
146
|
+
description: "The round this move is for. Must be the current round or a later one — a " +
|
|
147
|
+
"move for a round already played cannot be bound.",
|
|
148
|
+
},
|
|
149
|
+
...props,
|
|
150
|
+
},
|
|
151
|
+
required: ["round", ...required],
|
|
152
|
+
},
|
|
153
|
+
},
|
|
154
|
+
},
|
|
155
|
+
required: [PLAN_KEY],
|
|
156
|
+
};
|
|
157
|
+
}
|
|
158
|
+
export function moveTool(game, provider = "openai", planRounds) {
|
|
159
|
+
let schema = SCHEMAS[game];
|
|
160
|
+
if (!schema) {
|
|
161
|
+
throw new Error(`pyyol.moveTool: no move tool for game ${JSON.stringify(game)}; ` +
|
|
162
|
+
`known games are ${Object.keys(TOOL_BY_GAME).sort().join(", ")}`);
|
|
163
|
+
}
|
|
164
|
+
if (planRounds !== undefined) {
|
|
165
|
+
if (planRounds < 1)
|
|
166
|
+
throw new Error("pyyol.moveTool: planRounds must be at least 1");
|
|
167
|
+
schema = planSchema(schema, planRounds);
|
|
168
|
+
}
|
|
169
|
+
const name = TOOL_BY_GAME[game];
|
|
170
|
+
const description = DESCRIPTIONS[game];
|
|
171
|
+
const p = (provider || "").toLowerCase();
|
|
172
|
+
if (p === "anthropic")
|
|
173
|
+
return { name, description, input_schema: schema };
|
|
174
|
+
if (p === "google")
|
|
175
|
+
return { name, description, parameters: schema };
|
|
176
|
+
// OpenAI chat completions. The Responses API accepts the flattened form; both are
|
|
177
|
+
// understood by moveFromResponse, so an agent that uses either is bound the same.
|
|
178
|
+
return { type: "function", function: { name, description, parameters: schema } };
|
|
179
|
+
}
|
|
180
|
+
/**
|
|
181
|
+
* The provider-specific way to REQUIRE the move tool.
|
|
182
|
+
*
|
|
183
|
+
* Worth using. Without it a model may answer in prose, and a turn with no tool call is
|
|
184
|
+
* unverified — the agent keeps playing but earns no completion binding.
|
|
185
|
+
*/
|
|
186
|
+
export function moveToolChoice(game, provider = "openai") {
|
|
187
|
+
const name = moveToolName(game);
|
|
188
|
+
if (!name)
|
|
189
|
+
throw new Error(`pyyol.moveToolChoice: no move tool for game ${JSON.stringify(game)}`);
|
|
190
|
+
const p = (provider || "").toLowerCase();
|
|
191
|
+
if (p === "anthropic")
|
|
192
|
+
return { type: "tool", name };
|
|
193
|
+
if (p === "google") {
|
|
194
|
+
return { function_calling_config: { mode: "ANY", allowed_function_names: [name] } };
|
|
195
|
+
}
|
|
196
|
+
return { type: "function", function: { name } };
|
|
197
|
+
}
|
|
198
|
+
/**
|
|
199
|
+
* The move arguments the model emitted, or null if it emitted no usable move call.
|
|
200
|
+
*
|
|
201
|
+
* STRUCTURAL, not per-provider. Every provider that has ever expressed a tool call has
|
|
202
|
+
* expressed it as a name beside an arguments blob, as SIBLINGS in one object:
|
|
203
|
+
*
|
|
204
|
+
* OpenAI {"function": {"name": "play_card", "arguments": "{\"card\":7}"}}
|
|
205
|
+
* Responses {"type": "function_call", "name": "play_card", "arguments": "{...}"}
|
|
206
|
+
* Anthropic {"type": "tool_use", "name": "play_card", "input": {"card": 7}}
|
|
207
|
+
* Google {"functionCall": {"name": "play_card", "args": {"card": 7}}}
|
|
208
|
+
* Bedrock {"toolUse": {"name": "play_card", "input": {"card": 7}}}
|
|
209
|
+
* Ollama {"function": {"name": "play_card", "arguments": {"card": 7}}}
|
|
210
|
+
*
|
|
211
|
+
* So the walk looks for that structure anywhere in the document and a provider nobody has heard
|
|
212
|
+
* of works on the day it ships. Enumerating shapes loses by construction: new providers appear
|
|
213
|
+
* constantly, every self-hosted server has its own dialect, and an unlisted one fails SILENTLY —
|
|
214
|
+
* the turn is never bound and nobody learns why.
|
|
215
|
+
*
|
|
216
|
+
* SAFE because the tool NAME is the discriminator and it is ours. The one near-miss is a response
|
|
217
|
+
* echoing the tool DEFINITION, which is why "parameters" is NOT accepted as an arguments key: a
|
|
218
|
+
* JSON Schema yields no card and falls through to null rather than to a wrong move. That
|
|
219
|
+
* direction matters — a wrong move REJECTS an honest turn, a miss only leaves it unverified.
|
|
220
|
+
*
|
|
221
|
+
* Returns the LAST matching call: a model that corrected itself stands behind its final answer.
|
|
222
|
+
*/
|
|
223
|
+
export function moveFromResponse(game, resp) {
|
|
224
|
+
const want = moveToolName(game);
|
|
225
|
+
if (!want)
|
|
226
|
+
return null;
|
|
227
|
+
const found = findToolCalls(resp, want);
|
|
228
|
+
return found.length ? found[found.length - 1] : null;
|
|
229
|
+
}
|
|
230
|
+
// The sibling fields that carry a tool call's arguments, across every provider shape seen so far.
|
|
231
|
+
// "parameters" is EXCLUDED on purpose — it is the JSON Schema keyword, so accepting it would let a
|
|
232
|
+
// tool DEFINITION echoed back in a response be read as a tool CALL.
|
|
233
|
+
const ARGS_KEYS = ["arguments", "input", "args"];
|
|
234
|
+
/**
|
|
235
|
+
* Collect every (name === toolName, arguments) pair in the document, in document order.
|
|
236
|
+
*
|
|
237
|
+
* Object keys are walked in SORTED order so the result is deterministic. The caller takes the
|
|
238
|
+
* last match, so an unstable walk would make which move gets bound depend on key insertion
|
|
239
|
+
* order — a coin flip deciding whether an honest turn is accepted.
|
|
240
|
+
*/
|
|
241
|
+
function findToolCalls(node, toolName) {
|
|
242
|
+
const out = [];
|
|
243
|
+
if (Array.isArray(node)) {
|
|
244
|
+
for (const item of node)
|
|
245
|
+
out.push(...findToolCalls(item, toolName));
|
|
246
|
+
return out;
|
|
247
|
+
}
|
|
248
|
+
if (node === null || typeof node !== "object")
|
|
249
|
+
return out;
|
|
250
|
+
const obj = node;
|
|
251
|
+
if (obj.name === toolName) {
|
|
252
|
+
for (const k of ARGS_KEYS) {
|
|
253
|
+
if (!(k in obj))
|
|
254
|
+
continue;
|
|
255
|
+
const args = decodeArgsValue(obj[k]);
|
|
256
|
+
if (args) {
|
|
257
|
+
out.push(args);
|
|
258
|
+
break;
|
|
259
|
+
}
|
|
260
|
+
}
|
|
261
|
+
}
|
|
262
|
+
for (const k of Object.keys(obj).sort())
|
|
263
|
+
out.push(...findToolCalls(obj[k], toolName));
|
|
264
|
+
return out;
|
|
265
|
+
}
|
|
266
|
+
/**
|
|
267
|
+
* Accept an already-decoded object, or a JSON string containing one.
|
|
268
|
+
*
|
|
269
|
+
* OpenAI-family providers send arguments as a STRING; Anthropic, Google, Bedrock and Ollama send
|
|
270
|
+
* an object. Both land here so no caller needs to know which.
|
|
271
|
+
*/
|
|
272
|
+
function decodeArgsValue(raw) {
|
|
273
|
+
if (raw && typeof raw === "object" && !Array.isArray(raw)) {
|
|
274
|
+
return raw;
|
|
275
|
+
}
|
|
276
|
+
if (typeof raw === "string") {
|
|
277
|
+
const t = raw.trim();
|
|
278
|
+
if (!t)
|
|
279
|
+
return null;
|
|
280
|
+
try {
|
|
281
|
+
const decoded = JSON.parse(t);
|
|
282
|
+
if (decoded && typeof decoded === "object" && !Array.isArray(decoded)) {
|
|
283
|
+
return decoded;
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
catch {
|
|
287
|
+
return null;
|
|
288
|
+
}
|
|
289
|
+
}
|
|
290
|
+
return null;
|
|
291
|
+
}
|
|
292
|
+
/**
|
|
293
|
+
* Read an integer argument.
|
|
294
|
+
*
|
|
295
|
+
* Tolerant of a model that quoted the number, because that is a formatting habit rather than a
|
|
296
|
+
* different decision. NOT tolerant of a fractional value: 7.5 is not a card, and rounding it
|
|
297
|
+
* would invent a move the model did not make — which would then reject the agent's real one.
|
|
298
|
+
*/
|
|
299
|
+
function intArg(args, key) {
|
|
300
|
+
const v = args[key];
|
|
301
|
+
if (typeof v === "boolean")
|
|
302
|
+
return [0, false];
|
|
303
|
+
if (typeof v === "number")
|
|
304
|
+
return Number.isInteger(v) ? [v, true] : [0, false];
|
|
305
|
+
if (typeof v === "string") {
|
|
306
|
+
const s = v.trim();
|
|
307
|
+
// Number() accepts "" and " " as 0 and "1e3" as 1000; neither is a move a model wrote as
|
|
308
|
+
// an integer, and accepting them would bind a value the model did not name.
|
|
309
|
+
if (!/^[+-]?\d+$/.test(s))
|
|
310
|
+
return [0, false];
|
|
311
|
+
return [Number(s), true];
|
|
312
|
+
}
|
|
313
|
+
return [0, false];
|
|
314
|
+
}
|
|
315
|
+
function strArg(args, key) {
|
|
316
|
+
const v = args[key];
|
|
317
|
+
if (typeof v === "string")
|
|
318
|
+
return v;
|
|
319
|
+
if (typeof v === "number")
|
|
320
|
+
return String(v);
|
|
321
|
+
return "";
|
|
322
|
+
}
|
|
323
|
+
/** The bound form of a Goofspiel move: the card, and nothing else. */
|
|
324
|
+
export function canonGoofspiel(card) {
|
|
325
|
+
return `card:${Math.trunc(card)}`;
|
|
326
|
+
}
|
|
327
|
+
/**
|
|
328
|
+
* The bound form of a Mafia action: the verb and its target seat.
|
|
329
|
+
*
|
|
330
|
+
* The PHASE is deliberately excluded — it is server state, not the model's choice, and binding
|
|
331
|
+
* it would reject an honest turn over a field the model had no say in.
|
|
332
|
+
*
|
|
333
|
+
* Negative targets collapse to one token; ZERO DOES NOT. Seat 0 is an ordinary player, and
|
|
334
|
+
* abstaining is its own action kind rather than a sentinel target, so "no seat" is only ever an
|
|
335
|
+
* absent or negative field. Collapsing 0 too would let a move against that one player be
|
|
336
|
+
* substituted for doing nothing.
|
|
337
|
+
*/
|
|
338
|
+
export function canonMafia(kind, target) {
|
|
339
|
+
const t = Math.trunc(target) < 0 ? "none" : String(Math.trunc(target));
|
|
340
|
+
return `${kind.trim().toLowerCase()}:${t}`;
|
|
341
|
+
}
|
|
342
|
+
/**
|
|
343
|
+
* The bound form of a Monopoly action: verb, property, amount.
|
|
344
|
+
*
|
|
345
|
+
* All three are always rendered, including zeros. Omitting an absent field would let "mortgage
|
|
346
|
+
* property 0 for 50" and "mortgage property 50 for 0" reduce to the same string, and two
|
|
347
|
+
* different decisions sharing one canonical form is the one thing this mechanism cannot tolerate.
|
|
348
|
+
*/
|
|
349
|
+
export function canonMonopoly(kind, property = 0, amount = 0) {
|
|
350
|
+
return `${kind.trim().toLowerCase()}:${Math.trunc(property)}:${Math.trunc(amount)}`;
|
|
351
|
+
}
|
|
352
|
+
/**
|
|
353
|
+
* Reduce move arguments to the canonical string a bound decision stores.
|
|
354
|
+
*
|
|
355
|
+
* null means "nothing bindable here", which callers must treat as an unverified turn and never
|
|
356
|
+
* as a wrong move.
|
|
357
|
+
*/
|
|
358
|
+
export function canonMove(game, args) {
|
|
359
|
+
if (!args)
|
|
360
|
+
return null;
|
|
361
|
+
if (game === GAME_GOOFSPIEL) {
|
|
362
|
+
const [card, ok] = intArg(args, "card");
|
|
363
|
+
return ok ? canonGoofspiel(card) : null;
|
|
364
|
+
}
|
|
365
|
+
if (game === GAME_MAFIA) {
|
|
366
|
+
const kind = strArg(args, "kind").trim();
|
|
367
|
+
if (!kind)
|
|
368
|
+
return null;
|
|
369
|
+
const [target, has] = intArg(args, "target");
|
|
370
|
+
return canonMafia(kind, has ? target : NO_TARGET);
|
|
371
|
+
}
|
|
372
|
+
if (game === GAME_MONOPOLY) {
|
|
373
|
+
const kind = strArg(args, "kind").trim();
|
|
374
|
+
if (!kind)
|
|
375
|
+
return null;
|
|
376
|
+
const [property] = intArg(args, "property");
|
|
377
|
+
const [amount] = intArg(args, "amount");
|
|
378
|
+
return canonMonopoly(kind, property, amount);
|
|
379
|
+
}
|
|
380
|
+
return null;
|
|
381
|
+
}
|
|
382
|
+
/**
|
|
383
|
+
* The canonical move the platform will bind for this response, or null.
|
|
384
|
+
*
|
|
385
|
+
* The one call worth making in a test: it is exactly what the gateway does, so an agent that
|
|
386
|
+
* asserts on this locally cannot be surprised by a rejection in a real match.
|
|
387
|
+
*/
|
|
388
|
+
export function boundMove(game, resp) {
|
|
389
|
+
return canonMove(game, moveFromResponse(game, resp));
|
|
390
|
+
}
|
|
391
|
+
/**
|
|
392
|
+
* The argument that carries a multi-round decision.
|
|
393
|
+
*
|
|
394
|
+
* Named once, and it must match the Go gateway and the Python SDK exactly: a mismatch would
|
|
395
|
+
* not throw, it would silently fall back to single-round binding and quietly restore the
|
|
396
|
+
* coverage problem range bindings exist to fix.
|
|
397
|
+
*/
|
|
398
|
+
export const PLAN_KEY = "plan";
|
|
399
|
+
/**
|
|
400
|
+
* The most rounds one completion may claim to have decided.
|
|
401
|
+
*
|
|
402
|
+
* Bounded because the plan is attacker-supplied — uncapped, a single call could assert a
|
|
403
|
+
* hundred thousand rounds and become that many database writes. Comfortably above any real
|
|
404
|
+
* game, so a legitimate agent never meets it.
|
|
405
|
+
*/
|
|
406
|
+
export const MAX_SPAN_ROUNDS = 64;
|
|
407
|
+
/**
|
|
408
|
+
* Reduce move arguments to EVERY round they decided.
|
|
409
|
+
*
|
|
410
|
+
* # Why a completion may cover more than one round
|
|
411
|
+
*
|
|
412
|
+
* Coverage used to count CALLS, so one completion bound one round. An agent that batches — one
|
|
413
|
+
* call planning three rounds — therefore scored about 33% on real staked tables while playing
|
|
414
|
+
* entirely model-backed, and cost optimisation is something this platform means to REWARD.
|
|
415
|
+
* Coverage now means "decisions a model made" rather than "calls made".
|
|
416
|
+
*
|
|
417
|
+
* # Why claiming a span is safe
|
|
418
|
+
*
|
|
419
|
+
* A span is a COMMITMENT, not a free coverage win. Match-time enforcement is unchanged, so
|
|
420
|
+
* submitting anything other than the bound move for a covered round is rejected exactly as a
|
|
421
|
+
* substitution is. An agent that over-claims has only tied its own hands.
|
|
422
|
+
*
|
|
423
|
+
* # The one thing a span must never do
|
|
424
|
+
*
|
|
425
|
+
* Rounds before `provenRound` are DROPPED. Those turns have already been played, so a binding
|
|
426
|
+
* over them is coverage nothing will ever check — an agent could retroactively claim turns it
|
|
427
|
+
* played unbound. Forward claims are self-limiting because they are enforced.
|
|
428
|
+
*
|
|
429
|
+
* null means nothing is bindable. A plan naming one round twice returns null WHOLE: two moves
|
|
430
|
+
* for one slot has no honest reading, and picking either would be guessing for the agent.
|
|
431
|
+
*/
|
|
432
|
+
export function canonPlan(game, args, provenRound) {
|
|
433
|
+
if (!args)
|
|
434
|
+
return null;
|
|
435
|
+
const entries = planEntries(args);
|
|
436
|
+
if (entries === null) {
|
|
437
|
+
// No plan: the ordinary single-round call, unchanged.
|
|
438
|
+
const move = canonMove(game, args);
|
|
439
|
+
return move ? [{ round: provenRound, move }] : null;
|
|
440
|
+
}
|
|
441
|
+
if (entries.length > MAX_SPAN_ROUNDS)
|
|
442
|
+
return null;
|
|
443
|
+
const seen = new Set();
|
|
444
|
+
const out = [];
|
|
445
|
+
for (const entry of entries) {
|
|
446
|
+
const [round, has] = intArg(entry, "round");
|
|
447
|
+
// Backward or unplaceable: skipped, never fatal. A model that emitted one bad entry has
|
|
448
|
+
// still honestly decided the others.
|
|
449
|
+
if (!has || round < provenRound)
|
|
450
|
+
continue;
|
|
451
|
+
if (seen.has(round))
|
|
452
|
+
return null;
|
|
453
|
+
const move = canonMove(game, entry);
|
|
454
|
+
if (!move)
|
|
455
|
+
continue;
|
|
456
|
+
seen.add(round);
|
|
457
|
+
out.push({ round, move });
|
|
458
|
+
}
|
|
459
|
+
if (out.length === 0)
|
|
460
|
+
return null;
|
|
461
|
+
out.sort((a, b) => a.round - b.round);
|
|
462
|
+
return out;
|
|
463
|
+
}
|
|
464
|
+
/**
|
|
465
|
+
* The per-round argument objects in a plan, or null when this call carries no plan.
|
|
466
|
+
*
|
|
467
|
+
* null must mean "no plan" rather than "empty plan": the caller falls back to single-round
|
|
468
|
+
* binding on null, and binding nothing would break every agent shipping today.
|
|
469
|
+
*/
|
|
470
|
+
function planEntries(args) {
|
|
471
|
+
const raw = args[PLAN_KEY];
|
|
472
|
+
if (!Array.isArray(raw) || raw.length === 0)
|
|
473
|
+
return null;
|
|
474
|
+
const out = raw.filter((item) => typeof item === "object" && item !== null && !Array.isArray(item));
|
|
475
|
+
return out.length > 0 ? out : null;
|
|
476
|
+
}
|
|
477
|
+
/**
|
|
478
|
+
* Every round this response will bind, exactly as the gateway will read it.
|
|
479
|
+
*
|
|
480
|
+
* Worth calling in a test before shipping a batching agent: if this does not list the round you
|
|
481
|
+
* are about to play, that turn will not be bound, and if it lists a DIFFERENT move than you
|
|
482
|
+
* intend to submit, the match will reject it.
|
|
483
|
+
*/
|
|
484
|
+
export function boundPlan(game, resp, provenRound) {
|
|
485
|
+
return canonPlan(game, moveFromResponse(game, resp), provenRound);
|
|
486
|
+
}
|
package/dist/pricing.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
export declare const PRICING_VERSION = "2026-
|
|
1
|
+
export declare const PRICING_VERSION = "2026-08-06";
|
|
2
2
|
export interface Rate {
|
|
3
3
|
/** USD per 1M input tokens. */
|
|
4
4
|
input: number;
|
|
@@ -16,18 +16,30 @@ export declare function canonical(model: string, provider?: string): string | nu
|
|
|
16
16
|
export declare function rateFor(model: string, provider?: string): Rate;
|
|
17
17
|
/** True if the model maps to an explicit table entry (not the fallback). */
|
|
18
18
|
export declare function isKnown(model: string): boolean;
|
|
19
|
+
/** USD per 1M tokens for writing a prompt into the provider's cache. */
|
|
20
|
+
export declare function cacheWriteRate(model: string, provider?: string): number;
|
|
19
21
|
export interface CostArgs {
|
|
20
22
|
promptTokens?: number;
|
|
21
23
|
completionTokens?: number;
|
|
24
|
+
/** Prompt-cache READ tokens (a subset of promptTokens). */
|
|
22
25
|
cachedTokens?: number;
|
|
26
|
+
/** Prompt-cache WRITE/creation tokens (also a subset of promptTokens). */
|
|
27
|
+
cachedWriteTokens?: number;
|
|
23
28
|
reasoningTokens?: number;
|
|
24
29
|
/** WHO served the call. Decides whether there is a bill at all: the same model id
|
|
25
30
|
* is billed on a hosted provider and free on the developer's own hardware. */
|
|
26
31
|
provider?: string;
|
|
27
32
|
}
|
|
28
33
|
/**
|
|
29
|
-
* USD cost estimate for one model call.
|
|
30
|
-
*
|
|
31
|
-
*
|
|
34
|
+
* USD cost estimate for one model call.
|
|
35
|
+
*
|
|
36
|
+
* `promptTokens` is the TOTAL billable input, and `cachedTokens` (reads) and
|
|
37
|
+
* `cachedWriteTokens` (creations) are SUBSETS of it — so the three partition the input
|
|
38
|
+
* into full-rate, read-rate and write-rate portions. Normalizing onto that convention is
|
|
39
|
+
* the caller's job (extractUsage does it): providers disagree about whether cache tokens
|
|
40
|
+
* sit inside their reported input count, and pricing must not have to know which.
|
|
41
|
+
*
|
|
42
|
+
* `reasoningTokens` are output tokens already counted in `completionTokens` (kept for
|
|
43
|
+
* reporting).
|
|
32
44
|
*/
|
|
33
45
|
export declare function estimateCost(model: string, a?: CostArgs): number;
|
package/dist/pricing.js
CHANGED
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
// always be traced to the table that produced it.
|
|
9
9
|
import { isSelfHosted } from "./providers.js";
|
|
10
10
|
// Bump whenever any rate below changes. Stamped onto every estimate.
|
|
11
|
-
export const PRICING_VERSION = "2026-
|
|
11
|
+
export const PRICING_VERSION = "2026-08-06";
|
|
12
12
|
// Canonical model id -> Rate. Lowercase, provider-agnostic.
|
|
13
13
|
const TABLE = {
|
|
14
14
|
// OpenAI
|
|
@@ -130,18 +130,65 @@ export function rateFor(model, provider = "") {
|
|
|
130
130
|
export function isKnown(model) {
|
|
131
131
|
return canonical(model) !== null;
|
|
132
132
|
}
|
|
133
|
+
// Cache-WRITE multipliers, applied to a model's input rate.
|
|
134
|
+
//
|
|
135
|
+
// Writing a prompt into a provider's cache is a separately-billed event from reading it
|
|
136
|
+
// back, and the two go in OPPOSITE directions: Anthropic surcharges a write to 1.25x
|
|
137
|
+
// input and discounts a read to 0.1x, while OpenAI does not bill writes at all. Recording
|
|
138
|
+
// only reads therefore does not merely lose a number — it prices the expensive half of
|
|
139
|
+
// caching at zero, and does so for the agents that cache hardest.
|
|
140
|
+
//
|
|
141
|
+
// Expressed as a multiplier rather than a per-model rate because that is how providers
|
|
142
|
+
// publish it: one ratio per model family. A multiplier also cannot drift out of step with
|
|
143
|
+
// a model's input rate the way a duplicated absolute number can.
|
|
144
|
+
const CACHE_WRITE_MULTIPLIER = [
|
|
145
|
+
["claude-", 1.25], // Anthropic bills a cache write at 1.25x input
|
|
146
|
+
["gpt-", 0.0], // OpenAI prompt caching is automatic; writes are not billed
|
|
147
|
+
["o1", 0.0],
|
|
148
|
+
["o3", 0.0],
|
|
149
|
+
["o4", 0.0],
|
|
150
|
+
["gemini-", 0.0], // implicit context caching is free
|
|
151
|
+
];
|
|
152
|
+
// Multiplier for a family with no published cache-write behaviour: a write costs what an
|
|
153
|
+
// ordinary input token costs. Not 0.0, which would make an unrecognised model's caching
|
|
154
|
+
// silently free — the flattering direction.
|
|
155
|
+
const DEFAULT_CACHE_WRITE_MULTIPLIER = 1.0;
|
|
156
|
+
/** USD per 1M tokens for writing a prompt into the provider's cache. */
|
|
157
|
+
export function cacheWriteRate(model, provider = "") {
|
|
158
|
+
const rate = rateFor(model, provider);
|
|
159
|
+
const key = canonical(model, provider) ?? "";
|
|
160
|
+
for (const [prefix, mult] of CACHE_WRITE_MULTIPLIER) {
|
|
161
|
+
if (key.startsWith(prefix))
|
|
162
|
+
return rate.input * mult;
|
|
163
|
+
}
|
|
164
|
+
return rate.input * DEFAULT_CACHE_WRITE_MULTIPLIER;
|
|
165
|
+
}
|
|
133
166
|
/**
|
|
134
|
-
* USD cost estimate for one model call.
|
|
135
|
-
*
|
|
136
|
-
*
|
|
167
|
+
* USD cost estimate for one model call.
|
|
168
|
+
*
|
|
169
|
+
* `promptTokens` is the TOTAL billable input, and `cachedTokens` (reads) and
|
|
170
|
+
* `cachedWriteTokens` (creations) are SUBSETS of it — so the three partition the input
|
|
171
|
+
* into full-rate, read-rate and write-rate portions. Normalizing onto that convention is
|
|
172
|
+
* the caller's job (extractUsage does it): providers disagree about whether cache tokens
|
|
173
|
+
* sit inside their reported input count, and pricing must not have to know which.
|
|
174
|
+
*
|
|
175
|
+
* `reasoningTokens` are output tokens already counted in `completionTokens` (kept for
|
|
176
|
+
* reporting).
|
|
137
177
|
*/
|
|
138
178
|
export function estimateCost(model, a = {}) {
|
|
139
179
|
const rate = rateFor(model, a.provider ?? "");
|
|
140
180
|
const prompt = Math.max(0, a.promptTokens ?? 0);
|
|
141
181
|
const completion = Math.max(0, a.completionTokens ?? 0);
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
const
|
|
145
|
-
const
|
|
182
|
+
// Reads are taken out first, then writes from what remains, so the two subsets can
|
|
183
|
+
// never overlap and bill the same token twice.
|
|
184
|
+
const read = Math.max(0, Math.min(a.cachedTokens ?? 0, prompt));
|
|
185
|
+
const write = Math.max(0, Math.min(a.cachedWriteTokens ?? 0, prompt - read));
|
|
186
|
+
const fullInput = prompt - read - write;
|
|
187
|
+
const readRate = rate.cachedInput ?? rate.input;
|
|
188
|
+
const cost = (fullInput * rate.input +
|
|
189
|
+
read * readRate +
|
|
190
|
+
write * cacheWriteRate(model, a.provider ?? "") +
|
|
191
|
+
completion * rate.output) /
|
|
192
|
+
1_000_000;
|
|
146
193
|
return Math.round(cost * 1e8) / 1e8;
|
|
147
194
|
}
|