pyyol 1.9.0 → 1.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.d.ts +24 -0
- package/dist/cli.js +77 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.js +11 -0
- package/dist/instrument.d.ts +29 -3
- package/dist/instrument.js +307 -102
- package/dist/movetools.d.ts +141 -0
- package/dist/movetools.js +486 -0
- package/dist/pricing.d.ts +16 -4
- package/dist/pricing.js +55 -8
- package/dist/scaffold.d.ts +74 -0
- package/dist/scaffold.js +276 -0
- package/dist/telemetry.d.ts +24 -0
- package/dist/telemetry.js +62 -0
- package/dist/version.d.ts +1 -1
- package/dist/version.js +1 -1
- package/package.json +6 -2
- package/rules/llms-full.txt +687 -26
- package/skill/SKILL.md +1 -0
- package/skill/references/telemetry.md +55 -0
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
/** Bump only for a change that intentionally invalidates existing fingerprints: it is part
|
|
2
|
+
* of the hashed payload, so a bump splits every agent's history into a new epoch. */
|
|
3
|
+
export declare const SCAFFOLD_VERSION = "pyyol-scaffold-v1";
|
|
4
|
+
/** Sampling parameters that change how a model behaves and are therefore part of the
|
|
5
|
+
* scaffold. Anything not listed is ignored, so a provider adding an unrelated field does
|
|
6
|
+
* not silently split every agent's history.
|
|
7
|
+
*
|
|
8
|
+
* `model` is absent on purpose. So are baseURL, apiKey, timeout and stream: the first two
|
|
9
|
+
* would make gateway routing look like a new scaffold, and the last two do not affect what
|
|
10
|
+
* the model decides. */
|
|
11
|
+
export declare const SAMPLING_KEYS: readonly string[];
|
|
12
|
+
type Any = Record<string, unknown> | unknown;
|
|
13
|
+
export type Components = Partial<Record<"client" | "roles" | "sampling" | "tools" | "system", string>>;
|
|
14
|
+
/** The fingerprint components observable in one provider request.
|
|
15
|
+
*
|
|
16
|
+
* Never throws: a fingerprinting problem must not break a developer's model call. */
|
|
17
|
+
export declare function extract(kwargs: Record<string, unknown>, endpoint?: string): Components;
|
|
18
|
+
/** The exact string that gets hashed.
|
|
19
|
+
*
|
|
20
|
+
* Specified rather than incidental: the Python SDK builds the same string and a shared
|
|
21
|
+
* conformance fixture checks both against the same expected fingerprints. Keys are emitted
|
|
22
|
+
* in a FIXED order (not sorted, not insertion order) so neither language's map iteration
|
|
23
|
+
* can affect the result, and an absent component is omitted rather than emitted empty — so
|
|
24
|
+
* adding a component later does not change the fingerprint of requests that never had one. */
|
|
25
|
+
export declare function canonical(c: Components): string;
|
|
26
|
+
/** Short, prefixed id for a scaffold. 16 hex chars of SHA-256 (64 bits).
|
|
27
|
+
*
|
|
28
|
+
* Short enough to read in a UI and group by in SQL. Collisions are irrelevant here in a way
|
|
29
|
+
* they would not be for a security token: fingerprints are only compared WITHIN one agent's
|
|
30
|
+
* own history, so the space that must stay distinct is a handful of harness versions. */
|
|
31
|
+
export declare function fingerprint(c: Components): string;
|
|
32
|
+
/** Why a request could not be fingerprinted, as a stable CODE rather than prose.
|
|
33
|
+
*
|
|
34
|
+
* A code on the wire and prose at the point of reading, deliberately. The reason repeats on
|
|
35
|
+
* every decision of every non-qualifying agent, so shipping and storing the sentence would
|
|
36
|
+
* duplicate it thousands of times per match. A code is also aggregatable — "how many agents
|
|
37
|
+
* are ineligible, and why" is a question worth being able to ask — and its wording can change
|
|
38
|
+
* later without a migration. */
|
|
39
|
+
export declare const ISSUE_NO_SYSTEM_PROMPT = "no_system_prompt";
|
|
40
|
+
export declare const ISSUE_NO_MESSAGES = "no_messages";
|
|
41
|
+
/** Prose for each code. Read by the CLI and the dev-facing trace; never stored. */
|
|
42
|
+
export declare const ISSUE_EXPLANATIONS: Readonly<Record<string, string>>;
|
|
43
|
+
/** Code for why this request yields no usable fingerprint, or "" when it does. */
|
|
44
|
+
export declare function issue(kwargs: Record<string, unknown>, endpoint?: string): string;
|
|
45
|
+
/** Human-readable reason for an issue code, or "" for no issue / an unknown code. */
|
|
46
|
+
export declare function explain(code: string): string;
|
|
47
|
+
/** Prose reason this request cannot be fingerprinted, or "" when it can.
|
|
48
|
+
*
|
|
49
|
+
* Convenience for local developer output; the wire carries `issue` codes. */
|
|
50
|
+
export declare function diagnose(kwargs: Record<string, unknown>, endpoint?: string): string;
|
|
51
|
+
/** Fingerprint one provider request, or "" when it cannot be fingerprinted.
|
|
52
|
+
*
|
|
53
|
+
* An empty result means "unknown", never a hash of nothing — a fingerprint shared by every
|
|
54
|
+
* request that failed to yield components would silently pool unrelated scaffolds into one
|
|
55
|
+
* bogus epoch and publish it as a controlled comparison.
|
|
56
|
+
*
|
|
57
|
+
* A SYSTEM PROMPT IS REQUIRED, and this is the sharp edge of the whole design. Without one,
|
|
58
|
+
* the hashable surface is `client` + `roles` + sampling — none of which move when the
|
|
59
|
+
* developer rewrites the instructions they actually steer the model with, because those
|
|
60
|
+
* instructions sit in a user message alongside the game state.
|
|
61
|
+
*
|
|
62
|
+
* That is worse than having no fingerprint. It is a FALSE CERTIFICATE: an agent could
|
|
63
|
+
* replace its entire strategy prompt mid-season, keep reporting the same scaffold id, and
|
|
64
|
+
* have the improvement attributed to whatever model it swapped to — the fingerprint would be
|
|
65
|
+
* manufacturing the confound it exists to remove. Hashing more cannot fix it, because the
|
|
66
|
+
* instructions and the game state are the same string. See `diagnose`. */
|
|
67
|
+
export declare function fromRequest(kwargs: Record<string, unknown>, endpoint?: string): string;
|
|
68
|
+
/** True when a set of observations can support a within-scaffold model comparison.
|
|
69
|
+
*
|
|
70
|
+
* Requires exactly one KNOWN scaffold. An unknown ("") fingerprint disqualifies rather than
|
|
71
|
+
* being ignored: treating "we could not tell" as "the same as the others" is how a
|
|
72
|
+
* confounded comparison gets published as a clean one. */
|
|
73
|
+
export declare function eligibleForPairing(fingerprints: ReadonlyArray<string | undefined>): boolean;
|
|
74
|
+
export type { Any };
|
package/dist/scaffold.js
ADDED
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
// Scaffold fingerprinting: what stays the same when you swap the model.
|
|
2
|
+
// Mirrors sdk/python/pyyol/scaffold.py — see that file for the full rationale, and
|
|
3
|
+
// sdk/conformance/scaffold.json for the fixtures that hold the two to the same bytes.
|
|
4
|
+
//
|
|
5
|
+
// # The problem this exists to solve
|
|
6
|
+
//
|
|
7
|
+
// "Claude is better than GPT" is not a claim our data could support before this. Every
|
|
8
|
+
// Pyyol match confounds two things: the MODEL a developer chose and the HARNESS they wrote
|
|
9
|
+
// around it — the system prompt, the tools, the sampling settings, how the game state is
|
|
10
|
+
// framed. A strong agent on a weak model beats a weak agent on a strong model, and the
|
|
11
|
+
// leaderboard cannot tell you which happened.
|
|
12
|
+
//
|
|
13
|
+
// The clean way out is a PAIRED comparison: the same harness, run with model A and with
|
|
14
|
+
// model B. Then the harness cancels and the difference is the model.
|
|
15
|
+
//
|
|
16
|
+
// # What a fingerprint is, and what it is not
|
|
17
|
+
//
|
|
18
|
+
// A stable id for "the scaffold", derived from what the SDK observes on each model call
|
|
19
|
+
// and deliberately EXCLUDING the model name. Excluding the model is the whole point: with
|
|
20
|
+
// the model in the hash, every model would get its own scaffold id and nothing could ever
|
|
21
|
+
// be paired.
|
|
22
|
+
//
|
|
23
|
+
// It is not compared across developers, and it is not a way to read anyone's prompt — the
|
|
24
|
+
// system prompt enters as a digest, so the fingerprint proves "same prompt" without
|
|
25
|
+
// revealing it.
|
|
26
|
+
import { createHash } from "node:crypto";
|
|
27
|
+
/** Bump only for a change that intentionally invalidates existing fingerprints: it is part
|
|
28
|
+
* of the hashed payload, so a bump splits every agent's history into a new epoch. */
|
|
29
|
+
export const SCAFFOLD_VERSION = "pyyol-scaffold-v1";
|
|
30
|
+
/** Sampling parameters that change how a model behaves and are therefore part of the
|
|
31
|
+
* scaffold. Anything not listed is ignored, so a provider adding an unrelated field does
|
|
32
|
+
* not silently split every agent's history.
|
|
33
|
+
*
|
|
34
|
+
* `model` is absent on purpose. So are baseURL, apiKey, timeout and stream: the first two
|
|
35
|
+
* would make gateway routing look like a new scaffold, and the last two do not affect what
|
|
36
|
+
* the model decides. */
|
|
37
|
+
export const SAMPLING_KEYS = [
|
|
38
|
+
"frequency_penalty",
|
|
39
|
+
"max_completion_tokens",
|
|
40
|
+
"max_output_tokens",
|
|
41
|
+
"max_tokens",
|
|
42
|
+
"presence_penalty",
|
|
43
|
+
"reasoning_effort",
|
|
44
|
+
"seed",
|
|
45
|
+
"stop",
|
|
46
|
+
"stop_sequences",
|
|
47
|
+
"temperature",
|
|
48
|
+
"thinking",
|
|
49
|
+
"top_k",
|
|
50
|
+
"top_p",
|
|
51
|
+
"verbosity",
|
|
52
|
+
];
|
|
53
|
+
function isRecord(v) {
|
|
54
|
+
return typeof v === "object" && v !== null && !Array.isArray(v);
|
|
55
|
+
}
|
|
56
|
+
/** Canonical number formatting, shared with the Python SDK.
|
|
57
|
+
*
|
|
58
|
+
* The obvious approach diverges: `String(1.0)` is "1" in JavaScript while Python's
|
|
59
|
+
* `str(1.0)` is "1.0", so an agent using temperature=1.0 would fingerprint differently in
|
|
60
|
+
* the two SDKs and its history would split at the language boundary for no reason. Six
|
|
61
|
+
* significant figures is what both can agree on exactly (Python's "%.6g"). */
|
|
62
|
+
function num(v) {
|
|
63
|
+
if (typeof v === "boolean")
|
|
64
|
+
return v ? "true" : "false";
|
|
65
|
+
if (!Number.isFinite(v))
|
|
66
|
+
return "0";
|
|
67
|
+
if (Number.isInteger(v))
|
|
68
|
+
return String(v);
|
|
69
|
+
// toPrecision(6) then back through Number strips the trailing zeros that %.6g also drops.
|
|
70
|
+
return String(Number(v.toPrecision(6)));
|
|
71
|
+
}
|
|
72
|
+
function scalar(v) {
|
|
73
|
+
if (v === null || v === undefined)
|
|
74
|
+
return "";
|
|
75
|
+
if (typeof v === "number" || typeof v === "boolean")
|
|
76
|
+
return num(v);
|
|
77
|
+
if (typeof v === "string")
|
|
78
|
+
return v;
|
|
79
|
+
if (Array.isArray(v))
|
|
80
|
+
return "[" + v.map(scalar).join(",") + "]";
|
|
81
|
+
if (isRecord(v)) {
|
|
82
|
+
// A structured value (Anthropic's `thinking: {type, budget_tokens}`). Sorted so key
|
|
83
|
+
// order in the developer's literal cannot change the fingerprint.
|
|
84
|
+
const keys = Object.keys(v).sort();
|
|
85
|
+
return "{" + keys.map((k) => `${k}=${scalar(v[k])}`).join(";") + "}";
|
|
86
|
+
}
|
|
87
|
+
return String(v);
|
|
88
|
+
}
|
|
89
|
+
/** Flatten a message content field to text.
|
|
90
|
+
*
|
|
91
|
+
* Only text parts contribute: an image's bytes are not the scaffold, and hashing them
|
|
92
|
+
* would make the fingerprint depend on the board screenshot of the moment. */
|
|
93
|
+
function textOf(content) {
|
|
94
|
+
if (content === null || content === undefined)
|
|
95
|
+
return "";
|
|
96
|
+
if (typeof content === "string")
|
|
97
|
+
return content;
|
|
98
|
+
if (Array.isArray(content))
|
|
99
|
+
return content.map(textOf).join("");
|
|
100
|
+
if (isRecord(content)) {
|
|
101
|
+
const t = content.text;
|
|
102
|
+
return typeof t === "string" ? t : "";
|
|
103
|
+
}
|
|
104
|
+
return "";
|
|
105
|
+
}
|
|
106
|
+
function digest(text) {
|
|
107
|
+
return createHash("sha256").update(text, "utf8").digest("hex");
|
|
108
|
+
}
|
|
109
|
+
/** The fingerprint components observable in one provider request.
|
|
110
|
+
*
|
|
111
|
+
* Never throws: a fingerprinting problem must not break a developer's model call. */
|
|
112
|
+
export function extract(kwargs, endpoint = "") {
|
|
113
|
+
const roles = [];
|
|
114
|
+
const systemParts = [];
|
|
115
|
+
// Anthropic carries the system prompt as a top-level argument rather than a message.
|
|
116
|
+
// Treated as a leading system role so the two providers produce comparable shapes.
|
|
117
|
+
if (kwargs.system !== undefined && kwargs.system !== null) {
|
|
118
|
+
const text = textOf(kwargs.system);
|
|
119
|
+
if (text) {
|
|
120
|
+
roles.push("system");
|
|
121
|
+
systemParts.push(text);
|
|
122
|
+
}
|
|
123
|
+
}
|
|
124
|
+
const messages = kwargs.messages ?? kwargs.input ?? [];
|
|
125
|
+
if (Array.isArray(messages)) {
|
|
126
|
+
for (const m of messages) {
|
|
127
|
+
if (!isRecord(m))
|
|
128
|
+
continue;
|
|
129
|
+
const role = typeof m.role === "string" ? m.role : "";
|
|
130
|
+
if (!role)
|
|
131
|
+
continue;
|
|
132
|
+
roles.push(role);
|
|
133
|
+
// "developer" is OpenAI's newer name for the system role; both are scaffold.
|
|
134
|
+
if (role === "system" || role === "developer")
|
|
135
|
+
systemParts.push(textOf(m.content));
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
const sampling = [];
|
|
139
|
+
for (const key of SAMPLING_KEYS) {
|
|
140
|
+
const v = kwargs[key];
|
|
141
|
+
if (v !== undefined && v !== null)
|
|
142
|
+
sampling.push(`${key}=${scalar(v)}`);
|
|
143
|
+
}
|
|
144
|
+
const tools = [];
|
|
145
|
+
const declared = kwargs.tools;
|
|
146
|
+
if (Array.isArray(declared)) {
|
|
147
|
+
for (const t of declared) {
|
|
148
|
+
let name = "";
|
|
149
|
+
if (isRecord(t)) {
|
|
150
|
+
// OpenAI nests the name under `function`; Anthropic puts it at the top level.
|
|
151
|
+
if (isRecord(t.function) && typeof t.function.name === "string")
|
|
152
|
+
name = t.function.name;
|
|
153
|
+
if (!name && typeof t.name === "string")
|
|
154
|
+
name = t.name;
|
|
155
|
+
}
|
|
156
|
+
if (name)
|
|
157
|
+
tools.push(name);
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
const out = {
|
|
161
|
+
client: endpoint,
|
|
162
|
+
roles: roles.join(","),
|
|
163
|
+
sampling: sampling.join(";"),
|
|
164
|
+
// Sorted + deduped: declaring the same tools in a different order is the same scaffold.
|
|
165
|
+
tools: [...new Set(tools)].sort().join(","),
|
|
166
|
+
};
|
|
167
|
+
if (systemParts.length)
|
|
168
|
+
out.system = digest(systemParts.join("\n"));
|
|
169
|
+
return out;
|
|
170
|
+
}
|
|
171
|
+
/** The exact string that gets hashed.
|
|
172
|
+
*
|
|
173
|
+
* Specified rather than incidental: the Python SDK builds the same string and a shared
|
|
174
|
+
* conformance fixture checks both against the same expected fingerprints. Keys are emitted
|
|
175
|
+
* in a FIXED order (not sorted, not insertion order) so neither language's map iteration
|
|
176
|
+
* can affect the result, and an absent component is omitted rather than emitted empty — so
|
|
177
|
+
* adding a component later does not change the fingerprint of requests that never had one. */
|
|
178
|
+
export function canonical(c) {
|
|
179
|
+
const order = ["client", "roles", "sampling", "tools", "system"];
|
|
180
|
+
const lines = [SCAFFOLD_VERSION];
|
|
181
|
+
for (const key of order) {
|
|
182
|
+
const val = c[key];
|
|
183
|
+
if (val)
|
|
184
|
+
lines.push(`${key}=${val}`);
|
|
185
|
+
}
|
|
186
|
+
return lines.join("\n");
|
|
187
|
+
}
|
|
188
|
+
/** Short, prefixed id for a scaffold. 16 hex chars of SHA-256 (64 bits).
|
|
189
|
+
*
|
|
190
|
+
* Short enough to read in a UI and group by in SQL. Collisions are irrelevant here in a way
|
|
191
|
+
* they would not be for a security token: fingerprints are only compared WITHIN one agent's
|
|
192
|
+
* own history, so the space that must stay distinct is a handful of harness versions. */
|
|
193
|
+
export function fingerprint(c) {
|
|
194
|
+
return "sc_" + digest(canonical(c)).slice(0, 16);
|
|
195
|
+
}
|
|
196
|
+
/** Why a request could not be fingerprinted, as a stable CODE rather than prose.
|
|
197
|
+
*
|
|
198
|
+
* A code on the wire and prose at the point of reading, deliberately. The reason repeats on
|
|
199
|
+
* every decision of every non-qualifying agent, so shipping and storing the sentence would
|
|
200
|
+
* duplicate it thousands of times per match. A code is also aggregatable — "how many agents
|
|
201
|
+
* are ineligible, and why" is a question worth being able to ask — and its wording can change
|
|
202
|
+
* later without a migration. */
|
|
203
|
+
export const ISSUE_NO_SYSTEM_PROMPT = "no_system_prompt";
|
|
204
|
+
export const ISSUE_NO_MESSAGES = "no_messages";
|
|
205
|
+
/** Prose for each code. Read by the CLI and the dev-facing trace; never stored. */
|
|
206
|
+
export const ISSUE_EXPLANATIONS = {
|
|
207
|
+
[ISSUE_NO_SYSTEM_PROMPT]: "This request's instructions live in the user turn, mixed with the game state, where " +
|
|
208
|
+
"they cannot be told apart from it. Move your standing instructions into a system " +
|
|
209
|
+
"message to make this agent eligible for paired model comparison.",
|
|
210
|
+
[ISSUE_NO_MESSAGES]: "No messages were observed on this request, so there was nothing to fingerprint.",
|
|
211
|
+
};
|
|
212
|
+
/** Code for why this request yields no usable fingerprint, or "" when it does. */
|
|
213
|
+
export function issue(kwargs, endpoint = "") {
|
|
214
|
+
let c;
|
|
215
|
+
try {
|
|
216
|
+
c = extract(kwargs, endpoint);
|
|
217
|
+
}
|
|
218
|
+
catch {
|
|
219
|
+
return ISSUE_NO_MESSAGES;
|
|
220
|
+
}
|
|
221
|
+
if (!c.roles)
|
|
222
|
+
return ISSUE_NO_MESSAGES;
|
|
223
|
+
if (!c.system)
|
|
224
|
+
return ISSUE_NO_SYSTEM_PROMPT;
|
|
225
|
+
return "";
|
|
226
|
+
}
|
|
227
|
+
/** Human-readable reason for an issue code, or "" for no issue / an unknown code. */
|
|
228
|
+
export function explain(code) {
|
|
229
|
+
return ISSUE_EXPLANATIONS[code] ?? "";
|
|
230
|
+
}
|
|
231
|
+
/** Prose reason this request cannot be fingerprinted, or "" when it can.
|
|
232
|
+
*
|
|
233
|
+
* Convenience for local developer output; the wire carries `issue` codes. */
|
|
234
|
+
export function diagnose(kwargs, endpoint = "") {
|
|
235
|
+
return explain(issue(kwargs, endpoint));
|
|
236
|
+
}
|
|
237
|
+
/** Fingerprint one provider request, or "" when it cannot be fingerprinted.
|
|
238
|
+
*
|
|
239
|
+
* An empty result means "unknown", never a hash of nothing — a fingerprint shared by every
|
|
240
|
+
* request that failed to yield components would silently pool unrelated scaffolds into one
|
|
241
|
+
* bogus epoch and publish it as a controlled comparison.
|
|
242
|
+
*
|
|
243
|
+
* A SYSTEM PROMPT IS REQUIRED, and this is the sharp edge of the whole design. Without one,
|
|
244
|
+
* the hashable surface is `client` + `roles` + sampling — none of which move when the
|
|
245
|
+
* developer rewrites the instructions they actually steer the model with, because those
|
|
246
|
+
* instructions sit in a user message alongside the game state.
|
|
247
|
+
*
|
|
248
|
+
* That is worse than having no fingerprint. It is a FALSE CERTIFICATE: an agent could
|
|
249
|
+
* replace its entire strategy prompt mid-season, keep reporting the same scaffold id, and
|
|
250
|
+
* have the improvement attributed to whatever model it swapped to — the fingerprint would be
|
|
251
|
+
* manufacturing the confound it exists to remove. Hashing more cannot fix it, because the
|
|
252
|
+
* instructions and the game state are the same string. See `diagnose`. */
|
|
253
|
+
export function fromRequest(kwargs, endpoint = "") {
|
|
254
|
+
let c;
|
|
255
|
+
try {
|
|
256
|
+
c = extract(kwargs, endpoint);
|
|
257
|
+
}
|
|
258
|
+
catch {
|
|
259
|
+
return ""; // fingerprinting must never break a model call
|
|
260
|
+
}
|
|
261
|
+
if (!c.system)
|
|
262
|
+
return "";
|
|
263
|
+
return fingerprint(c);
|
|
264
|
+
}
|
|
265
|
+
/** True when a set of observations can support a within-scaffold model comparison.
|
|
266
|
+
*
|
|
267
|
+
* Requires exactly one KNOWN scaffold. An unknown ("") fingerprint disqualifies rather than
|
|
268
|
+
* being ignored: treating "we could not tell" as "the same as the others" is how a
|
|
269
|
+
* confounded comparison gets published as a clean one. */
|
|
270
|
+
export function eligibleForPairing(fingerprints) {
|
|
271
|
+
if (!fingerprints.length)
|
|
272
|
+
return false;
|
|
273
|
+
if (fingerprints.some((f) => !f))
|
|
274
|
+
return false;
|
|
275
|
+
return new Set(fingerprints).size === 1;
|
|
276
|
+
}
|
package/dist/telemetry.d.ts
CHANGED
|
@@ -37,7 +37,18 @@ export interface MoveUsage {
|
|
|
37
37
|
total_tokens: number;
|
|
38
38
|
reasoning_tokens?: number;
|
|
39
39
|
cached_tokens?: number;
|
|
40
|
+
cached_write_tokens?: number;
|
|
40
41
|
estimated_cost?: number;
|
|
42
|
+
/** Model calls made to reach this ONE decision. */
|
|
43
|
+
model_calls?: number;
|
|
44
|
+
/** Latency of each individual call, so a distribution is recoverable and not just a mean. */
|
|
45
|
+
call_latencies_ms?: number[];
|
|
46
|
+
/** Fingerprint of the harness this decision ran under, with the model excluded. */
|
|
47
|
+
scaffold?: string;
|
|
48
|
+
/** Set only when the fingerprint changed mid-turn, which makes the agent unpairable. */
|
|
49
|
+
scaffold_unstable?: boolean;
|
|
50
|
+
/** Code for why no fingerprint was produced, when none was — so the developer can act. */
|
|
51
|
+
scaffold_issue?: string;
|
|
41
52
|
model?: string | string[];
|
|
42
53
|
provider?: string | string[];
|
|
43
54
|
}
|
|
@@ -48,7 +59,9 @@ export interface UsageAdd {
|
|
|
48
59
|
completionTokens?: number;
|
|
49
60
|
reasoningTokens?: number;
|
|
50
61
|
cachedTokens?: number;
|
|
62
|
+
cachedWriteTokens?: number;
|
|
51
63
|
estimatedCost?: number;
|
|
64
|
+
latencyMs?: number;
|
|
52
65
|
}
|
|
53
66
|
/** Sums token usage + cost across every model call within a single turn. */
|
|
54
67
|
export declare class UsageAccumulator {
|
|
@@ -56,13 +69,24 @@ export declare class UsageAccumulator {
|
|
|
56
69
|
completionTokens: number;
|
|
57
70
|
reasoningTokens: number;
|
|
58
71
|
cachedTokens: number;
|
|
72
|
+
cachedWriteTokens: number;
|
|
59
73
|
estimatedCost: number;
|
|
60
74
|
calls: number;
|
|
75
|
+
readonly callLatenciesMs: number[];
|
|
76
|
+
scaffold: string;
|
|
77
|
+
scaffoldUnstable: boolean;
|
|
78
|
+
scaffoldIssue: string;
|
|
61
79
|
readonly models: string[];
|
|
62
80
|
readonly providers: string[];
|
|
63
81
|
matchId: string;
|
|
64
82
|
turn: number;
|
|
65
83
|
add(u: UsageAdd): void;
|
|
84
|
+
/** Record a scaffold fingerprint seen on one call this turn.
|
|
85
|
+
*
|
|
86
|
+
* An empty fingerprint means "could not tell" and is ignored rather than treated as a
|
|
87
|
+
* distinct scaffold: a failure to fingerprint is not evidence that the harness changed,
|
|
88
|
+
* and counting it as such would mark honest agents unstable. */
|
|
89
|
+
observeScaffold(fp: string, issue?: string): void;
|
|
66
90
|
get totalTokens(): number;
|
|
67
91
|
get empty(): boolean;
|
|
68
92
|
/** The `usage` block attached to a move — matches the arena's TokenUsage decode
|
package/dist/telemetry.js
CHANGED
|
@@ -95,9 +95,32 @@ export class UsageAccumulator {
|
|
|
95
95
|
promptTokens = 0;
|
|
96
96
|
completionTokens = 0;
|
|
97
97
|
reasoningTokens = 0;
|
|
98
|
+
// Cache READS (0.1x input on Anthropic) and cache WRITES (1.25x) are separate numbers
|
|
99
|
+
// because they are separate prices pointing opposite ways. Summing them into one
|
|
100
|
+
// "cached" figure makes the cost unrecoverable from what we stored.
|
|
98
101
|
cachedTokens = 0;
|
|
102
|
+
cachedWriteTokens = 0;
|
|
99
103
|
estimatedCost = 0;
|
|
100
104
|
calls = 0;
|
|
105
|
+
// Per-call latency, kept as a list so the platform can compute a distribution rather
|
|
106
|
+
// than only a mean. A turn's WALL time cannot separate one slow call from six quick
|
|
107
|
+
// ones, and "thinks for 40s" versus "makes 20 round trips" are different things about
|
|
108
|
+
// an agent.
|
|
109
|
+
callLatenciesMs = [];
|
|
110
|
+
// The SCAFFOLD this turn ran under: everything the developer built around the model,
|
|
111
|
+
// hashed with the model deliberately left out. It is what makes a paired model comparison
|
|
112
|
+
// possible — same scaffold, different model, so the harness cancels. See scaffold.ts.
|
|
113
|
+
scaffold = "";
|
|
114
|
+
// True when the fingerprint CHANGED between calls in one turn, which happens when variable
|
|
115
|
+
// game state sits in the system prompt. Such an agent cannot take part in a paired
|
|
116
|
+
// comparison, and saying so is more useful than silently keeping the first value seen.
|
|
117
|
+
scaffoldUnstable = false;
|
|
118
|
+
// CODE for why no fingerprint could be computed, when none could (see scaffold ISSUE_*).
|
|
119
|
+
// Carried to the developer rather than dropped: an agent that silently fails to qualify for
|
|
120
|
+
// the model board files a support ticket, where one told "move your instructions into a
|
|
121
|
+
// system message" fixes it in a line. A code rather than prose so it is small on the wire
|
|
122
|
+
// and aggregatable.
|
|
123
|
+
scaffoldIssue = "";
|
|
101
124
|
models = [];
|
|
102
125
|
providers = [];
|
|
103
126
|
// Turn context (for gateway attribution); set by runTurnUsage().
|
|
@@ -108,13 +131,33 @@ export class UsageAccumulator {
|
|
|
108
131
|
this.completionTokens += Math.max(0, Math.trunc(u.completionTokens ?? 0));
|
|
109
132
|
this.reasoningTokens += Math.max(0, Math.trunc(u.reasoningTokens ?? 0));
|
|
110
133
|
this.cachedTokens += Math.max(0, Math.trunc(u.cachedTokens ?? 0));
|
|
134
|
+
this.cachedWriteTokens += Math.max(0, Math.trunc(u.cachedWriteTokens ?? 0));
|
|
111
135
|
this.estimatedCost += Math.max(0, u.estimatedCost ?? 0);
|
|
112
136
|
this.calls += 1;
|
|
137
|
+
if (u.latencyMs)
|
|
138
|
+
this.callLatenciesMs.push(Math.max(0, Math.trunc(u.latencyMs)));
|
|
113
139
|
if (u.model && !this.models.includes(u.model))
|
|
114
140
|
this.models.push(u.model);
|
|
115
141
|
if (u.provider && !this.providers.includes(u.provider))
|
|
116
142
|
this.providers.push(u.provider);
|
|
117
143
|
}
|
|
144
|
+
/** Record a scaffold fingerprint seen on one call this turn.
|
|
145
|
+
*
|
|
146
|
+
* An empty fingerprint means "could not tell" and is ignored rather than treated as a
|
|
147
|
+
* distinct scaffold: a failure to fingerprint is not evidence that the harness changed,
|
|
148
|
+
* and counting it as such would mark honest agents unstable. */
|
|
149
|
+
observeScaffold(fp, issue = "") {
|
|
150
|
+
if (!fp) {
|
|
151
|
+
// First reason wins; later calls in the same turn usually repeat it.
|
|
152
|
+
if (issue && !this.scaffoldIssue)
|
|
153
|
+
this.scaffoldIssue = issue;
|
|
154
|
+
return;
|
|
155
|
+
}
|
|
156
|
+
if (!this.scaffold)
|
|
157
|
+
this.scaffold = fp;
|
|
158
|
+
else if (fp !== this.scaffold)
|
|
159
|
+
this.scaffoldUnstable = true;
|
|
160
|
+
}
|
|
118
161
|
get totalTokens() {
|
|
119
162
|
return this.promptTokens + this.completionTokens;
|
|
120
163
|
}
|
|
@@ -133,8 +176,27 @@ export class UsageAccumulator {
|
|
|
133
176
|
usage.reasoning_tokens = this.reasoningTokens;
|
|
134
177
|
if (this.cachedTokens)
|
|
135
178
|
usage.cached_tokens = this.cachedTokens;
|
|
179
|
+
if (this.cachedWriteTokens)
|
|
180
|
+
usage.cached_write_tokens = this.cachedWriteTokens;
|
|
136
181
|
if (this.estimatedCost)
|
|
137
182
|
usage.estimated_cost = Math.round(this.estimatedCost * 1e8) / 1e8;
|
|
183
|
+
// How many model calls this ONE decision took, and how long each took. A single
|
|
184
|
+
// aggregate hides the difference between an agent that answers in one call and one
|
|
185
|
+
// that runs a twelve-call chain to reach the same move — and that difference is most
|
|
186
|
+
// of what "efficient" means when comparing two agents on equal footing.
|
|
187
|
+
if (this.calls)
|
|
188
|
+
usage.model_calls = this.calls;
|
|
189
|
+
if (this.callLatenciesMs.length)
|
|
190
|
+
usage.call_latencies_ms = [...this.callLatenciesMs];
|
|
191
|
+
if (this.scaffold)
|
|
192
|
+
usage.scaffold = this.scaffold;
|
|
193
|
+
// Only sent when true. An absent flag and a false one mean the same thing, and shipping
|
|
194
|
+
// the false case on every move would be noise on the wire.
|
|
195
|
+
if (this.scaffoldUnstable)
|
|
196
|
+
usage.scaffold_unstable = true;
|
|
197
|
+
// Only when there is no fingerprint: with one, the code would be noise.
|
|
198
|
+
if (!this.scaffold && this.scaffoldIssue)
|
|
199
|
+
usage.scaffold_issue = this.scaffoldIssue;
|
|
138
200
|
if (this.models.length)
|
|
139
201
|
usage.model = this.models.length === 1 ? this.models[0] : this.models;
|
|
140
202
|
if (this.providers.length)
|
package/dist/version.d.ts
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
export declare const SDK_VERSION = "1.
|
|
1
|
+
export declare const SDK_VERSION = "1.10.0";
|
package/dist/version.js
CHANGED
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pyyol",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.10.0",
|
|
4
4
|
"description": "Official JS/TS SDK for pyyol — run AI game-playing agents locally over a WebSocket (Beta)",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|
|
@@ -32,6 +32,7 @@
|
|
|
32
32
|
"build": "npm run genversion && npm run clean && tsc -p tsconfig.build.json",
|
|
33
33
|
"build:test": "npm run genversion && tsc -p tsconfig.json",
|
|
34
34
|
"test": "npm run build:test && node --test dist/test/*.test.js",
|
|
35
|
+
"lint": "eslint .",
|
|
35
36
|
"prepack": "npm run build"
|
|
36
37
|
},
|
|
37
38
|
"keywords": [
|
|
@@ -57,7 +58,10 @@
|
|
|
57
58
|
"access": "public"
|
|
58
59
|
},
|
|
59
60
|
"devDependencies": {
|
|
61
|
+
"@eslint/js": "^9.39.5",
|
|
62
|
+
"@types/node": "^22.0.0",
|
|
63
|
+
"eslint": "^9.39.5",
|
|
60
64
|
"typescript": "^5.4.0",
|
|
61
|
-
"
|
|
65
|
+
"typescript-eslint": "^8.66.0"
|
|
62
66
|
}
|
|
63
67
|
}
|