@forwardimpact/libharness 0.1.22 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -201
- package/README.md +196 -80
- package/bin/fit-benchmark.js +44 -0
- package/bin/fit-harness.js +358 -0
- package/bin/fit-selfedit.js +165 -0
- package/bin/fit-trace.js +510 -0
- package/package.json +41 -11
- package/src/agent-runner.js +256 -0
- package/src/benchmark/apm-installer.js +207 -0
- package/src/benchmark/env-loader.js +158 -0
- package/src/benchmark/hook-env.js +40 -0
- package/src/benchmark/invariants.js +141 -0
- package/src/benchmark/judge.js +187 -0
- package/src/benchmark/npm-installer.js +87 -0
- package/src/benchmark/report.js +522 -0
- package/src/benchmark/result.js +127 -0
- package/src/benchmark/runner.js +583 -0
- package/src/benchmark/task-family.js +260 -0
- package/src/benchmark/workdir.js +298 -0
- package/src/commands/assert.js +153 -0
- package/src/commands/benchmark-definition.js +165 -0
- package/src/commands/benchmark-invariants.js +73 -0
- package/src/commands/benchmark-report.js +51 -0
- package/src/commands/benchmark-run.js +111 -0
- package/src/commands/by-discussion.js +94 -0
- package/src/commands/callback.js +119 -0
- package/src/commands/discuss.js +132 -0
- package/src/commands/facilitate.js +123 -0
- package/src/commands/output.js +36 -0
- package/src/commands/run.js +152 -0
- package/src/commands/supervise.js +136 -0
- package/src/commands/task-input.js +54 -0
- package/src/commands/tee.js +53 -0
- package/src/commands/trace.js +630 -0
- package/src/commands/work-tracker.js +35 -0
- package/src/cost.js +79 -0
- package/src/discuss-tools.js +173 -0
- package/src/discusser.js +394 -0
- package/src/events/github.js +161 -0
- package/src/facilitator.js +205 -0
- package/src/inbox-poller.js +81 -0
- package/src/index.js +72 -2
- package/src/judge.js +210 -0
- package/src/message-bus.js +118 -0
- package/src/orchestration-loop.js +330 -0
- package/src/orchestration-toolkit.js +441 -0
- package/src/orchestrator-helpers.js +23 -0
- package/src/profile-prompt.js +266 -0
- package/src/redaction.js +253 -0
- package/src/render/line-renderer.js +54 -0
- package/src/render/orchestrator-filter.js +19 -0
- package/src/render/palette.js +63 -0
- package/src/render/tool-hints.js +154 -0
- package/src/render/turn-renderer.js +96 -0
- package/src/reply-emitter.js +47 -0
- package/src/sequence-counter.js +21 -0
- package/src/signature-filter.js +27 -0
- package/src/supervisor.js +236 -0
- package/src/tee-writer.js +150 -0
- package/src/trace-collector.js +444 -0
- package/src/trace-github.js +473 -0
- package/src/trace-multi.js +101 -0
- package/src/trace-query.js +748 -0
- package/src/trace-render.js +211 -0
- package/src/trace-usage.js +249 -0
- package/src/fixture/assertions.js +0 -42
- package/src/fixture/cache.js +0 -50
- package/src/fixture/eval.js +0 -146
- package/src/fixture/index.js +0 -9
- package/src/fixture/pathway.js +0 -451
- package/src/fixture/services.js +0 -56
- package/src/mock/clients.js +0 -135
- package/src/mock/config.js +0 -45
- package/src/mock/data.js +0 -46
- package/src/mock/fs.js +0 -111
- package/src/mock/grpc.js +0 -94
- package/src/mock/http.js +0 -60
- package/src/mock/index.js +0 -36
- package/src/mock/infra.js +0 -219
- package/src/mock/logger.js +0 -42
- package/src/mock/observer.js +0 -74
- package/src/mock/resource-index.js +0 -95
- package/src/mock/service-callbacks.js +0 -39
- package/src/mock/services.js +0 -79
- package/src/mock/spy.js +0 -44
- package/src/mock/storage.js +0 -118
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Text renderers for `fit-trace` query output.
|
|
3
|
+
*
|
|
4
|
+
* One named export per renderable verb. Each renderer accepts the query result
|
|
5
|
+
* plus `{multi, signatures}` and returns a string. `multi` controls
|
|
6
|
+
* source-attribution prefixing (`grep -H` convention); record-per-line
|
|
7
|
+
* renderers prepend `<basename>:`, block renderers emit `# <basename>` headers.
|
|
8
|
+
*
|
|
9
|
+
* Internal module — imported by `commands/trace.js` and tests by relative
|
|
10
|
+
* path, never re-exported from `src/index.js`.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
/** Collapse newlines/tabs in a value to a single-line, grep-friendly string. */
|
|
14
|
+
function oneLine(value) {
|
|
15
|
+
const str = typeof value === "string" ? value : JSON.stringify(value ?? null);
|
|
16
|
+
return str.replace(/[\r\n\t]+/g, " ").trim();
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/** Group records by their `source` field (multi-file path), preserving order. */
|
|
20
|
+
function groupBySource(records) {
|
|
21
|
+
const groups = new Map();
|
|
22
|
+
for (const record of records) {
|
|
23
|
+
const key = record.source ?? "";
|
|
24
|
+
if (!groups.has(key)) groups.set(key, []);
|
|
25
|
+
groups.get(key).push(record);
|
|
26
|
+
}
|
|
27
|
+
return groups;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Render record-per-line output, prefixing each line with `<source>:` when
|
|
32
|
+
* multi-file. `lineOf` maps one record to its text line.
|
|
33
|
+
* @param {object[]} records
|
|
34
|
+
* @param {(record: object) => string} lineOf
|
|
35
|
+
* @param {{multi: boolean}} opts
|
|
36
|
+
* @returns {string}
|
|
37
|
+
*/
|
|
38
|
+
function renderLines(records, lineOf, { multi }) {
|
|
39
|
+
return records
|
|
40
|
+
.map((r) => (multi && r.source ? `${r.source}:${lineOf(r)}` : lineOf(r)))
|
|
41
|
+
.join("\n");
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/**
|
|
45
|
+
* Render a block per source. `blockOf` maps one record to a multi-line string;
|
|
46
|
+
* multi-file output separates groups with `# <source>` headers.
|
|
47
|
+
* @param {object[]} records
|
|
48
|
+
* @param {(record: object) => string} blockOf
|
|
49
|
+
* @param {{multi: boolean}} opts
|
|
50
|
+
* @returns {string}
|
|
51
|
+
*/
|
|
52
|
+
function renderBlocks(records, blockOf, { multi }) {
|
|
53
|
+
if (!multi) return records.map(blockOf).join("\n");
|
|
54
|
+
const out = [];
|
|
55
|
+
for (const [source, group] of groupBySource(records)) {
|
|
56
|
+
out.push(`# ${source}`);
|
|
57
|
+
out.push(...group.map(blockOf));
|
|
58
|
+
}
|
|
59
|
+
return out.join("\n");
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/** `[turnIdx] <Tool> <toolUseId>` / ` in:` / ` out:` per block. */
|
|
63
|
+
export function renderToolCalls(records, opts = {}) {
|
|
64
|
+
return renderBlocks(
|
|
65
|
+
records,
|
|
66
|
+
(r) => {
|
|
67
|
+
const head = `[${r.turnIndex}] ${r.name} ${r.toolUseId}`;
|
|
68
|
+
const input = ` in: ${oneLine(r.input)}`;
|
|
69
|
+
const out = ` out: ${
|
|
70
|
+
r.result ? oneLine(r.result.content) : "(no result)"
|
|
71
|
+
}`;
|
|
72
|
+
return [head, input, out].join("\n");
|
|
73
|
+
},
|
|
74
|
+
opts,
|
|
75
|
+
);
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/** `[turnIdx] <command>` per line, newlines escaped. */
|
|
79
|
+
export function renderCommands(records, opts = {}) {
|
|
80
|
+
return renderLines(
|
|
81
|
+
records,
|
|
82
|
+
(r) => `[${r.turnIndex}] ${oneLine(r.command)}`,
|
|
83
|
+
opts,
|
|
84
|
+
);
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
/** `<count>\t<path>` frequency-sorted. */
|
|
88
|
+
export function renderPaths(records, opts = {}) {
|
|
89
|
+
return renderLines(records, (r) => `${r.count}\t${r.path}`, opts);
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/** Metadata header, per-row metrics, then Tool and Path delta tables. */
|
|
93
|
+
export function renderCompare(result) {
|
|
94
|
+
const { a, b, toolDelta, pathDelta } = result;
|
|
95
|
+
const part = (p) => (p == null ? "(none)" : p);
|
|
96
|
+
const lines = [];
|
|
97
|
+
lines.push(
|
|
98
|
+
`A: ${a.metadata.caseName} / ${part(a.metadata.participant)}${
|
|
99
|
+
a.metadata.marker ? ` ${a.metadata.marker}` : ""
|
|
100
|
+
}`,
|
|
101
|
+
);
|
|
102
|
+
lines.push(
|
|
103
|
+
`B: ${b.metadata.caseName} / ${part(b.metadata.participant)}${
|
|
104
|
+
b.metadata.marker ? ` ${b.metadata.marker}` : ""
|
|
105
|
+
}`,
|
|
106
|
+
);
|
|
107
|
+
lines.push("");
|
|
108
|
+
lines.push(`turns | ${a.turnCount} | ${b.turnCount}`);
|
|
109
|
+
lines.push(`tools | ${a.tools.length} | ${b.tools.length}`);
|
|
110
|
+
lines.push(`paths | ${a.pathCount} | ${b.pathCount}`);
|
|
111
|
+
lines.push(`cost | ${a.cost} | ${b.cost}`);
|
|
112
|
+
lines.push("");
|
|
113
|
+
lines.push("Tool | A | B | Δ");
|
|
114
|
+
for (const d of toolDelta) {
|
|
115
|
+
lines.push(`${d.tool} | ${d.a} | ${d.b} | ${d.diff}`);
|
|
116
|
+
}
|
|
117
|
+
lines.push("");
|
|
118
|
+
lines.push("Path | A | B | Δ");
|
|
119
|
+
for (const d of pathDelta) {
|
|
120
|
+
lines.push(`${d.path} | ${d.a} | ${d.b} | ${d.diff}`);
|
|
121
|
+
}
|
|
122
|
+
return lines.join("\n");
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/** `Tool | Turns | In | Out | Share` sorted Share desc. */
|
|
126
|
+
export function renderStatsByTool(result) {
|
|
127
|
+
const lines = ["Tool | Turns | In | Out | Share"];
|
|
128
|
+
for (const b of result.perTool) {
|
|
129
|
+
lines.push(
|
|
130
|
+
`${b.tool} | ${b.turns} | ${Math.round(b.inputTokens)} | ${Math.round(
|
|
131
|
+
b.outputTokens,
|
|
132
|
+
)} | ${b.costShare.toFixed(4)}`,
|
|
133
|
+
);
|
|
134
|
+
}
|
|
135
|
+
return lines.join("\n");
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/** Totals block only. */
|
|
139
|
+
export function renderStatsSummary(result) {
|
|
140
|
+
const t = result.totals;
|
|
141
|
+
return [
|
|
142
|
+
`inputTokens: ${t.inputTokens}`,
|
|
143
|
+
`outputTokens: ${t.outputTokens}`,
|
|
144
|
+
`cacheReadInputTokens: ${t.cacheReadInputTokens}`,
|
|
145
|
+
`cacheCreationInputTokens: ${t.cacheCreationInputTokens}`,
|
|
146
|
+
`totalCostUsd: ${t.totalCostUsd}`,
|
|
147
|
+
`durationMs: ${t.durationMs}`,
|
|
148
|
+
].join("\n");
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
/** `[turnIdx] <prefix>: <excerpt>` per match. */
|
|
152
|
+
export function renderSearch(records, opts = {}) {
|
|
153
|
+
const lines = [];
|
|
154
|
+
for (const hit of records) {
|
|
155
|
+
const idx = hit.turn?.index;
|
|
156
|
+
const prefix = multiPrefix(hit, opts);
|
|
157
|
+
for (const match of hit.matches ?? []) {
|
|
158
|
+
lines.push(`${prefix}[${idx}] ${oneLine(match)}`);
|
|
159
|
+
}
|
|
160
|
+
}
|
|
161
|
+
return lines.join("\n");
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/** Source prefix for a multi-file record (search/default), or "". */
|
|
165
|
+
function multiPrefix(record, { multi }) {
|
|
166
|
+
return multi && record.source ? `${record.source}:` : "";
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/**
|
|
170
|
+
* Default renderer for every other renderable verb: one record per block,
|
|
171
|
+
* fields rendered as `key: value` lines (no JSON braces or quotes, so the
|
|
172
|
+
* default output is grep/awk-friendly and does not parse as JSON). Nested
|
|
173
|
+
* values are collapsed to a single grep-friendly line. Multi-file output
|
|
174
|
+
* separates source groups with `# <source>` headers (`renderBlocks`
|
|
175
|
+
* convention).
|
|
176
|
+
* @param {object[]|object} result
|
|
177
|
+
* @param {{multi: boolean}} opts
|
|
178
|
+
* @returns {string}
|
|
179
|
+
*/
|
|
180
|
+
export function renderDefault(result, opts = {}) {
|
|
181
|
+
const records = Array.isArray(result) ? result : [result];
|
|
182
|
+
return renderBlocks(records, (r) => recordBlock(stripSource(r)), opts);
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/**
|
|
186
|
+
* Render one record as `key: value` lines. Scalars render verbatim; objects
|
|
187
|
+
* and arrays collapse to a single line via `oneLine`. A non-object record
|
|
188
|
+
* (string/number) renders as its own single line.
|
|
189
|
+
* @param {*} record
|
|
190
|
+
* @returns {string}
|
|
191
|
+
*/
|
|
192
|
+
function recordBlock(record) {
|
|
193
|
+
if (record == null || typeof record !== "object" || Array.isArray(record)) {
|
|
194
|
+
return oneLine(record);
|
|
195
|
+
}
|
|
196
|
+
return Object.entries(record)
|
|
197
|
+
.map(([key, value]) => {
|
|
198
|
+
const scalar = value == null || typeof value !== "object";
|
|
199
|
+
return `${key}: ${scalar ? String(value) : oneLine(value)}`;
|
|
200
|
+
})
|
|
201
|
+
.join("\n");
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/** Drop the orchestrator-injected `source` field before textifying. */
|
|
205
|
+
function stripSource(record) {
|
|
206
|
+
if (record == null || typeof record !== "object" || Array.isArray(record)) {
|
|
207
|
+
return record;
|
|
208
|
+
}
|
|
209
|
+
const { source, ...rest } = record;
|
|
210
|
+
return rest;
|
|
211
|
+
}
|
|
@@ -0,0 +1,249 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Token-usage accounting for structured trace documents.
|
|
3
|
+
*
|
|
4
|
+
* `stats()` reports totals that name their population: result-event sums when
|
|
5
|
+
* the trace carries them (authoritative), the per-message fallback otherwise,
|
|
6
|
+
* or the carried last-wins summary for a pre-change document. These helpers
|
|
7
|
+
* compute the per-message accounting and surface any divergence against the
|
|
8
|
+
* result-event sums so a mismatch is reported, never silently absorbed.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
/** Zero-valued token usage, used as the carried-document fallback. */
|
|
12
|
+
export const ZERO_USAGE = {
|
|
13
|
+
inputTokens: 0,
|
|
14
|
+
outputTokens: 0,
|
|
15
|
+
cacheReadInputTokens: 0,
|
|
16
|
+
cacheCreationInputTokens: 0,
|
|
17
|
+
};
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Per-stream-event breakdown for a pre-change document, labeled as carried —
|
|
21
|
+
* old documents lack message identity, so rows stay keyed by turn index.
|
|
22
|
+
* @param {object[]} turns
|
|
23
|
+
* @returns {object[]}
|
|
24
|
+
*/
|
|
25
|
+
export function carriedPerTurn(turns) {
|
|
26
|
+
const perTurn = [];
|
|
27
|
+
for (const turn of turns) {
|
|
28
|
+
if (turn.role !== "assistant" || !turn.usage) continue;
|
|
29
|
+
perTurn.push({
|
|
30
|
+
index: turn.index,
|
|
31
|
+
inputTokens: turn.usage.inputTokens ?? 0,
|
|
32
|
+
outputTokens: turn.usage.outputTokens ?? 0,
|
|
33
|
+
cacheReadInputTokens: turn.usage.cacheReadInputTokens ?? 0,
|
|
34
|
+
cacheCreationInputTokens: turn.usage.cacheCreationInputTokens ?? 0,
|
|
35
|
+
population: "carried-document-per-turn",
|
|
36
|
+
});
|
|
37
|
+
}
|
|
38
|
+
return perTurn;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Whether a structured-document version predates per-message accounting
|
|
43
|
+
* (1.2.0). A trace with no version (collected by this build from NDJSON) is
|
|
44
|
+
* not pre-change. Compares numeric version parts so 1.10.0 reads as post-change.
|
|
45
|
+
* @param {string|undefined|null} version
|
|
46
|
+
* @returns {boolean}
|
|
47
|
+
*/
|
|
48
|
+
export function isPreChangeDoc(version) {
|
|
49
|
+
if (typeof version !== "string") return false;
|
|
50
|
+
const [major = 0, minor = 0] = version
|
|
51
|
+
.split(".")
|
|
52
|
+
.map((part) => parseInt(part, 10) || 0);
|
|
53
|
+
if (major !== 1) return major < 1;
|
|
54
|
+
// Per-message accounting arrived in 1.2.0; any 1.2.x is post-change.
|
|
55
|
+
return minor < 2;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Account assistant usage once per API message. Turns are grouped by
|
|
60
|
+
* `messageId` (a null id is its own singleton message); per message the
|
|
61
|
+
* field-wise max across its snapshots is taken — order-insensitive, equal to
|
|
62
|
+
* the single value when a message's duplicate snapshots are byte-identical
|
|
63
|
+
* (zero residual against result-event sums), and a floor for output (the
|
|
64
|
+
* largest streaming snapshot, never an overstatement).
|
|
65
|
+
* @param {object[]} turns
|
|
66
|
+
* @returns {{perMessage: object[], totals: object}}
|
|
67
|
+
*/
|
|
68
|
+
export function perMessageUsage(turns) {
|
|
69
|
+
const byMessage = new Map();
|
|
70
|
+
let singletonSeq = 0;
|
|
71
|
+
|
|
72
|
+
for (const turn of turns) {
|
|
73
|
+
if (turn.role !== "assistant" || !turn.usage) continue;
|
|
74
|
+
const key = turn.messageId ?? `__null__${singletonSeq++}`;
|
|
75
|
+
accumulateMessage(byMessage, key, turn);
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
const totals = {
|
|
79
|
+
inputTokens: 0,
|
|
80
|
+
outputTokens: 0,
|
|
81
|
+
cacheReadInputTokens: 0,
|
|
82
|
+
cacheCreationInputTokens: 0,
|
|
83
|
+
};
|
|
84
|
+
const perMessage = [];
|
|
85
|
+
for (const row of byMessage.values()) {
|
|
86
|
+
totals.inputTokens += row.inputTokens;
|
|
87
|
+
totals.outputTokens += row.outputTokens;
|
|
88
|
+
totals.cacheReadInputTokens += row.cacheReadInputTokens;
|
|
89
|
+
totals.cacheCreationInputTokens += row.cacheCreationInputTokens;
|
|
90
|
+
perMessage.push({
|
|
91
|
+
...row,
|
|
92
|
+
outputIsStreamingSnapshot: true,
|
|
93
|
+
population: "api-message",
|
|
94
|
+
});
|
|
95
|
+
}
|
|
96
|
+
return { perMessage, totals };
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/**
|
|
100
|
+
* Fold one assistant turn's usage into its message bucket by field-wise max.
|
|
101
|
+
* @param {Map<string, object>} byMessage
|
|
102
|
+
* @param {string} key
|
|
103
|
+
* @param {object} turn
|
|
104
|
+
*/
|
|
105
|
+
function accumulateMessage(byMessage, key, turn) {
|
|
106
|
+
const u = turn.usage;
|
|
107
|
+
const prev = byMessage.get(key);
|
|
108
|
+
if (!prev) {
|
|
109
|
+
byMessage.set(key, {
|
|
110
|
+
messageId: turn.messageId ?? null,
|
|
111
|
+
inputTokens: u.inputTokens ?? 0,
|
|
112
|
+
outputTokens: u.outputTokens ?? 0,
|
|
113
|
+
cacheReadInputTokens: u.cacheReadInputTokens ?? 0,
|
|
114
|
+
cacheCreationInputTokens: u.cacheCreationInputTokens ?? 0,
|
|
115
|
+
});
|
|
116
|
+
return;
|
|
117
|
+
}
|
|
118
|
+
prev.inputTokens = Math.max(prev.inputTokens, u.inputTokens ?? 0);
|
|
119
|
+
prev.outputTokens = Math.max(prev.outputTokens, u.outputTokens ?? 0);
|
|
120
|
+
prev.cacheReadInputTokens = Math.max(
|
|
121
|
+
prev.cacheReadInputTokens,
|
|
122
|
+
u.cacheReadInputTokens ?? 0,
|
|
123
|
+
);
|
|
124
|
+
prev.cacheCreationInputTokens = Math.max(
|
|
125
|
+
prev.cacheCreationInputTokens,
|
|
126
|
+
u.cacheCreationInputTokens ?? 0,
|
|
127
|
+
);
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
/**
|
|
131
|
+
* Compare per-message sums against the result-event sums on the fields the
|
|
132
|
+
* spec guarantees parity for (input, cacheRead, cacheCreation — never output,
|
|
133
|
+
* which always diverges by mechanism 2). Returns the first divergent field as
|
|
134
|
+
* `{field, perMessageSum, resultEventSum}`, or null when all agree.
|
|
135
|
+
* @param {object} perMessageTotals
|
|
136
|
+
* @param {object} resultEventUsage
|
|
137
|
+
* @returns {object|null}
|
|
138
|
+
*/
|
|
139
|
+
export function computeDivergence(perMessageTotals, resultEventUsage) {
|
|
140
|
+
for (const field of [
|
|
141
|
+
"inputTokens",
|
|
142
|
+
"cacheReadInputTokens",
|
|
143
|
+
"cacheCreationInputTokens",
|
|
144
|
+
]) {
|
|
145
|
+
const perMessageSum = perMessageTotals[field] ?? 0;
|
|
146
|
+
const resultEventSum = resultEventUsage[field] ?? 0;
|
|
147
|
+
if (perMessageSum !== resultEventSum) {
|
|
148
|
+
return { field, perMessageSum, resultEventSum };
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
return null;
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/** Sentinel bucket name for assistant turns that ran no tool call. */
|
|
155
|
+
const NO_TOOL = "(no-tool)";
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* Attribute per-turn usage to per-tool buckets: each `tool_use` block gets an
|
|
159
|
+
* equal share of its host turn's usage; assistant turns with no `tool_use`
|
|
160
|
+
* block contribute full usage to the `(no-tool)` bucket.
|
|
161
|
+
* @param {object[]} turns
|
|
162
|
+
* @returns {{buckets: Map<string, {inputTokens: number, outputTokens: number}>, bucketTurns: Map<string, Set<number>>}}
|
|
163
|
+
*/
|
|
164
|
+
export function bucketUsageByTool(turns) {
|
|
165
|
+
const buckets = new Map();
|
|
166
|
+
const bucketTurns = new Map();
|
|
167
|
+
const ensure = (name) => {
|
|
168
|
+
if (!buckets.has(name)) {
|
|
169
|
+
buckets.set(name, { inputTokens: 0, outputTokens: 0 });
|
|
170
|
+
bucketTurns.set(name, new Set());
|
|
171
|
+
}
|
|
172
|
+
return buckets.get(name);
|
|
173
|
+
};
|
|
174
|
+
|
|
175
|
+
for (const turn of turns) {
|
|
176
|
+
if (turn.role !== "assistant" || !turn.usage) continue;
|
|
177
|
+
const input = turn.usage.inputTokens ?? 0;
|
|
178
|
+
const output = turn.usage.outputTokens ?? 0;
|
|
179
|
+
const toolBlocks = turn.content.filter((b) => b.type === "tool_use");
|
|
180
|
+
const targets = toolBlocks.length === 0 ? [NO_TOOL] : toolBlocks;
|
|
181
|
+
const shareIn = input / targets.length;
|
|
182
|
+
const shareOut = output / targets.length;
|
|
183
|
+
for (const target of targets) {
|
|
184
|
+
const name = typeof target === "string" ? target : target.name;
|
|
185
|
+
const bucket = ensure(name);
|
|
186
|
+
bucket.inputTokens += shareIn;
|
|
187
|
+
bucket.outputTokens += shareOut;
|
|
188
|
+
bucketTurns.get(name).add(turn.index);
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
return { buckets, bucketTurns };
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/**
|
|
195
|
+
* Scale per-tool buckets onto the headline totals so the input, output, and
|
|
196
|
+
* `costShare` columns each sum to the corresponding `totals` value (and 1.0)
|
|
197
|
+
* exactly, regardless of population (result-event-sum, per-message-fallback, or
|
|
198
|
+
* carried-document). The largest bucket absorbs the rounding residual on each
|
|
199
|
+
* axis (criterion-6 invariant).
|
|
200
|
+
* @param {Map<string, {inputTokens: number, outputTokens: number}>} buckets
|
|
201
|
+
* @param {Map<string, Set<number>>} bucketTurns
|
|
202
|
+
* @param {object} totals
|
|
203
|
+
* @returns {Array<{tool: string, turns: number, inputTokens: number, outputTokens: number, costShare: number}>}
|
|
204
|
+
*/
|
|
205
|
+
export function reconcileBucketsToTotals(buckets, bucketTurns, totals) {
|
|
206
|
+
const rawIn = sumField(buckets, "inputTokens");
|
|
207
|
+
const rawOut = sumField(buckets, "outputTokens");
|
|
208
|
+
const scaleIn = rawIn === 0 ? 0 : (totals.inputTokens ?? 0) / rawIn;
|
|
209
|
+
const scaleOut = rawOut === 0 ? 0 : (totals.outputTokens ?? 0) / rawOut;
|
|
210
|
+
for (const b of buckets.values()) {
|
|
211
|
+
b.inputTokens *= scaleIn;
|
|
212
|
+
b.outputTokens *= scaleOut;
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
const totalTokens =
|
|
216
|
+
sumField(buckets, "inputTokens") + sumField(buckets, "outputTokens");
|
|
217
|
+
const perTool = [...buckets.entries()].map(([tool, b]) => ({
|
|
218
|
+
tool,
|
|
219
|
+
turns: bucketTurns.get(tool).size,
|
|
220
|
+
inputTokens: b.inputTokens,
|
|
221
|
+
outputTokens: b.outputTokens,
|
|
222
|
+
costShare:
|
|
223
|
+
totalTokens === 0 ? 0 : (b.inputTokens + b.outputTokens) / totalTokens,
|
|
224
|
+
}));
|
|
225
|
+
perTool.sort(
|
|
226
|
+
(x, y) => y.costShare - x.costShare || x.tool.localeCompare(y.tool),
|
|
227
|
+
);
|
|
228
|
+
|
|
229
|
+
if (perTool.length > 0) {
|
|
230
|
+
const top = perTool[0];
|
|
231
|
+
const sum = (field) => perTool.reduce((s, r) => s + r[field], 0);
|
|
232
|
+
top.inputTokens += (totals.inputTokens ?? 0) - sum("inputTokens");
|
|
233
|
+
top.outputTokens += (totals.outputTokens ?? 0) - sum("outputTokens");
|
|
234
|
+
top.costShare += 1 - sum("costShare");
|
|
235
|
+
}
|
|
236
|
+
return perTool;
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
/**
|
|
240
|
+
* Sum one numeric field across every bucket value.
|
|
241
|
+
* @param {Map<string, object>} buckets
|
|
242
|
+
* @param {string} field
|
|
243
|
+
* @returns {number}
|
|
244
|
+
*/
|
|
245
|
+
function sumField(buckets, field) {
|
|
246
|
+
let total = 0;
|
|
247
|
+
for (const b of buckets.values()) total += b[field];
|
|
248
|
+
return total;
|
|
249
|
+
}
|
|
@@ -1,42 +0,0 @@
|
|
|
1
|
-
import assert from "node:assert";
|
|
2
|
-
|
|
3
|
-
/**
|
|
4
|
-
* Asserts that a function throws with a matching message
|
|
5
|
-
* @param {Function} fn - Function to test
|
|
6
|
-
* @param {RegExp|string} pattern - Pattern to match
|
|
7
|
-
* @param {string} message - Assertion message
|
|
8
|
-
*/
|
|
9
|
-
export function assertThrowsMessage(fn, pattern, message) {
|
|
10
|
-
assert.throws(
|
|
11
|
-
fn,
|
|
12
|
-
{ message: pattern instanceof RegExp ? pattern : new RegExp(pattern) },
|
|
13
|
-
message,
|
|
14
|
-
);
|
|
15
|
-
}
|
|
16
|
-
|
|
17
|
-
/**
|
|
18
|
-
* Asserts that an async function rejects with a matching message
|
|
19
|
-
* @param {Function} fn - Async function to test
|
|
20
|
-
* @param {RegExp|string} pattern - Pattern to match
|
|
21
|
-
* @param {string} message - Assertion message
|
|
22
|
-
*/
|
|
23
|
-
export async function assertRejectsMessage(fn, pattern, message) {
|
|
24
|
-
await assert.rejects(
|
|
25
|
-
fn,
|
|
26
|
-
{ message: pattern instanceof RegExp ? pattern : new RegExp(pattern) },
|
|
27
|
-
message,
|
|
28
|
-
);
|
|
29
|
-
}
|
|
30
|
-
|
|
31
|
-
/**
|
|
32
|
-
* Creates a deferred promise for async testing
|
|
33
|
-
* @returns {object} Object with promise, resolve, and reject
|
|
34
|
-
*/
|
|
35
|
-
export function createDeferred() {
|
|
36
|
-
let resolve, reject;
|
|
37
|
-
const promise = new Promise((res, rej) => {
|
|
38
|
-
resolve = res;
|
|
39
|
-
reject = rej;
|
|
40
|
-
});
|
|
41
|
-
return { promise, resolve, reject };
|
|
42
|
-
}
|
package/src/fixture/cache.js
DELETED
|
@@ -1,50 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Cross-file fixture caching helpers. The test runner currently executes one
|
|
3
|
-
* test file per process, but fixtures loaded inside a file (e.g. starter
|
|
4
|
-
* standard YAML via `createDataLoader().loadAllData(dir)`) are re-parsed for
|
|
5
|
-
* every `test(...)` case unless hoisted. See spec 620.
|
|
6
|
-
*/
|
|
7
|
-
|
|
8
|
-
const caches = new WeakMap();
|
|
9
|
-
const stringCaches = new Map();
|
|
10
|
-
|
|
11
|
-
/**
|
|
12
|
-
* Wraps an async factory so it is invoked at most once per unique key.
|
|
13
|
-
*
|
|
14
|
-
* @template T
|
|
15
|
-
* @param {string} key - Cache key. Using the same key returns the cached value.
|
|
16
|
-
* @param {() => Promise<T>} factory - Factory invoked on miss.
|
|
17
|
-
* @returns {Promise<T>}
|
|
18
|
-
*/
|
|
19
|
-
export async function memoizeAsync(key, factory) {
|
|
20
|
-
if (stringCaches.has(key)) return stringCaches.get(key);
|
|
21
|
-
const promise = Promise.resolve().then(factory);
|
|
22
|
-
stringCaches.set(key, promise);
|
|
23
|
-
try {
|
|
24
|
-
return await promise;
|
|
25
|
-
} catch (err) {
|
|
26
|
-
stringCaches.delete(key);
|
|
27
|
-
throw err;
|
|
28
|
-
}
|
|
29
|
-
}
|
|
30
|
-
|
|
31
|
-
/**
|
|
32
|
-
* Caches the result of `fn(subject)` keyed by identity of `subject`. Useful
|
|
33
|
-
* for expensive derivations over a frozen input object.
|
|
34
|
-
*
|
|
35
|
-
* @template S, T
|
|
36
|
-
* @param {S} subject
|
|
37
|
-
* @param {(subject: S) => T} fn
|
|
38
|
-
* @returns {T}
|
|
39
|
-
*/
|
|
40
|
-
export function memoizeOnSubject(subject, fn) {
|
|
41
|
-
if (!caches.has(subject)) caches.set(subject, fn(subject));
|
|
42
|
-
return caches.get(subject);
|
|
43
|
-
}
|
|
44
|
-
|
|
45
|
-
/**
|
|
46
|
-
* Clears all memoization caches. Only useful in self-tests of this helper.
|
|
47
|
-
*/
|
|
48
|
-
export function __resetMemoCaches() {
|
|
49
|
-
stringCaches.clear();
|
|
50
|
-
}
|
package/src/fixture/eval.js
DELETED
|
@@ -1,146 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Test helpers for libeval-style streams, tool-use messages, and traces.
|
|
3
|
-
*
|
|
4
|
-
* Before these helpers existed, every test file under libraries/libeval/test
|
|
5
|
-
* inlined its own copy of concludeMsg / redirectMsg / tellMsg / stripAnsi /
|
|
6
|
-
* collect / writeLines / buildTrace. See spec 620.
|
|
7
|
-
*/
|
|
8
|
-
|
|
9
|
-
/**
|
|
10
|
-
* Builds a tool-use message envelope as emitted by the agent SDK.
|
|
11
|
-
* Replaces per-file concludeMsg / redirectMsg / tellMsg / shareMsg helpers.
|
|
12
|
-
*
|
|
13
|
-
* @param {string} name - Tool name, e.g. "Conclude", "Redirect", "Tell", "Share".
|
|
14
|
-
* @param {object} input - Tool input payload.
|
|
15
|
-
* @param {object} [options]
|
|
16
|
-
* @param {string} [options.id] - Explicit tool_use id. Defaults to `${name.toLowerCase()}-1`.
|
|
17
|
-
* @returns {object} Assistant message with a single tool_use content block.
|
|
18
|
-
*/
|
|
19
|
-
export function createToolUseMsg(name, input, { id } = {}) {
|
|
20
|
-
return {
|
|
21
|
-
type: "assistant",
|
|
22
|
-
message: {
|
|
23
|
-
content: [
|
|
24
|
-
{
|
|
25
|
-
type: "tool_use",
|
|
26
|
-
id: id || `${name.toLowerCase()}-1`,
|
|
27
|
-
name,
|
|
28
|
-
input,
|
|
29
|
-
},
|
|
30
|
-
],
|
|
31
|
-
},
|
|
32
|
-
};
|
|
33
|
-
}
|
|
34
|
-
|
|
35
|
-
/**
|
|
36
|
-
* Builds an assistant text message envelope as emitted by the agent SDK.
|
|
37
|
-
* @param {string} text - Text content.
|
|
38
|
-
* @returns {object} Assistant message with a single text content block.
|
|
39
|
-
*/
|
|
40
|
-
export function createTextBlockMsg(text) {
|
|
41
|
-
return {
|
|
42
|
-
type: "assistant",
|
|
43
|
-
message: {
|
|
44
|
-
content: [{ type: "text", text }],
|
|
45
|
-
},
|
|
46
|
-
};
|
|
47
|
-
}
|
|
48
|
-
|
|
49
|
-
/**
|
|
50
|
-
* Reads a PassThrough / Readable stream to a single string.
|
|
51
|
-
* @param {import("node:stream").Readable} stream
|
|
52
|
-
* @returns {string}
|
|
53
|
-
*/
|
|
54
|
-
export function collectStream(stream) {
|
|
55
|
-
const data = stream.read();
|
|
56
|
-
return data ? data.toString() : "";
|
|
57
|
-
}
|
|
58
|
-
|
|
59
|
-
/**
|
|
60
|
-
* Reads a stream and returns non-empty lines.
|
|
61
|
-
* @param {import("node:stream").Readable} stream
|
|
62
|
-
* @returns {string[]}
|
|
63
|
-
*/
|
|
64
|
-
export function collectLines(stream) {
|
|
65
|
-
return collectStream(stream).split("\n").filter(Boolean);
|
|
66
|
-
}
|
|
67
|
-
|
|
68
|
-
/**
|
|
69
|
-
* Strips ANSI SGR escape sequences from a string.
|
|
70
|
-
* @param {string} s
|
|
71
|
-
* @returns {string}
|
|
72
|
-
*/
|
|
73
|
-
export function stripAnsi(s) {
|
|
74
|
-
// biome-ignore lint/suspicious/noControlCharactersInRegex: ANSI SGR detection is intentional.
|
|
75
|
-
return s.replace(/\[[0-9;]*m/g, "");
|
|
76
|
-
}
|
|
77
|
-
|
|
78
|
-
/**
|
|
79
|
-
* Writes each line followed by a newline to the writer, then ends it.
|
|
80
|
-
* @param {import("node:stream").Writable} writer
|
|
81
|
-
* @param {string[]} lines
|
|
82
|
-
* @returns {Promise<void>}
|
|
83
|
-
*/
|
|
84
|
-
export async function writeLines(writer, lines) {
|
|
85
|
-
for (const line of lines) writer.write(line + "\n");
|
|
86
|
-
await new Promise((resolve) => writer.end(resolve));
|
|
87
|
-
}
|
|
88
|
-
|
|
89
|
-
/**
|
|
90
|
-
* Builds a minimal-but-valid trace object for TraceQuery tests.
|
|
91
|
-
* Replaces a 155-line inline buildTrace in trace-query.test.js.
|
|
92
|
-
*
|
|
93
|
-
* @param {object} [overrides]
|
|
94
|
-
* @param {object} [overrides.metadata]
|
|
95
|
-
* @param {object[]} [overrides.turns]
|
|
96
|
-
* @param {object} [overrides.summary]
|
|
97
|
-
* @returns {object} Trace object.
|
|
98
|
-
*/
|
|
99
|
-
export function createTestTrace(overrides = {}) {
|
|
100
|
-
const { metadata = {}, turns = [], summary = {} } = overrides;
|
|
101
|
-
return {
|
|
102
|
-
schema_version: "1.1",
|
|
103
|
-
metadata: {
|
|
104
|
-
session_id: "sess-test",
|
|
105
|
-
model: "claude-opus-4-7",
|
|
106
|
-
started_at: "2026-01-01T00:00:00Z",
|
|
107
|
-
ended_at: "2026-01-01T00:00:05Z",
|
|
108
|
-
tools: [],
|
|
109
|
-
...metadata,
|
|
110
|
-
},
|
|
111
|
-
turns,
|
|
112
|
-
summary: {
|
|
113
|
-
total_turns: turns.length,
|
|
114
|
-
total_tokens: 0,
|
|
115
|
-
tool_calls: 0,
|
|
116
|
-
errors: 0,
|
|
117
|
-
...summary,
|
|
118
|
-
},
|
|
119
|
-
};
|
|
120
|
-
}
|
|
121
|
-
|
|
122
|
-
/**
|
|
123
|
-
* Creates an async-generator agent query stub. The shape of `messages`
|
|
124
|
-
* determines per-call behaviour:
|
|
125
|
-
* - Flat array of message objects: yielded on every invocation.
|
|
126
|
-
* - Array of arrays (batches): the Nth invocation yields messages from the
|
|
127
|
-
* Nth batch. Subsequent calls past the last batch repeat the last one.
|
|
128
|
-
*
|
|
129
|
-
* @param {object[] | object[][]} messages
|
|
130
|
-
* @param {(params: object) => void} [onParams] - Invoked with call params.
|
|
131
|
-
* @returns {Function} Async generator mimicking `query({...})`.
|
|
132
|
-
*/
|
|
133
|
-
export function createMockAgentQuery(messages, onParams) {
|
|
134
|
-
const isBatched = Array.isArray(messages[0]);
|
|
135
|
-
let callIndex = 0;
|
|
136
|
-
return async function* mockQuery(params) {
|
|
137
|
-
if (onParams) onParams(params);
|
|
138
|
-
if (!isBatched) {
|
|
139
|
-
for (const msg of messages) yield msg;
|
|
140
|
-
return;
|
|
141
|
-
}
|
|
142
|
-
const batch = messages[callIndex] ?? messages[messages.length - 1] ?? [];
|
|
143
|
-
callIndex += 1;
|
|
144
|
-
for (const msg of batch) yield msg;
|
|
145
|
-
};
|
|
146
|
-
}
|