@forwardimpact/libharness 0.1.22 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -201
- package/README.md +196 -80
- package/bin/fit-benchmark.js +44 -0
- package/bin/fit-harness.js +358 -0
- package/bin/fit-selfedit.js +165 -0
- package/bin/fit-trace.js +510 -0
- package/package.json +41 -11
- package/src/agent-runner.js +256 -0
- package/src/benchmark/apm-installer.js +207 -0
- package/src/benchmark/env-loader.js +158 -0
- package/src/benchmark/hook-env.js +40 -0
- package/src/benchmark/invariants.js +141 -0
- package/src/benchmark/judge.js +187 -0
- package/src/benchmark/npm-installer.js +87 -0
- package/src/benchmark/report.js +604 -0
- package/src/benchmark/result.js +127 -0
- package/src/benchmark/runner.js +688 -0
- package/src/benchmark/scheduler.js +78 -0
- package/src/benchmark/task-family.js +260 -0
- package/src/benchmark/workdir.js +344 -0
- package/src/commands/assert.js +153 -0
- package/src/commands/benchmark-definition.js +175 -0
- package/src/commands/benchmark-invariants.js +73 -0
- package/src/commands/benchmark-report.js +51 -0
- package/src/commands/benchmark-run.js +175 -0
- package/src/commands/by-discussion.js +94 -0
- package/src/commands/callback.js +119 -0
- package/src/commands/discuss.js +132 -0
- package/src/commands/facilitate.js +123 -0
- package/src/commands/output.js +36 -0
- package/src/commands/run.js +152 -0
- package/src/commands/supervise.js +136 -0
- package/src/commands/task-input.js +54 -0
- package/src/commands/tee.js +53 -0
- package/src/commands/trace.js +630 -0
- package/src/commands/work-tracker.js +35 -0
- package/src/cost.js +79 -0
- package/src/discuss-tools.js +173 -0
- package/src/discusser.js +394 -0
- package/src/events/github.js +161 -0
- package/src/facilitator.js +205 -0
- package/src/inbox-poller.js +81 -0
- package/src/index.js +72 -2
- package/src/judge.js +210 -0
- package/src/message-bus.js +118 -0
- package/src/orchestration-loop.js +330 -0
- package/src/orchestration-toolkit.js +441 -0
- package/src/orchestrator-helpers.js +23 -0
- package/src/profile-prompt.js +266 -0
- package/src/redaction.js +253 -0
- package/src/render/line-renderer.js +54 -0
- package/src/render/orchestrator-filter.js +19 -0
- package/src/render/palette.js +63 -0
- package/src/render/tool-hints.js +154 -0
- package/src/render/turn-renderer.js +96 -0
- package/src/reply-emitter.js +47 -0
- package/src/sequence-counter.js +21 -0
- package/src/signature-filter.js +27 -0
- package/src/supervisor.js +236 -0
- package/src/tee-writer.js +150 -0
- package/src/trace-collector.js +444 -0
- package/src/trace-github.js +473 -0
- package/src/trace-multi.js +101 -0
- package/src/trace-query.js +748 -0
- package/src/trace-render.js +211 -0
- package/src/trace-usage.js +249 -0
- package/src/fixture/assertions.js +0 -42
- package/src/fixture/cache.js +0 -50
- package/src/fixture/eval.js +0 -146
- package/src/fixture/index.js +0 -9
- package/src/fixture/pathway.js +0 -451
- package/src/fixture/services.js +0 -56
- package/src/mock/clients.js +0 -135
- package/src/mock/config.js +0 -45
- package/src/mock/data.js +0 -46
- package/src/mock/fs.js +0 -111
- package/src/mock/grpc.js +0 -94
- package/src/mock/http.js +0 -60
- package/src/mock/index.js +0 -36
- package/src/mock/infra.js +0 -219
- package/src/mock/logger.js +0 -42
- package/src/mock/observer.js +0 -74
- package/src/mock/resource-index.js +0 -95
- package/src/mock/service-callbacks.js +0 -39
- package/src/mock/services.js +0 -79
- package/src/mock/spy.js +0 -44
- package/src/mock/storage.js +0 -118
|
@@ -0,0 +1,748 @@
|
|
|
1
|
+
import {
|
|
2
|
+
ZERO_USAGE,
|
|
3
|
+
bucketUsageByTool,
|
|
4
|
+
carriedPerTurn,
|
|
5
|
+
computeDivergence,
|
|
6
|
+
isPreChangeDoc,
|
|
7
|
+
perMessageUsage,
|
|
8
|
+
reconcileBucketsToTotals,
|
|
9
|
+
} from "./trace-usage.js";
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Query engine for structured trace documents produced by TraceCollector.
|
|
13
|
+
*
|
|
14
|
+
* Loads a structured JSON trace into memory and provides methods for
|
|
15
|
+
* paging, searching, filtering, and summarizing turns — the operations
|
|
16
|
+
* agents need to analyze large traces efficiently.
|
|
17
|
+
*/
|
|
18
|
+
export class TraceQuery {
|
|
19
|
+
/**
|
|
20
|
+
* @param {object} trace - Structured trace document (output of TraceCollector.toJSON())
|
|
21
|
+
*/
|
|
22
|
+
constructor(trace) {
|
|
23
|
+
this.trace = trace;
|
|
24
|
+
this.metadata = trace.metadata ?? {};
|
|
25
|
+
this.turns = trace.turns ?? [];
|
|
26
|
+
this.summary = trace.summary ?? {};
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* High-level overview: metadata, summary, turn count, tool frequency,
|
|
31
|
+
* and the first user message text (taskPrompt) when present.
|
|
32
|
+
* @returns {object}
|
|
33
|
+
*/
|
|
34
|
+
overview() {
|
|
35
|
+
const firstUser = this.turns.find((t) => t.role === "user");
|
|
36
|
+
const taskPrompt = firstUser
|
|
37
|
+
? firstUser.content
|
|
38
|
+
.filter((b) => b.type === "text")
|
|
39
|
+
.map((b) => b.text)
|
|
40
|
+
.join("\n")
|
|
41
|
+
: null;
|
|
42
|
+
return {
|
|
43
|
+
metadata: this.metadata,
|
|
44
|
+
summary: this.summary,
|
|
45
|
+
turnCount: this.turns.length,
|
|
46
|
+
resultEventTurns: this.summary.numTurns ?? null,
|
|
47
|
+
turnPopulations: {
|
|
48
|
+
turnCount: "rendered-trace-turns",
|
|
49
|
+
resultEventTurns: "result-event-turns",
|
|
50
|
+
},
|
|
51
|
+
tools: this.toolFrequency(),
|
|
52
|
+
taskPrompt,
|
|
53
|
+
};
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Full system/init event — the single most diagnostic message for
|
|
58
|
+
* root-cause analysis. Returns null for traces collected before this
|
|
59
|
+
* field existed.
|
|
60
|
+
* @returns {object|null}
|
|
61
|
+
*/
|
|
62
|
+
init() {
|
|
63
|
+
return this.trace.initEvent ?? null;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/**
|
|
67
|
+
* Retrieve a single turn by its index.
|
|
68
|
+
* @param {number} index
|
|
69
|
+
* @returns {object|null}
|
|
70
|
+
*/
|
|
71
|
+
turn(index) {
|
|
72
|
+
return this.turns.find((t) => t.index === index) ?? null;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Filter turns by composable structural criteria. All criteria are
|
|
77
|
+
* combined as AND. `tool()` and `errors()` remain as convenience
|
|
78
|
+
* shortcuts for pre-existing workflows.
|
|
79
|
+
*
|
|
80
|
+
* `toolName` matches assistant turns only. Applying `toolName` without
|
|
81
|
+
* `role: "assistant"` still drops every non-assistant turn, because
|
|
82
|
+
* resolving tool_use → tool_result pairs requires the `tool()` method.
|
|
83
|
+
* `isError` matches tool_result turns only. Combining `toolName` with
|
|
84
|
+
* `isError` therefore always returns `[]` (no turn is both assistant
|
|
85
|
+
* and tool_result) — use `tool(name)` for "errors from Bash"–shaped
|
|
86
|
+
* queries.
|
|
87
|
+
*
|
|
88
|
+
* @param {object} [opts]
|
|
89
|
+
* @param {string} [opts.role] - Exact role match (system | user |
|
|
90
|
+
* assistant | tool_result).
|
|
91
|
+
* @param {string} [opts.toolName] - Matches assistant turns with a
|
|
92
|
+
* tool_use block of this name. Drops all non-assistant turns.
|
|
93
|
+
* @param {boolean} [opts.isError] - Matches tool_result turns by
|
|
94
|
+
* `isError` value. Drops all non-tool_result turns.
|
|
95
|
+
* @returns {object[]}
|
|
96
|
+
*/
|
|
97
|
+
filter(opts = {}) {
|
|
98
|
+
const { role, toolName, isError } = opts;
|
|
99
|
+
return this.turns.filter(
|
|
100
|
+
(turn) =>
|
|
101
|
+
matchesRole(turn, role) &&
|
|
102
|
+
matchesError(turn, isError) &&
|
|
103
|
+
matchesToolName(turn, toolName),
|
|
104
|
+
);
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/** @returns {number} */
|
|
108
|
+
count() {
|
|
109
|
+
return this.turns.length;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* Return turns in range [from, to) (zero-indexed).
|
|
114
|
+
* @param {number} from
|
|
115
|
+
* @param {number} to
|
|
116
|
+
* @returns {object[]}
|
|
117
|
+
*/
|
|
118
|
+
batch(from, to) {
|
|
119
|
+
return this.turns.slice(from, to);
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* First N turns.
|
|
124
|
+
* @param {number} [n=10]
|
|
125
|
+
* @returns {object[]}
|
|
126
|
+
*/
|
|
127
|
+
head(n = 10) {
|
|
128
|
+
return this.turns.slice(0, n);
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* Last N turns.
|
|
133
|
+
* @param {number} [n=10]
|
|
134
|
+
* @returns {object[]}
|
|
135
|
+
*/
|
|
136
|
+
tail(n = 10) {
|
|
137
|
+
return this.turns.slice(-n);
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/**
|
|
141
|
+
* Search all turn content for a regex pattern. Returns matching turns
|
|
142
|
+
* with the matched text highlighted by context.
|
|
143
|
+
*
|
|
144
|
+
* Searches: assistant text blocks, tool_use names and stringified input,
|
|
145
|
+
* and tool_result content.
|
|
146
|
+
*
|
|
147
|
+
* @param {string} pattern - Regex pattern (case-insensitive)
|
|
148
|
+
* @param {object} [opts]
|
|
149
|
+
* @param {number} [opts.context=0] - Number of surrounding turns to include
|
|
150
|
+
* @param {number} [opts.limit=50] - Max results
|
|
151
|
+
* @param {boolean} [opts.full=false] - Emit full content block text in
|
|
152
|
+
* match descriptions instead of the default narrow excerpt window.
|
|
153
|
+
* @returns {object[]} Array of {turn, matches, context?}
|
|
154
|
+
*/
|
|
155
|
+
search(pattern, opts = {}) {
|
|
156
|
+
const { context = 0, limit = 50, full = false } = opts;
|
|
157
|
+
const re = new RegExp(pattern, "gi");
|
|
158
|
+
const hits = [];
|
|
159
|
+
|
|
160
|
+
for (const turn of this.turns) {
|
|
161
|
+
const matches = matchTurn(turn, re, full);
|
|
162
|
+
if (matches.length > 0) {
|
|
163
|
+
const entry = { turn, matches };
|
|
164
|
+
if (context > 0) {
|
|
165
|
+
const idx = turn.index;
|
|
166
|
+
entry.context = this.turns.filter(
|
|
167
|
+
(t) =>
|
|
168
|
+
t.index !== idx &&
|
|
169
|
+
t.index >= idx - context &&
|
|
170
|
+
t.index <= idx + context,
|
|
171
|
+
);
|
|
172
|
+
}
|
|
173
|
+
hits.push(entry);
|
|
174
|
+
if (hits.length >= limit) break;
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
return hits;
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/**
|
|
181
|
+
* Tool usage frequency, sorted descending.
|
|
182
|
+
* @returns {Array<{tool: string, count: number}>}
|
|
183
|
+
*/
|
|
184
|
+
toolFrequency() {
|
|
185
|
+
const counts = {};
|
|
186
|
+
for (const turn of this.turns) {
|
|
187
|
+
if (turn.role !== "assistant") continue;
|
|
188
|
+
for (const block of turn.content) {
|
|
189
|
+
if (block.type === "tool_use") {
|
|
190
|
+
counts[block.name] = (counts[block.name] ?? 0) + 1;
|
|
191
|
+
}
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
return Object.entries(counts)
|
|
195
|
+
.map(([tool, count]) => ({ tool, count }))
|
|
196
|
+
.sort((a, b) => b.count - a.count);
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/**
|
|
200
|
+
* Filter turns involving a specific tool (both the tool_use and its result).
|
|
201
|
+
* @param {string} name - Tool name
|
|
202
|
+
* @returns {object[]}
|
|
203
|
+
*/
|
|
204
|
+
tool(name) {
|
|
205
|
+
const toolUseIds = collectToolUseIds(this.turns, name);
|
|
206
|
+
const assistantTurns = this.turns.filter(
|
|
207
|
+
(t) =>
|
|
208
|
+
t.role === "assistant" &&
|
|
209
|
+
t.content.some((b) => b.type === "tool_use" && b.name === name),
|
|
210
|
+
);
|
|
211
|
+
const resultTurns = this.turns.filter(
|
|
212
|
+
(t) => t.role === "tool_result" && toolUseIds.has(t.toolUseId),
|
|
213
|
+
);
|
|
214
|
+
return [...assistantTurns, ...resultTurns].sort(
|
|
215
|
+
(a, b) => a.index - b.index,
|
|
216
|
+
);
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/**
|
|
220
|
+
* All error turns (tool results with isError=true).
|
|
221
|
+
* @returns {object[]}
|
|
222
|
+
*/
|
|
223
|
+
errors() {
|
|
224
|
+
return this.turns.filter(
|
|
225
|
+
(t) => t.role === "tool_result" && t.isError === true,
|
|
226
|
+
);
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
/**
|
|
230
|
+
* Extract just the reasoning text from assistant turns.
|
|
231
|
+
* @param {object} [opts]
|
|
232
|
+
* @param {number} [opts.from] - Start turn index
|
|
233
|
+
* @param {number} [opts.to] - End turn index (exclusive)
|
|
234
|
+
* @returns {Array<{index: number, text: string}>}
|
|
235
|
+
*/
|
|
236
|
+
reasoning(opts = {}) {
|
|
237
|
+
const { from, to } = opts;
|
|
238
|
+
const results = [];
|
|
239
|
+
for (const turn of this.turns) {
|
|
240
|
+
if (turn.role !== "assistant") continue;
|
|
241
|
+
if (from !== undefined && turn.index < from) continue;
|
|
242
|
+
if (to !== undefined && turn.index >= to) continue;
|
|
243
|
+
const texts = turn.content
|
|
244
|
+
.filter((b) => b.type === "text")
|
|
245
|
+
.map((b) => b.text);
|
|
246
|
+
if (texts.length > 0) {
|
|
247
|
+
results.push({ index: turn.index, text: texts.join("\n") });
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
return results;
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/**
|
|
254
|
+
* Compact one-line-per-assistant-turn timeline showing tool names,
|
|
255
|
+
* reasoning snippet, and token usage. Thinking-only turns are marked
|
|
256
|
+
* as such and their content is omitted (it is model-internal).
|
|
257
|
+
* @returns {string[]}
|
|
258
|
+
*/
|
|
259
|
+
timeline() {
|
|
260
|
+
const lines = [];
|
|
261
|
+
for (const turn of this.turns) {
|
|
262
|
+
if (turn.role !== "assistant") continue;
|
|
263
|
+
|
|
264
|
+
const tools = turn.content
|
|
265
|
+
.filter((b) => b.type === "tool_use")
|
|
266
|
+
.map((b) => b.name);
|
|
267
|
+
|
|
268
|
+
const textBlocks = turn.content
|
|
269
|
+
.filter((b) => b.type === "text")
|
|
270
|
+
.map((b) => b.text);
|
|
271
|
+
|
|
272
|
+
const hasThinking = turn.content.some((b) => b.type === "thinking");
|
|
273
|
+
|
|
274
|
+
// Skip thinking-only turns (no user-visible content).
|
|
275
|
+
if (hasThinking && tools.length === 0 && textBlocks.length === 0)
|
|
276
|
+
continue;
|
|
277
|
+
|
|
278
|
+
const snippet = textBlocks.join(" ").slice(0, 80).replace(/\n/g, " ");
|
|
279
|
+
|
|
280
|
+
const input = turn.usage?.inputTokens ?? 0;
|
|
281
|
+
const output = turn.usage?.outputTokens ?? 0;
|
|
282
|
+
const cacheRead = turn.usage?.cacheReadInputTokens ?? 0;
|
|
283
|
+
|
|
284
|
+
const toolStr = tools.length > 0 ? tools.join(", ") : "(text only)";
|
|
285
|
+
const tokenStr = `in:${fmtK(input + cacheRead)} out:${fmtK(output)}`;
|
|
286
|
+
|
|
287
|
+
lines.push(
|
|
288
|
+
`[${turn.index}] ${toolStr.padEnd(30)} ${tokenStr.padEnd(18)} ${snippet}`,
|
|
289
|
+
);
|
|
290
|
+
}
|
|
291
|
+
return lines;
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
/**
|
|
295
|
+
* Token usage and cost breakdown, accounted once per API message, plus
|
|
296
|
+
* totals that name their population.
|
|
297
|
+
*
|
|
298
|
+
* A structured document collected before this change (version < 1.2.0)
|
|
299
|
+
* carries no message identity, so it reports its carried last-wins summary
|
|
300
|
+
* labeled as such — corrected figures come from re-running the NDJSON source.
|
|
301
|
+
*
|
|
302
|
+
* Otherwise: when the trace carries result events, totals are the SDK's
|
|
303
|
+
* accumulated result-event sums (authoritative); the per-message sums are
|
|
304
|
+
* compared against them and any divergence on input/cacheRead/cacheCreation
|
|
305
|
+
* is surfaced, never silently absorbed. A trace with no result event
|
|
306
|
+
* (truncated or in-flight) falls back to the per-message sums, with output
|
|
307
|
+
* flagged as a streaming-snapshot lower bound and cost/duration/turns
|
|
308
|
+
* reported as unavailable rather than a silent 0.
|
|
309
|
+
* @returns {object}
|
|
310
|
+
*/
|
|
311
|
+
stats() {
|
|
312
|
+
if (isPreChangeDoc(this.trace.version)) {
|
|
313
|
+
return this.#carriedDocumentStats();
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
const { perMessage, totals: perMessageTotals } = perMessageUsage(
|
|
317
|
+
this.turns,
|
|
318
|
+
);
|
|
319
|
+
const re = this.summary.tokenUsage;
|
|
320
|
+
|
|
321
|
+
if (re) {
|
|
322
|
+
return {
|
|
323
|
+
totals: {
|
|
324
|
+
inputTokens: re.inputTokens ?? 0,
|
|
325
|
+
outputTokens: re.outputTokens ?? 0,
|
|
326
|
+
cacheReadInputTokens: re.cacheReadInputTokens ?? 0,
|
|
327
|
+
cacheCreationInputTokens: re.cacheCreationInputTokens ?? 0,
|
|
328
|
+
totalCostUsd: this.summary.totalCostUsd ?? 0,
|
|
329
|
+
durationMs: this.summary.durationMs ?? 0,
|
|
330
|
+
durationLabel: "cumulative invocation time",
|
|
331
|
+
resultEventTurns: this.summary.numTurns ?? 0,
|
|
332
|
+
population: "result-event-sum",
|
|
333
|
+
resultEventsPresent: true,
|
|
334
|
+
},
|
|
335
|
+
perTurn: perMessage,
|
|
336
|
+
modelUsage: this.summary.modelUsage ?? null,
|
|
337
|
+
divergence: computeDivergence(perMessageTotals, re),
|
|
338
|
+
};
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
return {
|
|
342
|
+
totals: {
|
|
343
|
+
...perMessageTotals,
|
|
344
|
+
outputIsStreamingSnapshot: true,
|
|
345
|
+
totalCostUsd: null,
|
|
346
|
+
durationMs: null,
|
|
347
|
+
resultEventTurns: null,
|
|
348
|
+
population: "per-message-fallback",
|
|
349
|
+
resultEventsPresent: false,
|
|
350
|
+
},
|
|
351
|
+
perTurn: perMessage,
|
|
352
|
+
modelUsage: this.summary.modelUsage ?? null,
|
|
353
|
+
divergence: null,
|
|
354
|
+
};
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
/**
|
|
358
|
+
* Stats for a pre-change structured document: report the carried last-wins
|
|
359
|
+
* summary and per-stream-event breakdown, each labeled, without claiming
|
|
360
|
+
* result-event parity (the document lacks the message identity it needs).
|
|
361
|
+
* @returns {object}
|
|
362
|
+
*/
|
|
363
|
+
#carriedDocumentStats() {
|
|
364
|
+
const re = this.summary.tokenUsage ?? ZERO_USAGE;
|
|
365
|
+
return {
|
|
366
|
+
totals: {
|
|
367
|
+
inputTokens: re.inputTokens ?? 0,
|
|
368
|
+
outputTokens: re.outputTokens ?? 0,
|
|
369
|
+
cacheReadInputTokens: re.cacheReadInputTokens ?? 0,
|
|
370
|
+
cacheCreationInputTokens: re.cacheCreationInputTokens ?? 0,
|
|
371
|
+
totalCostUsd: this.summary.totalCostUsd ?? 0,
|
|
372
|
+
durationMs: this.summary.durationMs ?? 0,
|
|
373
|
+
population: "carried-document-summary",
|
|
374
|
+
},
|
|
375
|
+
perTurn: carriedPerTurn(this.turns),
|
|
376
|
+
modelUsage: this.summary.modelUsage ?? null,
|
|
377
|
+
divergence: null,
|
|
378
|
+
};
|
|
379
|
+
}
|
|
380
|
+
|
|
381
|
+
/**
|
|
382
|
+
* One record per `tool_use` block, each paired with its `tool_result`
|
|
383
|
+
* (joined by `toolUseId`) or `result: null` for orphaned calls.
|
|
384
|
+
* @returns {Array<{turnIndex: number, name: string, toolUseId: string, input: object, result: {content: *, isError: boolean}|null}>}
|
|
385
|
+
*/
|
|
386
|
+
toolCalls() {
|
|
387
|
+
const blocks = collectToolUseBlocks(this.turns);
|
|
388
|
+
const results = new Map();
|
|
389
|
+
for (const turn of this.turns) {
|
|
390
|
+
if (turn.role === "tool_result" && turn.toolUseId) {
|
|
391
|
+
results.set(turn.toolUseId, {
|
|
392
|
+
content: turn.content ?? null,
|
|
393
|
+
isError: turn.isError ?? false,
|
|
394
|
+
});
|
|
395
|
+
}
|
|
396
|
+
}
|
|
397
|
+
return [...blocks.entries()].map(([toolUseId, b]) => ({
|
|
398
|
+
turnIndex: b.turnIndex,
|
|
399
|
+
name: b.name,
|
|
400
|
+
toolUseId,
|
|
401
|
+
input: b.input,
|
|
402
|
+
result: results.get(toolUseId) ?? null,
|
|
403
|
+
}));
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
/**
|
|
407
|
+
* One record per `Bash` `tool_use` block, carrying its command text.
|
|
408
|
+
* @param {string} [re] - Optional regex source tested against `input.command`.
|
|
409
|
+
* @returns {Array<{turnIndex: number, toolUseId: string, command: string}>}
|
|
410
|
+
*/
|
|
411
|
+
commands(re) {
|
|
412
|
+
const filter = re === undefined ? null : new RegExp(re);
|
|
413
|
+
const out = [];
|
|
414
|
+
for (const [toolUseId, b] of collectToolUseBlocks(this.turns, "Bash")) {
|
|
415
|
+
const command = b.input?.command ?? "";
|
|
416
|
+
if (filter && !filter.test(command)) continue;
|
|
417
|
+
out.push({ turnIndex: b.turnIndex, toolUseId, command });
|
|
418
|
+
}
|
|
419
|
+
return out;
|
|
420
|
+
}
|
|
421
|
+
|
|
422
|
+
/**
|
|
423
|
+
* Distinct `file_path` arguments across `Read`/`Edit`/`Write` tool calls,
|
|
424
|
+
* frequency-sorted (count desc, path asc tiebreak).
|
|
425
|
+
* @param {string} [prefix] - Optional `startsWith` filter.
|
|
426
|
+
* @returns {Array<{path: string, count: number}>}
|
|
427
|
+
*/
|
|
428
|
+
paths(prefix) {
|
|
429
|
+
return [...collectFilePaths(this.turns).entries()]
|
|
430
|
+
.filter(([path]) => prefix === undefined || path.startsWith(prefix))
|
|
431
|
+
.map(([path, count]) => ({ path, count }))
|
|
432
|
+
.sort((a, b) => b.count - a.count || a.path.localeCompare(b.path));
|
|
433
|
+
}
|
|
434
|
+
|
|
435
|
+
/**
|
|
436
|
+
* Side-by-side comparison of this trace against another peer `TraceQuery`.
|
|
437
|
+
* Identity (case name, participant) comes from the caller — the trace
|
|
438
|
+
* carries no filename.
|
|
439
|
+
* @param {TraceQuery} other
|
|
440
|
+
* @param {{aIdentity: {caseName: string, participant: string|null}, bIdentity: {caseName: string, participant: string|null}}} identities
|
|
441
|
+
* @returns {{a: object, b: object, toolDelta: Array, pathDelta: Array}}
|
|
442
|
+
*/
|
|
443
|
+
compare(other, { aIdentity, bIdentity } = {}) {
|
|
444
|
+
const a = sideSummary(this, aIdentity);
|
|
445
|
+
const b = sideSummary(other, bIdentity);
|
|
446
|
+
|
|
447
|
+
const toolNames = [
|
|
448
|
+
...new Set([...a.toolFreq.keys(), ...b.toolFreq.keys()]),
|
|
449
|
+
];
|
|
450
|
+
const toolDelta = toolNames
|
|
451
|
+
.map((tool) => {
|
|
452
|
+
const av = a.toolFreq.get(tool) ?? 0;
|
|
453
|
+
const bv = b.toolFreq.get(tool) ?? 0;
|
|
454
|
+
return { tool, a: av, b: bv, diff: bv - av };
|
|
455
|
+
})
|
|
456
|
+
.sort(
|
|
457
|
+
(x, y) =>
|
|
458
|
+
Math.abs(y.diff) - Math.abs(x.diff) || x.tool.localeCompare(y.tool),
|
|
459
|
+
);
|
|
460
|
+
|
|
461
|
+
const pathNames = [
|
|
462
|
+
...new Set([...a.pathFreq.keys(), ...b.pathFreq.keys()]),
|
|
463
|
+
];
|
|
464
|
+
const pathDelta = pathNames
|
|
465
|
+
.map((path) => {
|
|
466
|
+
const av = a.pathFreq.get(path) ?? 0;
|
|
467
|
+
const bv = b.pathFreq.get(path) ?? 0;
|
|
468
|
+
return { path, a: av, b: bv, diff: bv - av };
|
|
469
|
+
})
|
|
470
|
+
.sort(
|
|
471
|
+
(x, y) =>
|
|
472
|
+
Math.abs(y.diff) - Math.abs(x.diff) || x.path.localeCompare(y.path),
|
|
473
|
+
);
|
|
474
|
+
|
|
475
|
+
return { a: a.surface, b: b.surface, toolDelta, pathDelta };
|
|
476
|
+
}
|
|
477
|
+
|
|
478
|
+
/**
|
|
479
|
+
* Per-tool token attribution: each `tool_use` block gets an equal share of
|
|
480
|
+
* its host turn's usage; assistant turns with no `tool_use` block contribute
|
|
481
|
+
* full usage to the `(no-tool)` bucket. Per-bucket sums are scaled onto
|
|
482
|
+
* `stats().totals` — the authoritative population (result-event sums when the
|
|
483
|
+
* trace carries them, the per-message fallback otherwise) — so the buckets
|
|
484
|
+
* answer "of the reported total, what share did each tool drive" rather than
|
|
485
|
+
* a separate per-turn re-count that drifts from the headline figure. The
|
|
486
|
+
* largest bucket absorbs the rounding residual on each axis, so the input,
|
|
487
|
+
* output, and `costShare` columns each sum to the corresponding `totals`
|
|
488
|
+
* value (and `1.0`) exactly (criterion-6 invariant).
|
|
489
|
+
* @returns {{perTool: Array<{tool: string, turns: number, inputTokens: number, outputTokens: number, costShare: number}>, totals: object}}
|
|
490
|
+
*/
|
|
491
|
+
statsByTool() {
|
|
492
|
+
const { buckets, bucketTurns } = bucketUsageByTool(this.turns);
|
|
493
|
+
const totals = this.stats().totals;
|
|
494
|
+
const perTool = reconcileBucketsToTotals(buckets, bucketTurns, totals);
|
|
495
|
+
return { perTool, totals };
|
|
496
|
+
}
|
|
497
|
+
|
|
498
|
+
/**
|
|
499
|
+
* Totals-only view — `stats().totals` with no per-turn array.
|
|
500
|
+
* @returns {{totals: object}}
|
|
501
|
+
*/
|
|
502
|
+
statsSummary() {
|
|
503
|
+
return { totals: this.stats().totals };
|
|
504
|
+
}
|
|
505
|
+
}
|
|
506
|
+
|
|
507
|
+
/**
|
|
508
|
+
* @param {object} turn
|
|
509
|
+
* @param {string|undefined} role
|
|
510
|
+
* @returns {boolean}
|
|
511
|
+
*/
|
|
512
|
+
function matchesRole(turn, role) {
|
|
513
|
+
return role === undefined || turn.role === role;
|
|
514
|
+
}
|
|
515
|
+
|
|
516
|
+
/**
|
|
517
|
+
* @param {object} turn
|
|
518
|
+
* @param {boolean|undefined} isError
|
|
519
|
+
* @returns {boolean}
|
|
520
|
+
*/
|
|
521
|
+
function matchesError(turn, isError) {
|
|
522
|
+
if (isError === undefined) return true;
|
|
523
|
+
return turn.role === "tool_result" && turn.isError === isError;
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
/**
|
|
527
|
+
* @param {object} turn
|
|
528
|
+
* @param {string|undefined} toolName
|
|
529
|
+
* @returns {boolean}
|
|
530
|
+
*/
|
|
531
|
+
function matchesToolName(turn, toolName) {
|
|
532
|
+
if (toolName === undefined) return true;
|
|
533
|
+
return (
|
|
534
|
+
turn.role === "assistant" &&
|
|
535
|
+
turn.content.some((b) => b.type === "tool_use" && b.name === toolName)
|
|
536
|
+
);
|
|
537
|
+
}
|
|
538
|
+
|
|
539
|
+
/**
|
|
540
|
+
* Collect every assistant `tool_use` block keyed by `toolUseId`, optionally
|
|
541
|
+
* filtered by tool name. The shared join-key source feeding `toolCalls()`,
|
|
542
|
+
* `commands()`, and `collectToolUseIds()`. Insertion order follows turn order.
|
|
543
|
+
* @param {object[]} turns
|
|
544
|
+
* @param {string} [name] - Optional tool-name filter.
|
|
545
|
+
* @returns {Map<string, {turnIndex: number, name: string, input: object}>}
|
|
546
|
+
*/
|
|
547
|
+
function collectToolUseBlocks(turns, name) {
|
|
548
|
+
const blocks = new Map();
|
|
549
|
+
for (const turn of turns) {
|
|
550
|
+
if (turn.role !== "assistant") continue;
|
|
551
|
+
for (const b of turn.content) {
|
|
552
|
+
if (b.type !== "tool_use" || !b.toolUseId) continue;
|
|
553
|
+
if (name !== undefined && b.name !== name) continue;
|
|
554
|
+
blocks.set(b.toolUseId, {
|
|
555
|
+
turnIndex: turn.index,
|
|
556
|
+
name: b.name,
|
|
557
|
+
input: b.input,
|
|
558
|
+
});
|
|
559
|
+
}
|
|
560
|
+
}
|
|
561
|
+
return blocks;
|
|
562
|
+
}
|
|
563
|
+
|
|
564
|
+
/**
|
|
565
|
+
* Collect all toolUseIds for a given tool name from assistant turns.
|
|
566
|
+
* @param {object[]} turns
|
|
567
|
+
* @param {string} name
|
|
568
|
+
* @returns {Set<string>}
|
|
569
|
+
*/
|
|
570
|
+
function collectToolUseIds(turns, name) {
|
|
571
|
+
return new Set(collectToolUseBlocks(turns, name).keys());
|
|
572
|
+
}
|
|
573
|
+
|
|
574
|
+
/** Tool names in `Read`/`Edit`/`Write` that carry a `file_path` argument. */
|
|
575
|
+
const PATH_TOOLS = new Set(["Read", "Edit", "Write"]);
|
|
576
|
+
|
|
577
|
+
/**
|
|
578
|
+
* Frequency map of distinct `file_path` arguments across `Read`/`Edit`/`Write`
|
|
579
|
+
* tool calls, in first-seen insertion order.
|
|
580
|
+
* @param {object[]} turns
|
|
581
|
+
* @returns {Map<string, number>}
|
|
582
|
+
*/
|
|
583
|
+
function collectFilePaths(turns) {
|
|
584
|
+
const counts = new Map();
|
|
585
|
+
for (const turn of turns) {
|
|
586
|
+
if (turn.role !== "assistant") continue;
|
|
587
|
+
for (const block of turn.content) {
|
|
588
|
+
if (block.type !== "tool_use" || !PATH_TOOLS.has(block.name)) continue;
|
|
589
|
+
const p = block.input?.file_path;
|
|
590
|
+
if (typeof p !== "string") continue;
|
|
591
|
+
counts.set(p, (counts.get(p) ?? 0) + 1);
|
|
592
|
+
}
|
|
593
|
+
}
|
|
594
|
+
return counts;
|
|
595
|
+
}
|
|
596
|
+
|
|
597
|
+
/**
|
|
598
|
+
* Build the per-side comparison surface plus the tool/path frequency maps
|
|
599
|
+
* the delta computation consumes. Empty traces emit a `(empty)` marker.
|
|
600
|
+
* @param {TraceQuery} query
|
|
601
|
+
* @param {{caseName: string, participant: string|null}} [identity]
|
|
602
|
+
* @returns {{surface: object, toolFreq: Map<string, number>, pathFreq: Map<string, number>}}
|
|
603
|
+
*/
|
|
604
|
+
function sideSummary(
|
|
605
|
+
query,
|
|
606
|
+
identity = { caseName: "(unknown)", participant: null },
|
|
607
|
+
) {
|
|
608
|
+
const toolFreq = new Map(query.toolFrequency().map((t) => [t.tool, t.count]));
|
|
609
|
+
const pathFreq = collectFilePaths(query.turns);
|
|
610
|
+
|
|
611
|
+
const isEmpty = query.turns.length === 0;
|
|
612
|
+
const metadata = {
|
|
613
|
+
caseName: identity.caseName,
|
|
614
|
+
participant: identity.participant ?? null,
|
|
615
|
+
};
|
|
616
|
+
if (isEmpty) metadata.marker = "(empty)";
|
|
617
|
+
|
|
618
|
+
const tools = [...toolFreq.keys()].sort();
|
|
619
|
+
const paths = [...pathFreq.keys()].sort();
|
|
620
|
+
|
|
621
|
+
return {
|
|
622
|
+
surface: {
|
|
623
|
+
metadata,
|
|
624
|
+
turnCount: query.turns.length,
|
|
625
|
+
tools,
|
|
626
|
+
paths,
|
|
627
|
+
pathCount: paths.length,
|
|
628
|
+
cost: query.stats().totals.totalCostUsd,
|
|
629
|
+
},
|
|
630
|
+
toolFreq,
|
|
631
|
+
pathFreq,
|
|
632
|
+
};
|
|
633
|
+
}
|
|
634
|
+
|
|
635
|
+
/**
|
|
636
|
+
* Search a single turn for regex matches. Returns array of match descriptions.
|
|
637
|
+
* @param {object} turn
|
|
638
|
+
* @param {RegExp} re
|
|
639
|
+
* @param {boolean} [full=false] - Emit full block text instead of an excerpt.
|
|
640
|
+
* @returns {string[]}
|
|
641
|
+
*/
|
|
642
|
+
function matchTurn(turn, re, full = false) {
|
|
643
|
+
if (turn.role === "assistant") return matchAssistantTurn(turn, re, full);
|
|
644
|
+
if (turn.role === "tool_result") return matchToolResultTurn(turn, re, full);
|
|
645
|
+
if (turn.role === "user") return matchUserTurn(turn, re, full);
|
|
646
|
+
return [];
|
|
647
|
+
}
|
|
648
|
+
|
|
649
|
+
function matchAssistantTurn(turn, re, full) {
|
|
650
|
+
const matches = [];
|
|
651
|
+
for (const block of turn.content) {
|
|
652
|
+
if (block.type === "text") {
|
|
653
|
+
const desc = describeText(block.text, re, "text", full);
|
|
654
|
+
if (desc) matches.push(desc);
|
|
655
|
+
} else if (block.type === "tool_use") {
|
|
656
|
+
matches.push(...matchToolUseBlock(block, re, full));
|
|
657
|
+
}
|
|
658
|
+
}
|
|
659
|
+
return matches;
|
|
660
|
+
}
|
|
661
|
+
|
|
662
|
+
function matchToolUseBlock(block, re, full) {
|
|
663
|
+
const matches = [];
|
|
664
|
+
if (re.test(block.name)) {
|
|
665
|
+
re.lastIndex = 0;
|
|
666
|
+
matches.push(`tool_name: ${block.name}`);
|
|
667
|
+
}
|
|
668
|
+
const inputStr = JSON.stringify(block.input);
|
|
669
|
+
const inputDesc = describeText(
|
|
670
|
+
inputStr,
|
|
671
|
+
re,
|
|
672
|
+
`tool_input(${block.name})`,
|
|
673
|
+
full,
|
|
674
|
+
);
|
|
675
|
+
if (inputDesc) matches.push(inputDesc);
|
|
676
|
+
return matches;
|
|
677
|
+
}
|
|
678
|
+
|
|
679
|
+
function matchToolResultTurn(turn, re, full) {
|
|
680
|
+
const content = turn.content ?? "";
|
|
681
|
+
const desc = describeText(content, re, "result", full);
|
|
682
|
+
return desc ? [desc] : [];
|
|
683
|
+
}
|
|
684
|
+
|
|
685
|
+
function matchUserTurn(turn, re, full) {
|
|
686
|
+
const matches = [];
|
|
687
|
+
for (const block of turn.content ?? []) {
|
|
688
|
+
if (block.type === "text") {
|
|
689
|
+
const desc = describeText(block.text, re, "user_text", full);
|
|
690
|
+
if (desc) matches.push(desc);
|
|
691
|
+
}
|
|
692
|
+
}
|
|
693
|
+
return matches;
|
|
694
|
+
}
|
|
695
|
+
|
|
696
|
+
/**
|
|
697
|
+
* Return a `<prefix>: <text-or-excerpt>` description when `text` matches
|
|
698
|
+
* the regex, or null when it does not. Centralises the full-vs-excerpt
|
|
699
|
+
* choice so each call site just supplies its prefix.
|
|
700
|
+
* @param {string} text
|
|
701
|
+
* @param {RegExp} re
|
|
702
|
+
* @param {string} prefix
|
|
703
|
+
* @param {boolean} full
|
|
704
|
+
* @returns {string|null}
|
|
705
|
+
*/
|
|
706
|
+
function describeText(text, re, prefix, full) {
|
|
707
|
+
if (!re.test(text)) return null;
|
|
708
|
+
re.lastIndex = 0;
|
|
709
|
+
return `${prefix}: ${full ? text : excerptAround(text, re)}`;
|
|
710
|
+
}
|
|
711
|
+
|
|
712
|
+
/**
|
|
713
|
+
* Extract a short excerpt around the first regex match in text.
|
|
714
|
+
* @param {string} text
|
|
715
|
+
* @param {RegExp} re
|
|
716
|
+
* @returns {string}
|
|
717
|
+
*/
|
|
718
|
+
function excerptAround(text, re) {
|
|
719
|
+
re.lastIndex = 0;
|
|
720
|
+
const m = re.exec(text);
|
|
721
|
+
if (!m) return text.slice(0, 100);
|
|
722
|
+
const start = Math.max(0, m.index - 40);
|
|
723
|
+
const end = Math.min(text.length, m.index + m[0].length + 40);
|
|
724
|
+
let excerpt = text.slice(start, end);
|
|
725
|
+
if (start > 0) excerpt = "..." + excerpt;
|
|
726
|
+
if (end < text.length) excerpt = excerpt + "...";
|
|
727
|
+
return excerpt;
|
|
728
|
+
}
|
|
729
|
+
|
|
730
|
+
/**
|
|
731
|
+
* Format a token count as compact K notation.
|
|
732
|
+
* @param {number} n
|
|
733
|
+
* @returns {string}
|
|
734
|
+
*/
|
|
735
|
+
function fmtK(n) {
|
|
736
|
+
if (n < 1000) return String(n);
|
|
737
|
+
return (n / 1000).toFixed(1) + "K";
|
|
738
|
+
}
|
|
739
|
+
|
|
740
|
+
/**
|
|
741
|
+
* Load a structured trace from a JSON string.
|
|
742
|
+
* @param {string} json
|
|
743
|
+
* @returns {TraceQuery}
|
|
744
|
+
*/
|
|
745
|
+
export function createTraceQuery(json) {
|
|
746
|
+
const trace = typeof json === "string" ? JSON.parse(json) : json;
|
|
747
|
+
return new TraceQuery(trace);
|
|
748
|
+
}
|