@forwardimpact/libharness 2.0.0 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +68 -65
- package/package.json +15 -13
- package/src/advisor.js +47 -41
- package/src/agent-runner.js +58 -48
- package/src/benchmark/apm-installer.js +28 -28
- package/src/benchmark/env-loader.js +24 -16
- package/src/benchmark/grade.js +44 -41
- package/src/benchmark/hidden-tests.js +25 -24
- package/src/benchmark/hook-env.js +11 -9
- package/src/benchmark/invariants.js +20 -17
- package/src/benchmark/judge.js +29 -28
- package/src/benchmark/npm-installer.js +9 -8
- package/src/benchmark/report.js +53 -50
- package/src/benchmark/result.js +24 -23
- package/src/benchmark/runner.js +75 -69
- package/src/benchmark/scheduler.js +17 -16
- package/src/benchmark/task-family.js +29 -27
- package/src/benchmark/trace-split.js +9 -8
- package/src/benchmark/workdir.js +27 -25
- package/src/claude-code-executable.js +11 -11
- package/src/commands/advisor-flags.js +8 -7
- package/src/commands/assert.js +16 -15
- package/src/commands/benchmark-definition.js +20 -20
- package/src/commands/benchmark-grade.js +13 -12
- package/src/commands/benchmark-report.js +5 -5
- package/src/commands/benchmark-run.js +31 -28
- package/src/commands/by-discussion.js +11 -11
- package/src/commands/callback.js +11 -11
- package/src/commands/discuss.js +8 -7
- package/src/commands/facilitate.js +16 -14
- package/src/commands/output.js +4 -3
- package/src/commands/run.js +15 -15
- package/src/commands/scan-logs.js +22 -20
- package/src/commands/selfedit.js +124 -0
- package/src/commands/supervise.js +13 -11
- package/src/commands/task-input.js +9 -9
- package/src/commands/tee.js +11 -10
- package/src/commands/trace.js +55 -42
- package/src/commands/work-tracker.js +4 -3
- package/src/cost.js +17 -17
- package/src/discuss-tools.js +16 -16
- package/src/discusser.js +39 -38
- package/src/events/github.js +54 -37
- package/src/facilitator.js +21 -21
- package/src/inbox-poller.js +4 -4
- package/src/judge.js +32 -30
- package/src/message-bus.js +12 -11
- package/src/orchestration-loop.js +35 -36
- package/src/orchestration-toolkit.js +58 -53
- package/src/orchestrator-helpers.js +2 -2
- package/src/profile-prompt.js +54 -53
- package/src/redaction.js +63 -57
- package/src/render/line-renderer.js +5 -5
- package/src/render/orchestrator-filter.js +3 -3
- package/src/render/palette.js +11 -9
- package/src/render/tool-hints.js +18 -15
- package/src/render/turn-renderer.js +4 -4
- package/src/reply-emitter.js +2 -2
- package/src/sequence-counter.js +4 -3
- package/src/signature-filter.js +7 -6
- package/src/supervisor.js +19 -18
- package/src/tee-writer.js +25 -25
- package/src/trace-collector.js +53 -48
- package/src/trace-github.js +53 -44
- package/src/trace-multi.js +16 -14
- package/src/trace-query.js +61 -52
- package/src/trace-render.js +19 -19
- package/src/trace-usage.js +31 -28
- package/src/transcript-recorder.js +24 -20
- package/bin/fit-benchmark.js +0 -44
- package/bin/fit-harness.js +0 -412
- package/bin/fit-selfedit.js +0 -165
- package/bin/fit-trace.js +0 -520
package/src/trace-query.js
CHANGED
|
@@ -9,11 +9,11 @@ import {
|
|
|
9
9
|
} from "./trace-usage.js";
|
|
10
10
|
|
|
11
11
|
/**
|
|
12
|
-
* Query engine for structured trace documents
|
|
12
|
+
* Query engine for structured trace documents that TraceCollector produces.
|
|
13
13
|
*
|
|
14
|
-
* Loads a structured JSON trace into memory
|
|
15
|
-
*
|
|
16
|
-
*
|
|
14
|
+
* Loads a structured JSON trace into memory. Provides methods to page,
|
|
15
|
+
* search, filter, and summarize turns. Agents need these operations to
|
|
16
|
+
* analyze large traces efficiently.
|
|
17
17
|
*/
|
|
18
18
|
export class TraceQuery {
|
|
19
19
|
/**
|
|
@@ -54,7 +54,7 @@ export class TraceQuery {
|
|
|
54
54
|
}
|
|
55
55
|
|
|
56
56
|
/**
|
|
57
|
-
* Full system/init event
|
|
57
|
+
* Full system/init event. It is the single most diagnostic message for
|
|
58
58
|
* root-cause analysis. Returns null for traces collected before this
|
|
59
59
|
* field existed.
|
|
60
60
|
* @returns {object|null}
|
|
@@ -73,16 +73,16 @@ export class TraceQuery {
|
|
|
73
73
|
}
|
|
74
74
|
|
|
75
75
|
/**
|
|
76
|
-
* Filter turns by composable structural criteria.
|
|
77
|
-
*
|
|
78
|
-
* shortcuts for
|
|
76
|
+
* Filter turns by composable structural criteria. The filter combines all
|
|
77
|
+
* criteria as AND. `tool()` and `errors()` remain as convenience
|
|
78
|
+
* shortcuts for workflows that already exist.
|
|
79
79
|
*
|
|
80
|
-
* `toolName` matches assistant turns only.
|
|
81
|
-
* `role: "assistant"` still drops every non-assistant turn, because
|
|
82
|
-
*
|
|
83
|
-
* `isError` matches tool_result turns only.
|
|
84
|
-
* `isError`
|
|
85
|
-
*
|
|
80
|
+
* `toolName` matches assistant turns only. `toolName` without
|
|
81
|
+
* `role: "assistant"` still drops every non-assistant turn, because only
|
|
82
|
+
* the `tool()` method resolves tool_use → tool_result pairs.
|
|
83
|
+
* `isError` matches tool_result turns only. So `toolName` with
|
|
84
|
+
* `isError` always returns `[]`. No turn is both assistant and
|
|
85
|
+
* tool_result. Use `tool(name)` for "errors from Bash"–shaped
|
|
86
86
|
* queries.
|
|
87
87
|
*
|
|
88
88
|
* @param {object} [opts]
|
|
@@ -138,15 +138,16 @@ export class TraceQuery {
|
|
|
138
138
|
}
|
|
139
139
|
|
|
140
140
|
/**
|
|
141
|
-
* Search all turn content for a regex pattern.
|
|
142
|
-
*
|
|
141
|
+
* Search all turn content for a regex pattern. Returns the turns that
|
|
142
|
+
* match and highlights the matched text with context.
|
|
143
143
|
*
|
|
144
144
|
* Searches: assistant text blocks, tool_use names and stringified input,
|
|
145
145
|
* and tool_result content.
|
|
146
146
|
*
|
|
147
147
|
* @param {string} pattern - Regex pattern (case-insensitive)
|
|
148
148
|
* @param {object} [opts]
|
|
149
|
-
* @param {number} [opts.context=0] - Number of
|
|
149
|
+
* @param {number} [opts.context=0] - Number of turns to include around the
|
|
150
|
+
* match
|
|
150
151
|
* @param {number} [opts.limit=50] - Max results
|
|
151
152
|
* @param {boolean} [opts.full=false] - Emit full content block text in
|
|
152
153
|
* match descriptions instead of the default narrow excerpt window.
|
|
@@ -197,7 +198,8 @@ export class TraceQuery {
|
|
|
197
198
|
}
|
|
198
199
|
|
|
199
200
|
/**
|
|
200
|
-
* Filter turns
|
|
201
|
+
* Filter turns that involve a specific tool (both the tool_use and its
|
|
202
|
+
* result).
|
|
201
203
|
* @param {string} name - Tool name
|
|
202
204
|
* @returns {object[]}
|
|
203
205
|
*/
|
|
@@ -251,9 +253,9 @@ export class TraceQuery {
|
|
|
251
253
|
}
|
|
252
254
|
|
|
253
255
|
/**
|
|
254
|
-
* Compact one-line-per-assistant-turn timeline
|
|
255
|
-
* reasoning snippet, and token usage.
|
|
256
|
-
* as such
|
|
256
|
+
* Compact one-line-per-assistant-turn timeline with tool names,
|
|
257
|
+
* reasoning snippet, and token usage. The timeline marks thinking-only
|
|
258
|
+
* turns as such. It omits their content (the content is model-internal).
|
|
257
259
|
* @returns {string[]}
|
|
258
260
|
*/
|
|
259
261
|
timeline() {
|
|
@@ -296,16 +298,18 @@ export class TraceQuery {
|
|
|
296
298
|
* totals that name their population.
|
|
297
299
|
*
|
|
298
300
|
* A structured document collected before this change (version < 1.2.0)
|
|
299
|
-
* carries no message identity
|
|
300
|
-
*
|
|
301
|
+
* carries no message identity. It reports its carried last-wins summary
|
|
302
|
+
* and labels it as such. To get corrected figures, run the NDJSON source
|
|
303
|
+
* again.
|
|
301
304
|
*
|
|
302
|
-
* Otherwise
|
|
303
|
-
* accumulated result-event sums (authoritative)
|
|
304
|
-
*
|
|
305
|
-
*
|
|
306
|
-
* (truncated or in-flight) falls
|
|
307
|
-
*
|
|
308
|
-
*
|
|
305
|
+
* Otherwise, when the trace carries result events, the totals are the SDK's
|
|
306
|
+
* accumulated result-event sums (authoritative). The method compares the
|
|
307
|
+
* per-message sums against them. It surfaces any divergence on
|
|
308
|
+
* input/cacheRead/cacheCreation. It never absorbs that divergence
|
|
309
|
+
* silently. A trace with no result event (truncated or in-flight) falls
|
|
310
|
+
* back to the per-message sums. The method flags output as a
|
|
311
|
+
* streaming-snapshot lower bound. It reports cost/duration/turns as
|
|
312
|
+
* unavailable. It never reports a silent 0.
|
|
309
313
|
* @returns {object}
|
|
310
314
|
*/
|
|
311
315
|
stats() {
|
|
@@ -355,8 +359,8 @@ export class TraceQuery {
|
|
|
355
359
|
}
|
|
356
360
|
|
|
357
361
|
/**
|
|
358
|
-
* Stats for a pre-change structured document
|
|
359
|
-
* summary and per-stream-event breakdown
|
|
362
|
+
* Stats for a pre-change structured document. Report the carried last-wins
|
|
363
|
+
* summary and the per-stream-event breakdown. Label each one. Do not claim
|
|
360
364
|
* result-event parity (the document lacks the message identity it needs).
|
|
361
365
|
* @returns {object}
|
|
362
366
|
*/
|
|
@@ -379,8 +383,8 @@ export class TraceQuery {
|
|
|
379
383
|
}
|
|
380
384
|
|
|
381
385
|
/**
|
|
382
|
-
* One record per `tool_use` block
|
|
383
|
-
* (joined by `toolUseId`)
|
|
386
|
+
* One record per `tool_use` block. Each record pairs with its `tool_result`
|
|
387
|
+
* (joined by `toolUseId`). An orphaned call gets `result: null`.
|
|
384
388
|
* @returns {Array<{turnIndex: number, name: string, toolUseId: string, input: object, result: {content: *, isError: boolean}|null}>}
|
|
385
389
|
*/
|
|
386
390
|
toolCalls() {
|
|
@@ -404,8 +408,10 @@ export class TraceQuery {
|
|
|
404
408
|
}
|
|
405
409
|
|
|
406
410
|
/**
|
|
407
|
-
* One record per `Bash` `tool_use` block
|
|
408
|
-
*
|
|
411
|
+
* One record per `Bash` `tool_use` block. Each record carries its command
|
|
412
|
+
* text.
|
|
413
|
+
* @param {string} [re] - Optional regex source to test against
|
|
414
|
+
* `input.command`.
|
|
409
415
|
* @returns {Array<{turnIndex: number, toolUseId: string, command: string}>}
|
|
410
416
|
*/
|
|
411
417
|
commands(re) {
|
|
@@ -434,7 +440,7 @@ export class TraceQuery {
|
|
|
434
440
|
|
|
435
441
|
/**
|
|
436
442
|
* Side-by-side comparison of this trace against another peer `TraceQuery`.
|
|
437
|
-
* Identity (case name, participant) comes from the caller
|
|
443
|
+
* Identity (case name, participant) comes from the caller. The trace
|
|
438
444
|
* carries no filename.
|
|
439
445
|
* @param {TraceQuery} other
|
|
440
446
|
* @param {{aIdentity: {caseName: string, participant: string|null}, bIdentity: {caseName: string, participant: string|null}}} identities
|
|
@@ -476,16 +482,17 @@ export class TraceQuery {
|
|
|
476
482
|
}
|
|
477
483
|
|
|
478
484
|
/**
|
|
479
|
-
* Per-tool token attribution
|
|
480
|
-
* its host turn's usage
|
|
481
|
-
* full usage to the `(no-tool)` bucket.
|
|
482
|
-
* `stats().totals
|
|
483
|
-
* trace carries them, the per-message fallback
|
|
484
|
-
* answer "of the reported total, what share did
|
|
485
|
-
* a separate per-turn re-count
|
|
486
|
-
*
|
|
487
|
-
*
|
|
488
|
-
*
|
|
485
|
+
* Per-tool token attribution. Each `tool_use` block gets an equal share of
|
|
486
|
+
* its host turn's usage. Assistant turns with no `tool_use` block
|
|
487
|
+
* contribute full usage to the `(no-tool)` bucket. The method scales the
|
|
488
|
+
* per-bucket sums onto `stats().totals`, the authoritative population
|
|
489
|
+
* (result-event sums when the trace carries them, the per-message fallback
|
|
490
|
+
* otherwise). So the buckets answer "of the reported total, what share did
|
|
491
|
+
* each tool drive". The method does not make a separate per-turn re-count.
|
|
492
|
+
* Such a re-count drifts from the headline figure. The largest bucket
|
|
493
|
+
* absorbs the rounding residual on each axis. So the input, output, and
|
|
494
|
+
* `costShare` columns each sum to the corresponding `totals` value (and
|
|
495
|
+
* `1.0`) exactly (criterion-6 invariant).
|
|
489
496
|
* @returns {{perTool: Array<{tool: string, turns: number, inputTokens: number, outputTokens: number, costShare: number}>, totals: object}}
|
|
490
497
|
*/
|
|
491
498
|
statsByTool() {
|
|
@@ -496,7 +503,7 @@ export class TraceQuery {
|
|
|
496
503
|
}
|
|
497
504
|
|
|
498
505
|
/**
|
|
499
|
-
* Totals-only view
|
|
506
|
+
* Totals-only view. Returns `stats().totals` with no per-turn array.
|
|
500
507
|
* @returns {{totals: object}}
|
|
501
508
|
*/
|
|
502
509
|
statsSummary() {
|
|
@@ -537,9 +544,10 @@ function matchesToolName(turn, toolName) {
|
|
|
537
544
|
}
|
|
538
545
|
|
|
539
546
|
/**
|
|
540
|
-
* Collect every assistant `tool_use` block
|
|
541
|
-
*
|
|
542
|
-
* `
|
|
547
|
+
* Collect every assistant `tool_use` block and key each one by `toolUseId`.
|
|
548
|
+
* Filter by tool name when the caller supplies a name. This is the shared
|
|
549
|
+
* join-key source for `toolCalls()`, `commands()`, and
|
|
550
|
+
* `collectToolUseIds()`. Insertion order follows turn order.
|
|
543
551
|
* @param {object[]} turns
|
|
544
552
|
* @param {string} [name] - Optional tool-name filter.
|
|
545
553
|
* @returns {Map<string, {turnIndex: number, name: string, input: object}>}
|
|
@@ -633,7 +641,8 @@ function sideSummary(
|
|
|
633
641
|
}
|
|
634
642
|
|
|
635
643
|
/**
|
|
636
|
-
* Search a single turn for regex matches. Returns array of match
|
|
644
|
+
* Search a single turn for regex matches. Returns an array of match
|
|
645
|
+
* descriptions.
|
|
637
646
|
* @param {object} turn
|
|
638
647
|
* @param {RegExp} re
|
|
639
648
|
* @param {boolean} [full=false] - Emit full block text instead of an excerpt.
|
package/src/trace-render.js
CHANGED
|
@@ -1,13 +1,13 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Text renderers for `
|
|
2
|
+
* Text renderers for `gemba-trace` query output.
|
|
3
3
|
*
|
|
4
4
|
* One named export per renderable verb. Each renderer accepts the query result
|
|
5
|
-
* plus `{multi, signatures}` and returns a string. `multi` controls
|
|
6
|
-
* source-attribution
|
|
7
|
-
*
|
|
5
|
+
* plus `{multi, signatures}` and returns a string. `multi` controls the
|
|
6
|
+
* source-attribution prefix (`grep -H` convention). A record-per-line renderer
|
|
7
|
+
* prepends `<basename>:`. A block renderer emits `# <basename>` headers.
|
|
8
8
|
*
|
|
9
|
-
*
|
|
10
|
-
* path
|
|
9
|
+
* This is an internal module. `commands/trace.js` and the tests import it by
|
|
10
|
+
* relative path. `src/index.js` never re-exports it.
|
|
11
11
|
*/
|
|
12
12
|
|
|
13
13
|
/** Collapse newlines/tabs in a value to a single-line, grep-friendly string. */
|
|
@@ -16,7 +16,7 @@ function oneLine(value) {
|
|
|
16
16
|
return str.replace(/[\r\n\t]+/g, " ").trim();
|
|
17
17
|
}
|
|
18
18
|
|
|
19
|
-
/** Group records by their `source` field (multi-file path)
|
|
19
|
+
/** Group records by their `source` field (multi-file path). Keep the order. */
|
|
20
20
|
function groupBySource(records) {
|
|
21
21
|
const groups = new Map();
|
|
22
22
|
for (const record of records) {
|
|
@@ -28,8 +28,8 @@ function groupBySource(records) {
|
|
|
28
28
|
}
|
|
29
29
|
|
|
30
30
|
/**
|
|
31
|
-
* Render record-per-line output
|
|
32
|
-
* multi
|
|
31
|
+
* Render record-per-line output. Prefix each line with `<source>:` when
|
|
32
|
+
* `multi` is true. `lineOf` maps one record to its text line.
|
|
33
33
|
* @param {object[]} records
|
|
34
34
|
* @param {(record: object) => string} lineOf
|
|
35
35
|
* @param {{multi: boolean}} opts
|
|
@@ -42,8 +42,8 @@ function renderLines(records, lineOf, { multi }) {
|
|
|
42
42
|
}
|
|
43
43
|
|
|
44
44
|
/**
|
|
45
|
-
* Render a block per source. `blockOf` maps one record to a multi-line
|
|
46
|
-
*
|
|
45
|
+
* Render a block per source. `blockOf` maps one record to a multi-line
|
|
46
|
+
* string. Multi-file output separates groups with `# <source>` headers.
|
|
47
47
|
* @param {object[]} records
|
|
48
48
|
* @param {(record: object) => string} blockOf
|
|
49
49
|
* @param {{multi: boolean}} opts
|
|
@@ -167,11 +167,11 @@ function multiPrefix(record, { multi }) {
|
|
|
167
167
|
}
|
|
168
168
|
|
|
169
169
|
/**
|
|
170
|
-
* Default renderer for every other renderable verb
|
|
171
|
-
* fields
|
|
172
|
-
*
|
|
173
|
-
*
|
|
174
|
-
* separates source groups with `# <source>` headers (`renderBlocks`
|
|
170
|
+
* Default renderer for every other renderable verb. It emits one record per
|
|
171
|
+
* block. It renders the fields as `key: value` lines. The output has no JSON
|
|
172
|
+
* braces or quotes, so it is grep/awk-friendly and does not parse as JSON.
|
|
173
|
+
* It collapses nested values to a single grep-friendly line. Multi-file
|
|
174
|
+
* output separates source groups with `# <source>` headers (`renderBlocks`
|
|
175
175
|
* convention).
|
|
176
176
|
* @param {object[]|object} result
|
|
177
177
|
* @param {{multi: boolean}} opts
|
|
@@ -183,8 +183,8 @@ export function renderDefault(result, opts = {}) {
|
|
|
183
183
|
}
|
|
184
184
|
|
|
185
185
|
/**
|
|
186
|
-
* Render one record as `key: value` lines. Scalars render verbatim
|
|
187
|
-
* and arrays collapse to a single line
|
|
186
|
+
* Render one record as `key: value` lines. Scalars render verbatim. Objects
|
|
187
|
+
* and arrays collapse to a single line through `oneLine`. A non-object record
|
|
188
188
|
* (string/number) renders as its own single line.
|
|
189
189
|
* @param {*} record
|
|
190
190
|
* @returns {string}
|
|
@@ -201,7 +201,7 @@ function recordBlock(record) {
|
|
|
201
201
|
.join("\n");
|
|
202
202
|
}
|
|
203
203
|
|
|
204
|
-
/** Drop the orchestrator-injected `source` field before
|
|
204
|
+
/** Drop the orchestrator-injected `source` field before text output. */
|
|
205
205
|
function stripSource(record) {
|
|
206
206
|
if (record == null || typeof record !== "object" || Array.isArray(record)) {
|
|
207
207
|
return record;
|
package/src/trace-usage.js
CHANGED
|
@@ -1,11 +1,12 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Token-usage accounting for structured trace documents.
|
|
3
3
|
*
|
|
4
|
-
* `stats()` reports totals that name their population
|
|
5
|
-
* the trace carries them (authoritative)
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
4
|
+
* `stats()` reports totals that name their population. It reports the
|
|
5
|
+
* result-event sums when the trace carries them (authoritative). It falls
|
|
6
|
+
* back to the per-message totals otherwise. For a pre-change document it
|
|
7
|
+
* reports the carried last-wins summary. These helpers compute the
|
|
8
|
+
* per-message accounting. They also report any divergence against the
|
|
9
|
+
* result-event sums. No mismatch disappears in silence.
|
|
9
10
|
*/
|
|
10
11
|
|
|
11
12
|
/** Zero-valued token usage, used as the carried-document fallback. */
|
|
@@ -17,8 +18,8 @@ export const ZERO_USAGE = {
|
|
|
17
18
|
};
|
|
18
19
|
|
|
19
20
|
/**
|
|
20
|
-
* Per-stream-event breakdown for a pre-change document, labeled as carried
|
|
21
|
-
*
|
|
21
|
+
* Per-stream-event breakdown for a pre-change document, labeled as carried.
|
|
22
|
+
* Old documents lack message identity, so rows stay keyed by turn index.
|
|
22
23
|
* @param {object[]} turns
|
|
23
24
|
* @returns {object[]}
|
|
24
25
|
*/
|
|
@@ -40,8 +41,9 @@ export function carriedPerTurn(turns) {
|
|
|
40
41
|
|
|
41
42
|
/**
|
|
42
43
|
* Whether a structured-document version predates per-message accounting
|
|
43
|
-
* (1.2.0). A trace with no version
|
|
44
|
-
*
|
|
44
|
+
* (1.2.0). A trace with no version is not pre-change. This build collects
|
|
45
|
+
* those traces from NDJSON. Compares numeric version parts so 1.10.0 reads
|
|
46
|
+
* as post-change.
|
|
45
47
|
* @param {string|undefined|null} version
|
|
46
48
|
* @returns {boolean}
|
|
47
49
|
*/
|
|
@@ -51,17 +53,17 @@ export function isPreChangeDoc(version) {
|
|
|
51
53
|
.split(".")
|
|
52
54
|
.map((part) => parseInt(part, 10) || 0);
|
|
53
55
|
if (major !== 1) return major < 1;
|
|
54
|
-
// Per-message accounting arrived in 1.2.0
|
|
56
|
+
// Per-message accounting arrived in 1.2.0. Any 1.2.x is post-change.
|
|
55
57
|
return minor < 2;
|
|
56
58
|
}
|
|
57
59
|
|
|
58
60
|
/**
|
|
59
|
-
* Account assistant usage once per API message.
|
|
60
|
-
* `messageId
|
|
61
|
-
* field-wise max across
|
|
62
|
-
* the single value when a message's duplicate snapshots are
|
|
63
|
-
* (zero residual against result-event sums)
|
|
64
|
-
* largest streaming snapshot
|
|
61
|
+
* Account assistant usage once per API message. This function groups turns
|
|
62
|
+
* by `messageId`. A null id is its own singleton message. For each message
|
|
63
|
+
* it takes the field-wise max across the snapshots. The max ignores order.
|
|
64
|
+
* The max equals the single value when a message's duplicate snapshots are
|
|
65
|
+
* byte-identical (zero residual against result-event sums). For output the
|
|
66
|
+
* max is a floor. It is the largest streaming snapshot. It never overstates.
|
|
65
67
|
* @param {object[]} turns
|
|
66
68
|
* @returns {{perMessage: object[], totals: object}}
|
|
67
69
|
*/
|
|
@@ -128,10 +130,11 @@ function accumulateMessage(byMessage, key, turn) {
|
|
|
128
130
|
}
|
|
129
131
|
|
|
130
132
|
/**
|
|
131
|
-
* Compare per-message sums against the result-event sums
|
|
132
|
-
*
|
|
133
|
-
*
|
|
134
|
-
* `{field, perMessageSum, resultEventSum}`, or
|
|
133
|
+
* Compare per-message sums against the result-event sums. The spec
|
|
134
|
+
* guarantees parity for input, cacheRead, and cacheCreation only. Output
|
|
135
|
+
* always diverges by mechanism 2, so the comparison skips it. Returns the
|
|
136
|
+
* first divergent field as `{field, perMessageSum, resultEventSum}`, or
|
|
137
|
+
* null when all agree.
|
|
135
138
|
* @param {object} perMessageTotals
|
|
136
139
|
* @param {object} resultEventUsage
|
|
137
140
|
* @returns {object|null}
|
|
@@ -155,9 +158,9 @@ export function computeDivergence(perMessageTotals, resultEventUsage) {
|
|
|
155
158
|
const NO_TOOL = "(no-tool)";
|
|
156
159
|
|
|
157
160
|
/**
|
|
158
|
-
* Attribute per-turn usage to per-tool buckets
|
|
159
|
-
* equal share of its host turn's usage
|
|
160
|
-
* block
|
|
161
|
+
* Attribute per-turn usage to per-tool buckets. Each `tool_use` block gets
|
|
162
|
+
* an equal share of its host turn's usage. An assistant turn with no
|
|
163
|
+
* `tool_use` block contributes full usage to the `(no-tool)` bucket.
|
|
161
164
|
* @param {object[]} turns
|
|
162
165
|
* @returns {{buckets: Map<string, {inputTokens: number, outputTokens: number}>, bucketTurns: Map<string, Set<number>>}}
|
|
163
166
|
*/
|
|
@@ -192,11 +195,11 @@ export function bucketUsageByTool(turns) {
|
|
|
192
195
|
}
|
|
193
196
|
|
|
194
197
|
/**
|
|
195
|
-
* Scale per-tool buckets onto the headline totals
|
|
196
|
-
* `costShare` columns each sum to the corresponding `totals`
|
|
197
|
-
*
|
|
198
|
-
* carried-document). The largest bucket absorbs
|
|
199
|
-
* axis (criterion-6 invariant).
|
|
198
|
+
* Scale per-tool buckets onto the headline totals. The input, output, and
|
|
199
|
+
* `costShare` columns then each sum exactly to the corresponding `totals`
|
|
200
|
+
* value (and 1.0). This holds for every population (result-event-sum,
|
|
201
|
+
* per-message-fallback, or carried-document). The largest bucket absorbs
|
|
202
|
+
* the rounding residual on each axis (criterion-6 invariant).
|
|
200
203
|
* @param {Map<string, {inputTokens: number, outputTokens: number}>} buckets
|
|
201
204
|
* @param {Map<string, Set<number>>} bucketTurns
|
|
202
205
|
* @param {object} totals
|
|
@@ -1,21 +1,23 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* TranscriptRecorder — per-participant in-memory record of the composed
|
|
3
|
-
* system prompt, delivered prompts, and session messages
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
2
|
+
* TranscriptRecorder — a per-participant in-memory record of the composed
|
|
3
|
+
* system prompt, the delivered prompts, and the session messages. The
|
|
4
|
+
* recorder renders that record into the context text an advisor consult
|
|
5
|
+
* forwards. The harness constructs it only when a session runs with an
|
|
6
|
+
* advisor model. Otherwise the harness keeps no per-participant record.
|
|
7
|
+
* Session lines then go straight to the trace stream.
|
|
7
8
|
*
|
|
8
|
-
*
|
|
9
|
-
* `AgentRunner.#recordLine
|
|
10
|
-
*
|
|
9
|
+
* The redaction path splits. The message tap arrives already redacted from
|
|
10
|
+
* `AgentRunner.#recordLine`. The seeded system prompt and the prompt tap
|
|
11
|
+
* are raw. The recorder redacts those two itself through the injected
|
|
11
12
|
* redactor.
|
|
12
13
|
*/
|
|
13
14
|
|
|
14
15
|
/**
|
|
15
16
|
* Normalize whatever the harness composed as a system prompt into plain
|
|
16
|
-
* text. In practice always a
|
|
17
|
-
*
|
|
18
|
-
*
|
|
17
|
+
* text. In practice it is always a
|
|
18
|
+
* `{type:"preset", preset:"claude_code", append}` object. Every recorded
|
|
19
|
+
* participant is an agent. The spec excludes leads. The function also
|
|
20
|
+
* accepts a plain string. It accepts `undefined` too.
|
|
19
21
|
* @param {string|{type: string, preset?: string, append?: string}|undefined} systemPrompt
|
|
20
22
|
* @returns {string|undefined}
|
|
21
23
|
*/
|
|
@@ -38,8 +40,8 @@ function wrapSection(tag, content) {
|
|
|
38
40
|
*
|
|
39
41
|
* @param {object} deps
|
|
40
42
|
* @param {string|object} [deps.systemPrompt] - The system prompt the harness
|
|
41
|
-
* composed for the participant, as
|
|
42
|
-
* construction.
|
|
43
|
+
* composed for the participant, exactly as it goes to the runner. The
|
|
44
|
+
* value arrives raw. The recorder redacts it at construction.
|
|
43
45
|
* @param {import("./redaction.js").Redactor} deps.redactor
|
|
44
46
|
* @returns {{recordPrompt: (text: string) => void, recordMessage: (line: string) => void, render: () => string}}
|
|
45
47
|
*/
|
|
@@ -56,25 +58,27 @@ export function createTranscriptRecorder({ systemPrompt, redactor }) {
|
|
|
56
58
|
|
|
57
59
|
return {
|
|
58
60
|
/**
|
|
59
|
-
* Record a delivered (amend-applied) prompt.
|
|
61
|
+
* Record a delivered (amend-applied) prompt. The text arrives raw.
|
|
62
|
+
* This method redacts it.
|
|
60
63
|
* @param {string} text
|
|
61
64
|
*/
|
|
62
65
|
recordPrompt(text) {
|
|
63
66
|
prompts.push(redactor.redactValue(text));
|
|
64
67
|
},
|
|
65
68
|
/**
|
|
66
|
-
* Record one NDJSON session line as-is
|
|
67
|
-
* from the runner's line path
|
|
69
|
+
* Record one NDJSON session line as-is. It arrives already redacted
|
|
70
|
+
* from the runner's line path.
|
|
68
71
|
* @param {string} line
|
|
69
72
|
*/
|
|
70
73
|
recordMessage(line) {
|
|
71
74
|
messages.push(line);
|
|
72
75
|
},
|
|
73
76
|
/**
|
|
74
|
-
* Render the record as the advisor's context text
|
|
75
|
-
*
|
|
76
|
-
* NDJSON lines
|
|
77
|
-
* construction
|
|
77
|
+
* Render the record as the advisor's context text. Blank lines join
|
|
78
|
+
* three tagged sections. Each section appears only when it is
|
|
79
|
+
* non-empty. The NDJSON lines stay verbatim. The forwarded context is
|
|
80
|
+
* uncurated by construction, because the spec excludes context-size
|
|
81
|
+
* curation.
|
|
78
82
|
* @returns {string}
|
|
79
83
|
*/
|
|
80
84
|
render() {
|
package/bin/fit-benchmark.js
DELETED
|
@@ -1,44 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
|
|
3
|
-
import "@forwardimpact/libpreflight/node22";
|
|
4
|
-
|
|
5
|
-
import { createCli } from "@forwardimpact/libcli";
|
|
6
|
-
import { createDefaultRuntime } from "@forwardimpact/libutil/runtime";
|
|
7
|
-
import { createLogger } from "@forwardimpact/libtelemetry";
|
|
8
|
-
|
|
9
|
-
import { definition } from "../src/commands/benchmark-definition.js";
|
|
10
|
-
|
|
11
|
-
const runtime = createDefaultRuntime();
|
|
12
|
-
const logger = createLogger("benchmark", runtime);
|
|
13
|
-
|
|
14
|
-
async function main() {
|
|
15
|
-
const cli = createCli(definition, {
|
|
16
|
-
runtime,
|
|
17
|
-
packageJsonUrl: new URL("../package.json", import.meta.url),
|
|
18
|
-
});
|
|
19
|
-
const parsed = cli.parse(runtime.proc.argv.slice(2));
|
|
20
|
-
if (!parsed) return runtime.proc.exit(0);
|
|
21
|
-
|
|
22
|
-
const { positionals } = parsed;
|
|
23
|
-
if (positionals.length === 0) {
|
|
24
|
-
cli.usageError("no command specified");
|
|
25
|
-
return runtime.proc.exit(2);
|
|
26
|
-
}
|
|
27
|
-
|
|
28
|
-
const command = positionals[0];
|
|
29
|
-
if (!definition.commands.some((c) => c.name === command)) {
|
|
30
|
-
cli.usageError(`unknown command "${command}"`);
|
|
31
|
-
return runtime.proc.exit(2);
|
|
32
|
-
}
|
|
33
|
-
|
|
34
|
-
const result = await cli.dispatch(parsed, { deps: { runtime } });
|
|
35
|
-
const envelope = result ?? { ok: true };
|
|
36
|
-
if (!envelope.ok && envelope.error) cli.error(envelope.error);
|
|
37
|
-
runtime.proc.exit(envelope.ok ? 0 : (envelope.code ?? 1));
|
|
38
|
-
}
|
|
39
|
-
|
|
40
|
-
main().catch((error) => {
|
|
41
|
-
logger.exception("main", error);
|
|
42
|
-
createCli(definition, { runtime }).error(error.message);
|
|
43
|
-
process.exit(1);
|
|
44
|
-
});
|