@forwardimpact/libharness 2.0.0 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +68 -65
- package/package.json +15 -13
- package/src/advisor.js +47 -41
- package/src/agent-runner.js +58 -48
- package/src/benchmark/apm-installer.js +28 -28
- package/src/benchmark/env-loader.js +24 -16
- package/src/benchmark/grade.js +44 -41
- package/src/benchmark/hidden-tests.js +25 -24
- package/src/benchmark/hook-env.js +11 -9
- package/src/benchmark/invariants.js +20 -17
- package/src/benchmark/judge.js +29 -28
- package/src/benchmark/npm-installer.js +9 -8
- package/src/benchmark/report.js +53 -50
- package/src/benchmark/result.js +24 -23
- package/src/benchmark/runner.js +75 -69
- package/src/benchmark/scheduler.js +17 -16
- package/src/benchmark/task-family.js +29 -27
- package/src/benchmark/trace-split.js +9 -8
- package/src/benchmark/workdir.js +27 -25
- package/src/claude-code-executable.js +11 -11
- package/src/commands/advisor-flags.js +8 -7
- package/src/commands/assert.js +16 -15
- package/src/commands/benchmark-definition.js +20 -20
- package/src/commands/benchmark-grade.js +13 -12
- package/src/commands/benchmark-report.js +5 -5
- package/src/commands/benchmark-run.js +31 -28
- package/src/commands/by-discussion.js +11 -11
- package/src/commands/callback.js +11 -11
- package/src/commands/discuss.js +8 -7
- package/src/commands/facilitate.js +16 -14
- package/src/commands/output.js +4 -3
- package/src/commands/run.js +15 -15
- package/src/commands/scan-logs.js +22 -20
- package/src/commands/selfedit.js +124 -0
- package/src/commands/supervise.js +13 -11
- package/src/commands/task-input.js +9 -9
- package/src/commands/tee.js +11 -10
- package/src/commands/trace.js +55 -42
- package/src/commands/work-tracker.js +4 -3
- package/src/cost.js +17 -17
- package/src/discuss-tools.js +16 -16
- package/src/discusser.js +39 -38
- package/src/events/github.js +54 -37
- package/src/facilitator.js +21 -21
- package/src/inbox-poller.js +4 -4
- package/src/judge.js +32 -30
- package/src/message-bus.js +12 -11
- package/src/orchestration-loop.js +35 -36
- package/src/orchestration-toolkit.js +58 -53
- package/src/orchestrator-helpers.js +2 -2
- package/src/profile-prompt.js +54 -53
- package/src/redaction.js +63 -57
- package/src/render/line-renderer.js +5 -5
- package/src/render/orchestrator-filter.js +3 -3
- package/src/render/palette.js +11 -9
- package/src/render/tool-hints.js +18 -15
- package/src/render/turn-renderer.js +4 -4
- package/src/reply-emitter.js +2 -2
- package/src/sequence-counter.js +4 -3
- package/src/signature-filter.js +7 -6
- package/src/supervisor.js +19 -18
- package/src/tee-writer.js +25 -25
- package/src/trace-collector.js +53 -48
- package/src/trace-github.js +53 -44
- package/src/trace-multi.js +16 -14
- package/src/trace-query.js +61 -52
- package/src/trace-render.js +19 -19
- package/src/trace-usage.js +31 -28
- package/src/transcript-recorder.js +24 -20
- package/bin/fit-benchmark.js +0 -44
- package/bin/fit-harness.js +0 -412
- package/bin/fit-selfedit.js +0 -165
- package/bin/fit-trace.js +0 -520
package/src/commands/trace.js
CHANGED
|
@@ -20,18 +20,18 @@ import {
|
|
|
20
20
|
// ctx.options — parsed flag values (`cli.parse().values`)
|
|
21
21
|
// ctx.args — named positionals declared on the subcommand
|
|
22
22
|
// ctx.deps — host-injected collaborators: `{ runtime, config }`
|
|
23
|
-
// Handlers read
|
|
24
|
-
// `ctx.deps.runtime
|
|
23
|
+
// Handlers read and write the filesystem and stdout only through
|
|
24
|
+
// `ctx.deps.runtime`. They return `{ ok: true }` on success.
|
|
25
25
|
|
|
26
|
-
/**
|
|
26
|
+
/** These characters mark a `--file` value as a glob. */
|
|
27
27
|
const GLOB_CHARS = /[*?[\]{}]/;
|
|
28
28
|
|
|
29
29
|
/**
|
|
30
30
|
* Resolve the cross-trace `--file` option (`ctx.options.file`) into a sorted
|
|
31
|
-
* flat list of file paths. A literal path passes through
|
|
32
|
-
* glob metacharacters expands
|
|
33
|
-
* fast path means the common single-file and shell-pre-expanded
|
|
34
|
-
* touch `globSync`.
|
|
31
|
+
* flat list of file paths. A literal path passes through. A value that
|
|
32
|
+
* carries glob metacharacters expands through `runtime.fsSync.globSync`. The
|
|
33
|
+
* literal-path fast path means the common single-file and shell-pre-expanded
|
|
34
|
+
* cases never touch `globSync`.
|
|
35
35
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
36
36
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
37
37
|
* @returns {string[]}
|
|
@@ -51,10 +51,11 @@ function resolveFiles(runtime, ctx) {
|
|
|
51
51
|
}
|
|
52
52
|
|
|
53
53
|
/**
|
|
54
|
-
* Emit a query result for a cross-trace verb
|
|
55
|
-
* JSON payload
|
|
56
|
-
* deep-equals today's output
|
|
57
|
-
*
|
|
54
|
+
* Emit a query result for a cross-trace verb. Under `--format json` this
|
|
55
|
+
* function writes the JSON payload. Single-object verbs unwrap when there is
|
|
56
|
+
* one file, so the envelope deep-equals today's output. Otherwise this
|
|
57
|
+
* function renders text to stdout. The renderer owns source attribution.
|
|
58
|
+
* `multi` gates it.
|
|
58
59
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
59
60
|
* @param {object|object[]} result
|
|
60
61
|
* @param {Function} renderer
|
|
@@ -78,7 +79,7 @@ function emit(runtime, result, renderer, ctx, multi, unwrap = false) {
|
|
|
78
79
|
// --- GitHub commands ---
|
|
79
80
|
|
|
80
81
|
/**
|
|
81
|
-
* List recent workflow runs
|
|
82
|
+
* List recent workflow runs that match a pattern.
|
|
82
83
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
83
84
|
*/
|
|
84
85
|
export async function runRunsCommand(ctx) {
|
|
@@ -332,8 +333,8 @@ export async function runStatsCommand(ctx) {
|
|
|
332
333
|
if (files.length === 0) return noFiles("stats");
|
|
333
334
|
const multi = files.length > 1;
|
|
334
335
|
const query = statsQuery(ctx);
|
|
335
|
-
// stats results are per-file objects
|
|
336
|
-
//
|
|
336
|
+
// stats results are per-file objects. Each file gets one block and no
|
|
337
|
+
// cross-file sum. Each block carries a source tag only when multi-file.
|
|
337
338
|
const results = files.map((file) => ({
|
|
338
339
|
result: query(loadTrace(runtime, file)),
|
|
339
340
|
source: multi ? basename(file) : undefined,
|
|
@@ -356,20 +357,22 @@ export async function runStatsCommand(ctx) {
|
|
|
356
357
|
}
|
|
357
358
|
|
|
358
359
|
/**
|
|
359
|
-
*
|
|
360
|
-
* named profile)
|
|
361
|
-
* per source. The combined trace from a
|
|
362
|
-
*
|
|
363
|
-
* run's spend.
|
|
364
|
-
* emits a GitHub-flavored
|
|
360
|
+
* Report the total run cost across every participant (agent, supervisor,
|
|
361
|
+
* judge, and any named profile). The command sums each `result` event in the
|
|
362
|
+
* trace. It attributes the cost per source. The combined trace from a
|
|
363
|
+
* supervised, facilitated, or discuss session already interleaves all
|
|
364
|
+
* participants, so one file yields the whole run's spend. The default output
|
|
365
|
+
* is `{totalCostUsd, bySource}` JSON. `--markdown` emits a GitHub-flavored
|
|
366
|
+
* block to redirect into `$GITHUB_STEP_SUMMARY`.
|
|
365
367
|
*
|
|
366
368
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
367
369
|
*/
|
|
368
370
|
export async function runCostCommand(ctx) {
|
|
369
371
|
const { runtime } = ctx.deps;
|
|
370
|
-
// Tolerate a missing
|
|
371
|
-
// so the trace may not exist
|
|
372
|
-
// nothing and exit 0
|
|
372
|
+
// Tolerate a missing or empty trace. A CI step reports cost under
|
|
373
|
+
// `always()`, so the trace may not exist. The run can fail before it
|
|
374
|
+
// produces one. Print nothing and exit 0. Do not throw. The caller then
|
|
375
|
+
// needs no `if [ -f ]`.
|
|
373
376
|
const file = ctx.args.file;
|
|
374
377
|
if (!file || !runtime.fsSync.existsSync(file)) return { ok: true };
|
|
375
378
|
const cost = computeTraceCost(runtime.fsSync.readFileSync(file, "utf8"));
|
|
@@ -383,7 +386,8 @@ export async function runCostCommand(ctx) {
|
|
|
383
386
|
|
|
384
387
|
/**
|
|
385
388
|
* Render a cost summary as a GitHub-flavored markdown block for a CI step
|
|
386
|
-
* summary
|
|
389
|
+
* summary. The block holds a headline total and a per-participant table. The
|
|
390
|
+
* table lists the highest cost first.
|
|
387
391
|
* @param {{totalCostUsd: number, bySource: Record<string, number>}} cost
|
|
388
392
|
* @returns {string}
|
|
389
393
|
*/
|
|
@@ -391,7 +395,7 @@ function renderCostMarkdown(cost) {
|
|
|
391
395
|
const lines = [
|
|
392
396
|
`### 💰 Run cost: $${cost.totalCostUsd.toFixed(4)}`,
|
|
393
397
|
"",
|
|
394
|
-
"
|
|
398
|
+
"This total covers every participant (agent, supervisor, judge, named profiles).",
|
|
395
399
|
];
|
|
396
400
|
const sources = Object.entries(cost.bySource).sort((a, b) => b[1] - a[1]);
|
|
397
401
|
if (sources.length > 0) {
|
|
@@ -492,21 +496,27 @@ export async function runCompareCommand(ctx) {
|
|
|
492
496
|
|
|
493
497
|
// --- Split command ---
|
|
494
498
|
|
|
495
|
-
/**
|
|
499
|
+
/**
|
|
500
|
+
* A valid source name starts with a lowercase letter. The rest uses lowercase
|
|
501
|
+
* alphanumeric characters or hyphens.
|
|
502
|
+
*/
|
|
496
503
|
const VALID_SOURCE_NAME = /^[a-z][a-z0-9-]*$/;
|
|
497
504
|
|
|
498
|
-
/**
|
|
505
|
+
/**
|
|
506
|
+
* Sources whose name is itself a structural role. The splitter classifies
|
|
507
|
+
* each one into the role it represents.
|
|
508
|
+
*/
|
|
499
509
|
const STRUCTURAL_ROLES = new Set(["agent", "supervisor", "facilitator"]);
|
|
500
510
|
|
|
501
511
|
/**
|
|
502
|
-
* Split a combined NDJSON trace into per-source files
|
|
503
|
-
* `trace--<case>--<participant>.<role>.ndjson` convention.
|
|
512
|
+
* Split a combined NDJSON trace into per-source files. The output names
|
|
513
|
+
* follow the `trace--<case>--<participant>.<role>.ndjson` convention.
|
|
504
514
|
*
|
|
505
515
|
* Each valid envelope source becomes one output file. Structural sources
|
|
506
|
-
* (`agent`, `supervisor`, `facilitator`) classify into the matching role
|
|
507
|
-
* use their own name as participant
|
|
516
|
+
* (`agent`, `supervisor`, `facilitator`) classify into the matching role.
|
|
517
|
+
* They use their own name as participant. Profile-named sources (e.g.
|
|
508
518
|
* `staff-engineer`) classify as agents with the profile in the participant
|
|
509
|
-
* slot.
|
|
519
|
+
* slot. The command drops orchestrator events and invalid source names.
|
|
510
520
|
*
|
|
511
521
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
512
522
|
*/
|
|
@@ -515,9 +525,10 @@ export async function runSplitCommand(ctx) {
|
|
|
515
525
|
const file = ctx.args.file;
|
|
516
526
|
if (!file) return { ok: false, code: 1, error: "split: missing input file" };
|
|
517
527
|
|
|
518
|
-
// `discuss` has the same lead + N-participants shape as `facilitate
|
|
519
|
-
// splitter buckets purely by envelope `source
|
|
520
|
-
//
|
|
528
|
+
// `discuss` has the same lead + N-participants shape as `facilitate`. The
|
|
529
|
+
// splitter buckets purely by envelope `source`, which is mode-independent.
|
|
530
|
+
// So the CLI accepts `discuss` alongside the structural modes. The CLI owns
|
|
531
|
+
// this rule. Callers do not.
|
|
521
532
|
const mode = ctx.options.mode;
|
|
522
533
|
if (!mode) return { ok: false, code: 1, error: "split: --mode is required" };
|
|
523
534
|
if (!["run", "supervise", "facilitate", "discuss"].includes(mode)) {
|
|
@@ -578,8 +589,9 @@ function parseBuckets(content) {
|
|
|
578
589
|
|
|
579
590
|
/**
|
|
580
591
|
* Compute total + per-source cost from raw file content. A structured JSON
|
|
581
|
-
* trace (from `
|
|
582
|
-
* but no per-source split
|
|
592
|
+
* trace (from `gemba-trace download`) carries its total in
|
|
593
|
+
* `summary.totalCostUsd` but no per-source split. `sumTraceCost` sums raw
|
|
594
|
+
* NDJSON.
|
|
583
595
|
* @param {string} content - Raw file content (structured JSON or NDJSON).
|
|
584
596
|
* @returns {{totalCostUsd: number, bySource: Record<string, number>}}
|
|
585
597
|
*/
|
|
@@ -590,7 +602,7 @@ function computeTraceCost(content) {
|
|
|
590
602
|
return { totalCostUsd: parsed.summary.totalCostUsd, bySource: {} };
|
|
591
603
|
}
|
|
592
604
|
} catch {
|
|
593
|
-
// Not a single JSON object
|
|
605
|
+
// Not a single JSON object. Treat it as NDJSON below.
|
|
594
606
|
}
|
|
595
607
|
return sumTraceCost(content.split("\n"));
|
|
596
608
|
}
|
|
@@ -610,7 +622,7 @@ export function loadTrace(runtime, file) {
|
|
|
610
622
|
return createTraceQuery(parsed);
|
|
611
623
|
}
|
|
612
624
|
} catch {
|
|
613
|
-
// Not valid JSON
|
|
625
|
+
// Not valid JSON. Fall through to NDJSON.
|
|
614
626
|
}
|
|
615
627
|
|
|
616
628
|
const collector = createTraceCollector({
|
|
@@ -623,9 +635,10 @@ export function loadTrace(runtime, file) {
|
|
|
623
635
|
}
|
|
624
636
|
|
|
625
637
|
/**
|
|
626
|
-
* Write JSON output to stdout. By default strips
|
|
627
|
-
* base64 blobs from the payload so they
|
|
628
|
-
*
|
|
638
|
+
* Write JSON output to stdout. By default the function strips
|
|
639
|
+
* `thinking.signature` base64 blobs from the payload so they do not dominate
|
|
640
|
+
* terminal output. Pass `--signatures` (surfaced as `values.signatures`) to
|
|
641
|
+
* keep them.
|
|
629
642
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
630
643
|
* @param {*} data
|
|
631
644
|
* @param {object} [values]
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The active work-item tracker selects which column of the work-trackers
|
|
3
3
|
* matrix realizes each coordination operation (see the agent reference
|
|
4
|
-
* `work-trackers.md`). `github` is the production binding
|
|
4
|
+
* `work-trackers.md`). `github` is the production binding. The offline
|
|
5
5
|
* coordination benchmark runs under `filesystem`.
|
|
6
6
|
*/
|
|
7
7
|
export const DEFAULT_WORK_TRACKER = "github";
|
|
@@ -14,10 +14,11 @@ export const KNOWN_WORK_TRACKERS = ["github", "filesystem"];
|
|
|
14
14
|
* flag, then an inherited `LIBHARNESS_WORK_TRACKER` on the environment (so a CI
|
|
15
15
|
* job or harness can select it without the flag), then the `github` default.
|
|
16
16
|
* The harness writes the result to `LIBHARNESS_WORK_TRACKER` on the agent
|
|
17
|
-
* environment
|
|
17
|
+
* environment. This mirrors `--agent-profile` → `LIBHARNESS_AGENT_PROFILE`.
|
|
18
18
|
* @param {Record<string, string|undefined>} values - Parsed option values
|
|
19
19
|
* @param {Record<string, string|undefined>} [env] - Process environment
|
|
20
|
-
* (e.g. `runtime.proc.env`)
|
|
20
|
+
* (e.g. `runtime.proc.env`). The function reads it for the
|
|
21
|
+
* `LIBHARNESS_WORK_TRACKER` fallback.
|
|
21
22
|
* @returns {string}
|
|
22
23
|
* @throws {Error} if the resolved tracker is unknown
|
|
23
24
|
*/
|
package/src/cost.js
CHANGED
|
@@ -1,20 +1,20 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Cost
|
|
3
|
-
*
|
|
2
|
+
* Cost totals over Claude Code NDJSON traces — the single source of truth for
|
|
3
|
+
* the cost of a run across every participant.
|
|
4
4
|
*
|
|
5
5
|
* The SDK reports the cumulative session cost on each `result` event as
|
|
6
6
|
* `total_cost_usd`. Supervised, facilitated, and discuss sessions interleave
|
|
7
|
-
* one runner's events with another's in a single combined trace
|
|
8
|
-
* each in a `{source, seq, event}` envelope
|
|
9
|
-
* events with no envelope. A judge runs as its own session in a separate
|
|
10
|
-
* trace. In every case the rule is the same
|
|
11
|
-
* `result` event
|
|
12
|
-
*
|
|
7
|
+
* one runner's events with another's in a single combined trace. They wrap
|
|
8
|
+
* each event in a `{source, seq, event}` envelope. A plain `run` trace carries
|
|
9
|
+
* bare events with no envelope. A judge runs as its own session in a separate
|
|
10
|
+
* trace. In every case the rule is the same. Sum the `total_cost_usd` of each
|
|
11
|
+
* `result` event. Keep a per-source breakdown so callers can attribute spend
|
|
12
|
+
* to the agent, supervisor, judge, or any named participant.
|
|
13
13
|
*
|
|
14
14
|
* This mirrors `TraceCollector.handleResult`, which accumulates the same
|
|
15
|
-
* figure for its summary footer
|
|
16
|
-
* benchmark runner, the callback command, and `
|
|
17
|
-
* implementation
|
|
15
|
+
* figure for its summary footer. This module stays a standalone pure helper.
|
|
16
|
+
* The benchmark runner, the callback command, and `gemba-trace cost` then
|
|
17
|
+
* share one implementation instead of each one re-deriving it and drifting.
|
|
18
18
|
*/
|
|
19
19
|
|
|
20
20
|
/** Bucket key for bare (un-enveloped) `run`-mode events: a lone agent session. */
|
|
@@ -24,9 +24,9 @@ export const UNSOURCED = "agent";
|
|
|
24
24
|
* Sum `total_cost_usd` across every `result` event in an NDJSON trace.
|
|
25
25
|
*
|
|
26
26
|
* @param {Iterable<string>} lines - NDJSON lines (e.g. `content.split("\n")`).
|
|
27
|
-
*
|
|
27
|
+
* The function skips blank and malformed lines.
|
|
28
28
|
* @returns {{totalCostUsd: number, bySource: Record<string, number>}}
|
|
29
|
-
* `totalCostUsd` is the sum across all participants
|
|
29
|
+
* `totalCostUsd` is the sum across all participants. `bySource` maps each
|
|
30
30
|
* envelope `source` (or {@link UNSOURCED} for bare events) to its subtotal.
|
|
31
31
|
*/
|
|
32
32
|
export function sumTraceCost(lines) {
|
|
@@ -46,9 +46,9 @@ export function sumTraceCost(lines) {
|
|
|
46
46
|
}
|
|
47
47
|
|
|
48
48
|
/**
|
|
49
|
-
* Parse a single NDJSON line
|
|
50
|
-
*
|
|
51
|
-
* no numeric `total_cost_usd`.
|
|
49
|
+
* Parse a single NDJSON line. Return its `result`-event cost contribution.
|
|
50
|
+
* Return null when the line is blank, malformed, not a result event, or
|
|
51
|
+
* carries no numeric `total_cost_usd`.
|
|
52
52
|
*
|
|
53
53
|
* @param {string} line
|
|
54
54
|
* @returns {{source: string, cost: number} | null}
|
|
@@ -64,7 +64,7 @@ function parseCostLine(line) {
|
|
|
64
64
|
return null;
|
|
65
65
|
}
|
|
66
66
|
|
|
67
|
-
// Unwrap the combined-trace envelope {source, seq, event}
|
|
67
|
+
// Unwrap the combined-trace envelope {source, seq, event}. Bare events
|
|
68
68
|
// (plain `run` traces) have a `type` and no `source`.
|
|
69
69
|
let source = UNSOURCED;
|
|
70
70
|
if (event.event && !event.type && typeof event.source === "string") {
|
package/src/discuss-tools.js
CHANGED
|
@@ -5,15 +5,15 @@
|
|
|
5
5
|
* - `Recess` suspends the session with a resumption trigger.
|
|
6
6
|
* - `Adjourn` ends the discussion with a verdict.
|
|
7
7
|
*
|
|
8
|
-
* `Conclude` is absent
|
|
8
|
+
* `Conclude` is absent. Discuss mode ends through Adjourn or Recess.
|
|
9
9
|
*
|
|
10
|
-
* `RequestForComment` is an agent-level coordination tool
|
|
10
|
+
* `RequestForComment` is an agent-level coordination tool. It is available on
|
|
11
11
|
* discuss agents and facilitated agents (not leads). It opens a new
|
|
12
12
|
* Discussion thread for long-horizon coordination on open questions.
|
|
13
13
|
*
|
|
14
|
-
* In discuss mode, each agent Answer routed to the lead
|
|
15
|
-
*
|
|
16
|
-
*
|
|
14
|
+
* In discuss mode, each agent Answer routed to the lead becomes a thread
|
|
15
|
+
* reply. The bridge callback delivers that reply. The lead surface needs no
|
|
16
|
+
* explicit reply tool.
|
|
17
17
|
*/
|
|
18
18
|
|
|
19
19
|
import { tool } from "@anthropic-ai/claude-agent-sdk";
|
|
@@ -30,13 +30,13 @@ import {
|
|
|
30
30
|
requireNoUnprocessedInbox,
|
|
31
31
|
} from "./orchestration-toolkit.js";
|
|
32
32
|
|
|
33
|
-
/** System prompt for discuss-mode agent participants. L0 mechanics only per
|
|
33
|
+
/** System prompt for discuss-mode agent participants. L0 mechanics only per JIDOKA. */
|
|
34
34
|
export const DISCUSS_AGENT_SYSTEM_PROMPT =
|
|
35
35
|
"You are a participant in a discussion.\n" +
|
|
36
36
|
"Each question arrives as `[ask#N] <name>: <text>` in your inbox.\n" +
|
|
37
37
|
"Quote N as askId on your `Answer` to route the reply correctly.\n" +
|
|
38
|
-
"
|
|
39
|
-
"
|
|
38
|
+
"The system posts your `Answer` to the discussion thread as a separate reply.\n" +
|
|
39
|
+
"The task can already contain a completed response with no new human input after it. In that case, `Answer` that no further action is needed.\n" +
|
|
40
40
|
"Do not redo completed work.";
|
|
41
41
|
|
|
42
42
|
const RESUME_TRIGGER_SCHEMA = z.discriminatedUnion("kind", [
|
|
@@ -66,7 +66,7 @@ export function createDiscussLeadToolServer(ctx) {
|
|
|
66
66
|
...baseTools(ctx, { from: "lead", defaultTo: undefined, broadcast: true }),
|
|
67
67
|
tool(
|
|
68
68
|
"Acknowledge",
|
|
69
|
-
"Post a brief message directly to the discussion thread. Use
|
|
69
|
+
"Post a brief message directly to the discussion thread. Use it to respond to a human follow-up. Use it to give a status update while participants work.",
|
|
70
70
|
{
|
|
71
71
|
message: z.string().describe("Message to post on the thread"),
|
|
72
72
|
},
|
|
@@ -104,7 +104,7 @@ export function createDiscussLeadToolServer(ctx) {
|
|
|
104
104
|
}
|
|
105
105
|
|
|
106
106
|
const ACKNOWLEDGE_DESC =
|
|
107
|
-
"Acknowledge an Ask before
|
|
107
|
+
"Acknowledge an Ask before you start work. Posts a visible comment on the thread. Does not discharge the Ask. You still owe an Answer.";
|
|
108
108
|
|
|
109
109
|
/** Discuss-mode agent tool server. */
|
|
110
110
|
export function createDiscussAgentToolServer(ctx, { from, extraTools = [] }) {
|
|
@@ -118,7 +118,7 @@ export function createDiscussAgentToolServer(ctx, { from, extraTools = [] }) {
|
|
|
118
118
|
message: z
|
|
119
119
|
.string()
|
|
120
120
|
.describe("Brief acknowledgement to post on the thread"),
|
|
121
|
-
askId: z.number().optional().describe("The ask
|
|
121
|
+
askId: z.number().optional().describe("The ask you acknowledge"),
|
|
122
122
|
},
|
|
123
123
|
async ({ message }) => {
|
|
124
124
|
const seq =
|
|
@@ -138,11 +138,11 @@ export function createDiscussAgentToolServer(ctx, { from, extraTools = [] }) {
|
|
|
138
138
|
}
|
|
139
139
|
|
|
140
140
|
/**
|
|
141
|
-
* Recess handler — ends the run with a structured pause
|
|
142
|
-
* trigger
|
|
143
|
-
* `concluded` flips true
|
|
144
|
-
* distinguishes them
|
|
145
|
-
*
|
|
141
|
+
* Recess handler — ends the run with a structured pause and a resumption
|
|
142
|
+
* trigger. It cancels any open Asks so askers see a synthetic null answer.
|
|
143
|
+
* `concluded` flips true, the same as Adjourn. The `recessed` verdict
|
|
144
|
+
* distinguishes them. `recessTrigger` carries the resume shape for the
|
|
145
|
+
* bridge.
|
|
146
146
|
*/
|
|
147
147
|
export function createRecessHandler(ctx) {
|
|
148
148
|
return async ({ reason, trigger }) => {
|
package/src/discusser.js
CHANGED
|
@@ -3,14 +3,14 @@
|
|
|
3
3
|
* `OrchestrationLoop`. The lead role uses `DiscussTools` (Adjourn / Recess)
|
|
4
4
|
* instead of the facilitator's Conclude.
|
|
5
5
|
*
|
|
6
|
-
* Discuss mode is a sibling of facilitate mode
|
|
7
|
-
* within-run turn loop
|
|
8
|
-
* role, tool set, system prompts, and participant
|
|
9
|
-
* mode-local.
|
|
6
|
+
* Discuss mode is a sibling of facilitate mode. It is not a subset of it.
|
|
7
|
+
* The two modes share the within-run turn loop through `OrchestrationLoop`.
|
|
8
|
+
* The lead role, the tool set, the system prompts, and the participant names
|
|
9
|
+
* all stay mode-local.
|
|
10
10
|
*
|
|
11
|
-
* Each agent Answer routed to the lead
|
|
12
|
-
*
|
|
13
|
-
*
|
|
11
|
+
* Each agent Answer routed to the lead becomes a thread reply. The bridge
|
|
12
|
+
* callback delivers that reply. The lead surface needs no explicit reply
|
|
13
|
+
* tool.
|
|
14
14
|
*/
|
|
15
15
|
|
|
16
16
|
import { Writable } from "node:stream";
|
|
@@ -40,20 +40,20 @@ import {
|
|
|
40
40
|
import { OrchestrationLoop } from "./orchestration-loop.js";
|
|
41
41
|
import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
|
|
42
42
|
|
|
43
|
-
/** System prompt for the discuss-mode lead. L0 mechanics only per
|
|
43
|
+
/** System prompt for the discuss-mode lead. L0 mechanics only per JIDOKA. */
|
|
44
44
|
export const DISCUSS_SYSTEM_PROMPT =
|
|
45
45
|
"You lead a discussion.\n" +
|
|
46
46
|
"You have no tools to perform work yourself.\n" +
|
|
47
47
|
"Use `RollCall` to list participants.\n" +
|
|
48
48
|
"Use `Ask` to delegate work to the best-suited participant.\n" +
|
|
49
|
-
"Participants are domain experts
|
|
50
|
-
"
|
|
49
|
+
"Participants are domain experts. State the task. Do not state how to do it.\n" +
|
|
50
|
+
"The system posts each participant's `Answer` to the discussion thread as a separate reply.\n" +
|
|
51
51
|
"`Ask` is async and returns {askIds:[N,…]} immediately.\n" +
|
|
52
52
|
"Answers arrive on your next turn as `[answer#N] <participant>: <text>` in your inbox.\n" +
|
|
53
53
|
"End your turn while Asks are pending. The system resumes you when answers arrive.\n" +
|
|
54
54
|
"Multiple `Ask` calls in one turn run participants in parallel.\n" +
|
|
55
|
-
"Use `Acknowledge` to post a brief message directly to the discussion thread
|
|
56
|
-
"
|
|
55
|
+
"Use `Acknowledge` to post a brief message directly to the discussion thread. Use it to respond to human follow-ups. Use it to give status updates while participants work.\n" +
|
|
56
|
+
"To end the discussion, call `Adjourn` with a verdict and summary. Call `Recess` instead only to wait on an external reply or duration.";
|
|
57
57
|
|
|
58
58
|
/**
|
|
59
59
|
* Augment a base orchestration context with discuss-mode fields.
|
|
@@ -78,8 +78,8 @@ const devNull = new Writable({
|
|
|
78
78
|
});
|
|
79
79
|
|
|
80
80
|
/**
|
|
81
|
-
* Async orchestrator for the `discuss` mode.
|
|
82
|
-
* `OrchestrationLoop` for the within-run turns
|
|
81
|
+
* Async orchestrator for the `discuss` mode. It composes an
|
|
82
|
+
* `OrchestrationLoop` for the within-run turns. It owns the discussion id,
|
|
83
83
|
* the resumption trigger, and the discuss-augmented terminal summary.
|
|
84
84
|
*/
|
|
85
85
|
export class Discusser {
|
|
@@ -115,10 +115,11 @@ export class Discusser {
|
|
|
115
115
|
}
|
|
116
116
|
|
|
117
117
|
/**
|
|
118
|
-
* Run the discussion.
|
|
119
|
-
* is set
|
|
120
|
-
* emits the discuss-augmented summary
|
|
121
|
-
* summary
|
|
118
|
+
* Run the discussion. This method emits the meta header first, when a
|
|
119
|
+
* discussion_id is set. It then delegates the within-run loop to
|
|
120
|
+
* `OrchestrationLoop`. It then emits the discuss-augmented summary, which
|
|
121
|
+
* overrides the loop's earlier summary. Trace consumers keep the last
|
|
122
|
+
* summary they see.
|
|
122
123
|
*
|
|
123
124
|
* @param {string} task
|
|
124
125
|
* @returns {Promise<{success: boolean, verdict: string, turns: number, replies: object[], trigger: object|null}>}
|
|
@@ -127,7 +128,7 @@ export class Discusser {
|
|
|
127
128
|
this.#emitMeta();
|
|
128
129
|
|
|
129
130
|
// The loop owns within-run turns. Its emitSummary fires once before
|
|
130
|
-
// run() returns
|
|
131
|
+
// run() returns. Ours replaces it as the last summary line.
|
|
131
132
|
await this.loop.run(task);
|
|
132
133
|
|
|
133
134
|
const verdict = this.ctx.verdict ?? "failed";
|
|
@@ -187,15 +188,15 @@ export class Discusser {
|
|
|
187
188
|
}
|
|
188
189
|
|
|
189
190
|
/**
|
|
190
|
-
* Factory — wires the lead and agent runners with `DiscussTools
|
|
191
|
-
* the `OrchestrationLoop`
|
|
192
|
-
*
|
|
191
|
+
* Factory — wires the lead and agent runners with `DiscussTools`. It builds
|
|
192
|
+
* the `OrchestrationLoop` with `leadName: "lead"` and a discuss-mode protocol
|
|
193
|
+
* tag. It then builds the `Discusser` that wraps the loop.
|
|
193
194
|
*
|
|
194
|
-
* Resume semantics: Recess ends the run
|
|
195
|
-
* `cancelPendingAsks
|
|
196
|
-
* ask so nothing dangles in the trace. The bridge later re-dispatches
|
|
197
|
-
*
|
|
198
|
-
*
|
|
195
|
+
* Resume semantics: Recess ends the run. It cancels any open Asks through
|
|
196
|
+
* `cancelPendingAsks`. It emits a synthetic null answer for each cancelled
|
|
197
|
+
* ask so nothing dangles in the trace. The bridge later re-dispatches the
|
|
198
|
+
* workflow against a fresh context. The human reads the trail of events to
|
|
199
|
+
* decide what to re-ask.
|
|
199
200
|
*
|
|
200
201
|
* @param {object} deps
|
|
201
202
|
* @param {string} [deps.leadProfile]
|
|
@@ -215,8 +216,8 @@ export class Discusser {
|
|
|
215
216
|
* @param {string|null} [deps.callbackUrl]
|
|
216
217
|
* @param {string|null} [deps.inboxUrl]
|
|
217
218
|
* @param {string|null} [deps.correlationId]
|
|
218
|
-
* @param {string} [deps.advisorModel] - Claude model for advisor consults
|
|
219
|
-
* @param {number} [deps.advisorMaxUses] - Session-wide consult budget
|
|
219
|
+
* @param {string} [deps.advisorModel] - Claude model for advisor consults. When absent, the factory offers no Advisor tool.
|
|
220
|
+
* @param {number} [deps.advisorMaxUses] - Session-wide consult budget that all agent participants share (default 3).
|
|
220
221
|
* @returns {Discusser}
|
|
221
222
|
*/
|
|
222
223
|
// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: factory wires N runners + resume hydration paths
|
|
@@ -254,9 +255,9 @@ export function createDiscusser({
|
|
|
254
255
|
discussionId ?? null,
|
|
255
256
|
);
|
|
256
257
|
|
|
257
|
-
// Hydrate resume context
|
|
258
|
-
//
|
|
259
|
-
// with a synthetic null answer, so
|
|
258
|
+
// Hydrate resume context: participants, replies, counters. The code does
|
|
259
|
+
// not restore `pendingAsks` on purpose. Recess cancelled every in-flight
|
|
260
|
+
// Ask with a synthetic null answer, so nothing meaningful remains to carry
|
|
260
261
|
// forward.
|
|
261
262
|
if (resumeContext) {
|
|
262
263
|
if (Array.isArray(resumeContext.participants))
|
|
@@ -292,7 +293,7 @@ export function createDiscusser({
|
|
|
292
293
|
})
|
|
293
294
|
: null;
|
|
294
295
|
|
|
295
|
-
// Intercept answers routed to the lead
|
|
296
|
+
// Intercept answers routed to the lead. Each one becomes a discussion reply.
|
|
296
297
|
const originalAnswer = messageBus.answer.bind(messageBus);
|
|
297
298
|
messageBus.answer = (from, to, text, askId) => {
|
|
298
299
|
if (to === "lead" && from !== "@orchestrator") {
|
|
@@ -319,12 +320,12 @@ export function createDiscusser({
|
|
|
319
320
|
let discusser;
|
|
320
321
|
const leadServer = createDiscussLeadToolServer(ctx);
|
|
321
322
|
|
|
322
|
-
// One budget per session
|
|
323
|
+
// One budget per session. Every agent's Advisor handler shares it.
|
|
323
324
|
const budget = advisorModel ? createAdvisorBudget(advisorMaxUses ?? 3) : null;
|
|
324
325
|
|
|
325
326
|
const agents = resolvedConfigs.map((config) => {
|
|
326
|
-
//
|
|
327
|
-
// composed prompt and tool surface
|
|
327
|
+
// `advisorModel` gates everything advisor-shaped. When it is unset, the
|
|
328
|
+
// composed prompt and tool surface stay byte-identical to today's.
|
|
328
329
|
const systemPrompt = composeSystemPrompt({
|
|
329
330
|
role: "agent",
|
|
330
331
|
profile: config.agentProfile,
|
|
@@ -338,8 +339,8 @@ export function createDiscusser({
|
|
|
338
339
|
let extraTools;
|
|
339
340
|
if (advisorModel) {
|
|
340
341
|
recorder = createTranscriptRecorder({ systemPrompt, redactor });
|
|
341
|
-
// Late-bound through the `let discusser` closure
|
|
342
|
-
// not exist yet when the advisor and tool
|
|
342
|
+
// Late-bound through the `let discusser` closure. The instance does
|
|
343
|
+
// not exist yet when the factory builds the advisor and the tool.
|
|
343
344
|
const advisor = createAdvisor({
|
|
344
345
|
model: advisorModel,
|
|
345
346
|
cwd: config.cwd ?? resolvedLeadCwd,
|