@forwardimpact/libharness 1.3.0 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/fit-harness.js +18 -0
- package/bin/fit-trace.js +10 -0
- package/package.json +1 -1
- package/src/advisor.js +218 -0
- package/src/agent-runner.js +6 -0
- package/src/benchmark/grade.js +222 -0
- package/src/benchmark/hidden-tests.js +180 -0
- package/src/benchmark/invariants.js +9 -10
- package/src/benchmark/judge.js +9 -6
- package/src/benchmark/report.js +150 -56
- package/src/benchmark/result.js +43 -8
- package/src/benchmark/runner.js +99 -109
- package/src/benchmark/task-family.js +139 -26
- package/src/benchmark/trace-split.js +73 -0
- package/src/benchmark/workdir.js +6 -1
- package/src/commands/advisor-flags.js +28 -0
- package/src/commands/assert.js +72 -0
- package/src/commands/benchmark-definition.js +11 -6
- package/src/commands/benchmark-grade.js +82 -0
- package/src/commands/benchmark-report.js +12 -2
- package/src/commands/discuss.js +4 -0
- package/src/commands/facilitate.js +4 -0
- package/src/commands/run.js +162 -67
- package/src/commands/supervise.js +4 -0
- package/src/discuss-tools.js +2 -1
- package/src/discusser.js +65 -10
- package/src/facilitator.js +64 -9
- package/src/index.js +9 -0
- package/src/orchestration-toolkit.js +65 -7
- package/src/supervisor.js +72 -10
- package/src/transcript-recorder.js +94 -0
- package/src/commands/benchmark-invariants.js +0 -73
package/src/commands/assert.js
CHANGED
|
@@ -62,12 +62,74 @@ export function evaluateAssertion(values, args, fsSync) {
|
|
|
62
62
|
|
|
63
63
|
const output = { test: testName, pass: result.pass };
|
|
64
64
|
if (result.message) output.message = result.message;
|
|
65
|
+
applyGradingFlags(values, output);
|
|
65
66
|
return output;
|
|
66
67
|
}
|
|
67
68
|
|
|
69
|
+
/**
|
|
70
|
+
* Attach the check-row grading role: `--gate` marks a gate check, `--weight`
|
|
71
|
+
* attaches a numeric weight (0 marks the row diagnostic). `--gate` with any
|
|
72
|
+
* `--weight` — 0 included — is invalid: a stray weight must never silently
|
|
73
|
+
* disarm a gate.
|
|
74
|
+
* @param {object} values
|
|
75
|
+
* @param {{test: string, pass: boolean, message?: string}} output - Mutated.
|
|
76
|
+
*/
|
|
77
|
+
function applyGradingFlags(values, output) {
|
|
78
|
+
const hasWeight = values.weight !== undefined;
|
|
79
|
+
if (values.gate && hasWeight) {
|
|
80
|
+
throw new Error("assert: --gate cannot be combined with --weight");
|
|
81
|
+
}
|
|
82
|
+
if (values.gate) output.gate = true;
|
|
83
|
+
if (hasWeight) {
|
|
84
|
+
const weight = parseWeight(values.weight);
|
|
85
|
+
if (weight === null) {
|
|
86
|
+
throw new Error(
|
|
87
|
+
`assert: invalid --weight '${values.weight}' (expected a finite number ≥ 0)`,
|
|
88
|
+
);
|
|
89
|
+
}
|
|
90
|
+
output.weight = weight;
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Parse a `--weight` value; null when invalid. A blank string is invalid —
|
|
96
|
+
* `Number("")` is 0, which would silently demote the check to a diagnostic.
|
|
97
|
+
* @param {string} raw
|
|
98
|
+
* @returns {number | null}
|
|
99
|
+
*/
|
|
100
|
+
function parseWeight(raw) {
|
|
101
|
+
if (typeof raw === "string" && raw.trim() === "") return null;
|
|
102
|
+
const weight = Number(raw);
|
|
103
|
+
return Number.isFinite(weight) && weight >= 0 ? weight : null;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* The grading role an emit-then-fail row keeps: a failing check must not
|
|
108
|
+
* lose its authored role — an errored gate that demoted to a scored row
|
|
109
|
+
* would let a broken scaffold earn partial credit instead of zeroing the
|
|
110
|
+
* score. Invalid or conflicting flags yield no role (the row fails as a
|
|
111
|
+
* unit-weight scored check).
|
|
112
|
+
* @param {object} values
|
|
113
|
+
* @returns {{gate?: true, weight?: number}}
|
|
114
|
+
*/
|
|
115
|
+
function errorRowRole(values) {
|
|
116
|
+
const hasWeight = values.weight !== undefined;
|
|
117
|
+
if (values.gate && !hasWeight) return { gate: true };
|
|
118
|
+
if (!values.gate && hasWeight) {
|
|
119
|
+
const weight = parseWeight(values.weight);
|
|
120
|
+
if (weight !== null) return { weight };
|
|
121
|
+
}
|
|
122
|
+
return {};
|
|
123
|
+
}
|
|
124
|
+
|
|
68
125
|
/**
|
|
69
126
|
* Run an assertion, write JSON to stdout, and return a failure envelope when
|
|
70
127
|
* the assertion does not pass.
|
|
128
|
+
*
|
|
129
|
+
* Emit-then-fail on every failure path: an invalid grading flag or an
|
|
130
|
+
* errored evaluation (e.g. `--grep` against a file the agent deleted) writes
|
|
131
|
+
* a failing row before the nonzero exit, so a typo or a vanished target
|
|
132
|
+
* shrinks the score, never the denominator.
|
|
71
133
|
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
72
134
|
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
73
135
|
*/
|
|
@@ -78,6 +140,16 @@ export async function runAssertCommand(ctx) {
|
|
|
78
140
|
try {
|
|
79
141
|
result = evaluateAssertion(ctx.options, args, runtime.fsSync);
|
|
80
142
|
} catch (err) {
|
|
143
|
+
const reason = err.message.startsWith("assert: ")
|
|
144
|
+
? err.message
|
|
145
|
+
: `assert: ${err.message}`;
|
|
146
|
+
const row = {
|
|
147
|
+
test: ctx.args["test-name"] ?? "(missing test name)",
|
|
148
|
+
pass: false,
|
|
149
|
+
...errorRowRole(ctx.options),
|
|
150
|
+
message: reason,
|
|
151
|
+
};
|
|
152
|
+
runtime.proc.stdout.write(JSON.stringify(row) + "\n");
|
|
81
153
|
return { ok: false, code: 1, error: err.message };
|
|
82
154
|
}
|
|
83
155
|
runtime.proc.stdout.write(JSON.stringify(result) + "\n");
|
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
*/
|
|
6
6
|
|
|
7
7
|
import { runBenchmarkRunCommand } from "./benchmark-run.js";
|
|
8
|
-
import {
|
|
8
|
+
import { runBenchmarkGradeCommand } from "./benchmark-grade.js";
|
|
9
9
|
import { runBenchmarkReportCommand } from "./benchmark-report.js";
|
|
10
10
|
import {
|
|
11
11
|
BENCHMARK_AGENT_MODEL,
|
|
@@ -95,11 +95,11 @@ export const definition = {
|
|
|
95
95
|
},
|
|
96
96
|
},
|
|
97
97
|
{
|
|
98
|
-
name: "
|
|
98
|
+
name: "grade",
|
|
99
99
|
args: [],
|
|
100
|
-
handler:
|
|
100
|
+
handler: runBenchmarkGradeCommand,
|
|
101
101
|
description:
|
|
102
|
-
"
|
|
102
|
+
"Grade a single task against a post-run workdir without invoking an agent: run the hidden test suite and the invariants script, then derive the verdict from the check rows (the exit mirrors it).",
|
|
103
103
|
options: {
|
|
104
104
|
family: {
|
|
105
105
|
type: "string",
|
|
@@ -112,7 +112,7 @@ export const definition = {
|
|
|
112
112
|
"run-dir": {
|
|
113
113
|
type: "string",
|
|
114
114
|
description:
|
|
115
|
-
"Post-run directory whose cwd/ subdir is the agent CWD;
|
|
115
|
+
"Post-run directory whose cwd/ subdir is the agent CWD; both producers run against that cwd — the path hooks receive as $AGENT_CWD",
|
|
116
116
|
},
|
|
117
117
|
output: {
|
|
118
118
|
type: "string",
|
|
@@ -140,6 +140,11 @@ export const definition = {
|
|
|
140
140
|
type: "string",
|
|
141
141
|
description: "Output format (json|text, default: json)",
|
|
142
142
|
},
|
|
143
|
+
detail: {
|
|
144
|
+
type: "string",
|
|
145
|
+
description:
|
|
146
|
+
"Text report verbosity (full|compact, default: full). compact omits per-task detail — useful for sharded run summaries.",
|
|
147
|
+
},
|
|
143
148
|
},
|
|
144
149
|
},
|
|
145
150
|
],
|
|
@@ -154,7 +159,7 @@ export const definition = {
|
|
|
154
159
|
"fit-benchmark run --family=./families/coding --work-tracker=filesystem",
|
|
155
160
|
"fit-benchmark run --family=./families/coding --skills-from=. --task=todo-api",
|
|
156
161
|
`fit-benchmark run --family=./families/coding --runs=10 --agent-model=${BENCHMARK_AGENT_MODEL}`,
|
|
157
|
-
"fit-benchmark
|
|
162
|
+
"fit-benchmark grade --family=./families/coding --task=todo-api --run-dir=./benchmark-runs/runs/todo-api/0",
|
|
158
163
|
"fit-benchmark report --format=text",
|
|
159
164
|
"fit-benchmark report --input=./runs/today --k=1,3,5 --format=text",
|
|
160
165
|
],
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `fit-benchmark grade` — run both check-row producers (the hidden test
|
|
3
|
+
* suite and the invariants script) against a post-run workdir directory and
|
|
4
|
+
* grade the merged rows with the same derivation the benchmark runner uses.
|
|
5
|
+
* No agent and no judge run, so authors validate a task's grading material
|
|
6
|
+
* against fixtures without paying for agent sessions; the process exit
|
|
7
|
+
* mirrors the graded verdict.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { join, resolve } from "node:path";
|
|
11
|
+
|
|
12
|
+
import { validateGradeRecord } from "../benchmark/result.js";
|
|
13
|
+
import { runInvariants } from "../benchmark/invariants.js";
|
|
14
|
+
import { runHiddenTests } from "../benchmark/hidden-tests.js";
|
|
15
|
+
import { runProducersAndGrade } from "../benchmark/grade.js";
|
|
16
|
+
import { loadTaskFamily } from "../benchmark/task-family.js";
|
|
17
|
+
import { probeFreePort } from "../benchmark/workdir.js";
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* @param {import("@forwardimpact/libcli").InvocationContext} ctx
|
|
21
|
+
* @returns {Promise<{ok: true} | {ok: false, code: number, error: string}>}
|
|
22
|
+
*/
|
|
23
|
+
export async function runBenchmarkGradeCommand(ctx) {
|
|
24
|
+
const values = ctx.options;
|
|
25
|
+
const runtime = ctx.deps.runtime;
|
|
26
|
+
const familyInput = values.family;
|
|
27
|
+
if (!familyInput)
|
|
28
|
+
return { ok: false, code: 1, error: "--family is required" };
|
|
29
|
+
const taskId = values.task;
|
|
30
|
+
if (!taskId) return { ok: false, code: 1, error: "--task is required" };
|
|
31
|
+
const runDirArg = values["run-dir"];
|
|
32
|
+
if (!runDirArg) return { ok: false, code: 1, error: "--run-dir is required" };
|
|
33
|
+
|
|
34
|
+
const family = await loadTaskFamily(familyInput, runtime);
|
|
35
|
+
const task = family.tasks().find((t) => t.id === taskId);
|
|
36
|
+
if (!task)
|
|
37
|
+
return { ok: false, code: 1, error: `task not found in family: ${taskId}` };
|
|
38
|
+
|
|
39
|
+
const runDir = resolve(runDirArg);
|
|
40
|
+
const cwd = join(runDir, "cwd");
|
|
41
|
+
const port = await probeFreePort();
|
|
42
|
+
const cellCtx = { cwd, port, runDir, familyDir: family.rootPath };
|
|
43
|
+
|
|
44
|
+
const { invariants, hiddenRows, engineError, healthy, grade } =
|
|
45
|
+
await runProducersAndGrade(task, cellCtx, runtime, {
|
|
46
|
+
runInvariants,
|
|
47
|
+
runHiddenTests,
|
|
48
|
+
});
|
|
49
|
+
// Same effective-score rule as the runner, minus the judge (none runs
|
|
50
|
+
// here): an unhealthy grader or a failing gate zeroes the score, so a
|
|
51
|
+
// crashed hook can never mint marks from the rows it emitted before dying.
|
|
52
|
+
// Unlike a runner record — where `grade.score` stays the raw fraction and
|
|
53
|
+
// the zeroing lands on the top-level `score` — this record has no second
|
|
54
|
+
// field, so `grade.score` carries the effective value here.
|
|
55
|
+
if (grade.score !== undefined && !(healthy && grade.gatesPass)) {
|
|
56
|
+
grade.score = 0;
|
|
57
|
+
}
|
|
58
|
+
const record = {
|
|
59
|
+
taskId: task.id,
|
|
60
|
+
grade,
|
|
61
|
+
invariants,
|
|
62
|
+
...(task.tests && {
|
|
63
|
+
hiddenTests: {
|
|
64
|
+
details: hiddenRows,
|
|
65
|
+
...(engineError && { error: engineError.message }),
|
|
66
|
+
},
|
|
67
|
+
}),
|
|
68
|
+
// Mirrors the script for diagnosis; the graded verdict drives the exit.
|
|
69
|
+
exitCode: invariants.exitCode,
|
|
70
|
+
};
|
|
71
|
+
validateGradeRecord(record);
|
|
72
|
+
|
|
73
|
+
const line = JSON.stringify(record) + "\n";
|
|
74
|
+
if (values.output) {
|
|
75
|
+
runtime.fsSync.writeFileSync(resolve(values.output), line);
|
|
76
|
+
} else {
|
|
77
|
+
runtime.proc.stdout.write(line);
|
|
78
|
+
}
|
|
79
|
+
return grade.verdict === "pass"
|
|
80
|
+
? { ok: true }
|
|
81
|
+
: { ok: false, code: 1, error: "" };
|
|
82
|
+
}
|
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* `fit-benchmark report` — aggregate `results.jsonl` into pass@k via the
|
|
3
3
|
* OpenAI HumanEval estimator. Output is JSON by default; pass --format=text
|
|
4
|
-
* to render a markdown table.
|
|
4
|
+
* to render a markdown table. --detail=compact drops the per-task detail
|
|
5
|
+
* sections so a sharded run's per-shard summary stays short (the merge job
|
|
6
|
+
* renders the full report over the combined ledger).
|
|
5
7
|
*/
|
|
6
8
|
|
|
7
9
|
import { resolve } from "node:path";
|
|
@@ -35,11 +37,19 @@ export async function runBenchmarkReportCommand(ctx) {
|
|
|
35
37
|
if (format !== "json" && format !== "text") {
|
|
36
38
|
return { ok: false, code: 1, error: "--format must be 'json' or 'text'" };
|
|
37
39
|
}
|
|
40
|
+
const detail = values.detail ?? "full";
|
|
41
|
+
if (detail !== "full" && detail !== "compact") {
|
|
42
|
+
return {
|
|
43
|
+
ok: false,
|
|
44
|
+
code: 1,
|
|
45
|
+
error: "--detail must be 'full' or 'compact'",
|
|
46
|
+
};
|
|
47
|
+
}
|
|
38
48
|
|
|
39
49
|
const report = await aggregate({
|
|
40
50
|
inputDir: resolve(inputDir),
|
|
41
51
|
kValues,
|
|
42
|
-
includeRuns: format === "text",
|
|
52
|
+
includeRuns: format === "text" && detail === "full",
|
|
43
53
|
runtime,
|
|
44
54
|
});
|
|
45
55
|
if (format === "text") {
|
package/src/commands/discuss.js
CHANGED
|
@@ -3,6 +3,7 @@ import { isoTimestamp } from "@forwardimpact/libutil";
|
|
|
3
3
|
import { createDiscusser } from "../discusser.js";
|
|
4
4
|
import { createRedactor } from "../redaction.js";
|
|
5
5
|
import { createTeeWriter } from "../tee-writer.js";
|
|
6
|
+
import { parseAdvisorOptions } from "./advisor-flags.js";
|
|
6
7
|
import { resolveTaskContent } from "./task-input.js";
|
|
7
8
|
import { resolveWorkTracker } from "./work-tracker.js";
|
|
8
9
|
import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
|
|
@@ -67,6 +68,7 @@ export function parseDiscussOptions(values, runtime) {
|
|
|
67
68
|
callbackUrl: runtime.proc.env.CALLBACK_URL ?? null,
|
|
68
69
|
inboxUrl: runtime.proc.env.INBOX_URL ?? null,
|
|
69
70
|
correlationId: runtime.proc.env.CORRELATION_ID ?? null,
|
|
71
|
+
...parseAdvisorOptions(values),
|
|
70
72
|
};
|
|
71
73
|
}
|
|
72
74
|
|
|
@@ -121,6 +123,8 @@ export async function runDiscussCommand(ctx) {
|
|
|
121
123
|
inboxUrl: opts.inboxUrl,
|
|
122
124
|
correlationId: opts.correlationId,
|
|
123
125
|
runtime,
|
|
126
|
+
advisorModel: opts.advisorModel,
|
|
127
|
+
advisorMaxUses: opts.advisorMaxUses,
|
|
124
128
|
});
|
|
125
129
|
|
|
126
130
|
const result = await discusser.run(opts.taskContent);
|
|
@@ -3,6 +3,7 @@ import { isoTimestamp } from "@forwardimpact/libutil";
|
|
|
3
3
|
import { createFacilitator } from "../facilitator.js";
|
|
4
4
|
import { createRedactor } from "../redaction.js";
|
|
5
5
|
import { createTeeWriter } from "../tee-writer.js";
|
|
6
|
+
import { parseAdvisorOptions } from "./advisor-flags.js";
|
|
6
7
|
import { resolveTaskContent } from "./task-input.js";
|
|
7
8
|
import { resolveWorkTracker } from "./work-tracker.js";
|
|
8
9
|
import { AGENT_MODEL, LEAD_MODEL } from "@forwardimpact/libutil/models";
|
|
@@ -60,6 +61,7 @@ export function parseFacilitateOptions(values, runtime) {
|
|
|
60
61
|
outputPath: values.output,
|
|
61
62
|
facilitatorProfile: values["lead-profile"] || undefined,
|
|
62
63
|
workTracker: resolveWorkTracker(values, runtime?.proc?.env),
|
|
64
|
+
...parseAdvisorOptions(values),
|
|
63
65
|
};
|
|
64
66
|
}
|
|
65
67
|
|
|
@@ -112,6 +114,8 @@ export async function runFacilitateCommand(ctx) {
|
|
|
112
114
|
taskAmend: opts.taskAmend,
|
|
113
115
|
redactor,
|
|
114
116
|
runtime,
|
|
117
|
+
advisorModel: opts.advisorModel,
|
|
118
|
+
advisorMaxUses: opts.advisorMaxUses,
|
|
115
119
|
});
|
|
116
120
|
|
|
117
121
|
const result = await facilitator.run(opts.taskContent);
|
package/src/commands/run.js
CHANGED
|
@@ -1,11 +1,23 @@
|
|
|
1
1
|
import { Writable } from "node:stream";
|
|
2
2
|
import { resolve } from "node:path";
|
|
3
3
|
import { isoTimestamp } from "@forwardimpact/libutil";
|
|
4
|
+
import { createSdkMcpServer } from "@anthropic-ai/claude-agent-sdk";
|
|
4
5
|
import { createAgentRunner } from "../agent-runner.js";
|
|
5
|
-
import {
|
|
6
|
+
import {
|
|
7
|
+
advisorGuidance,
|
|
8
|
+
createAdvisor,
|
|
9
|
+
createAdvisorBudget,
|
|
10
|
+
} from "../advisor.js";
|
|
11
|
+
import { advisorTool } from "../orchestration-toolkit.js";
|
|
12
|
+
import {
|
|
13
|
+
composeProfilePrompt,
|
|
14
|
+
composeSystemPrompt,
|
|
15
|
+
} from "../profile-prompt.js";
|
|
6
16
|
import { createRedactor } from "../redaction.js";
|
|
7
17
|
import { createTeeWriter } from "../tee-writer.js";
|
|
18
|
+
import { createTranscriptRecorder } from "../transcript-recorder.js";
|
|
8
19
|
import { SequenceCounter } from "../sequence-counter.js";
|
|
20
|
+
import { parseAdvisorOptions } from "./advisor-flags.js";
|
|
9
21
|
import { resolveWorkTracker } from "./work-tracker.js";
|
|
10
22
|
import { resolveTaskContent } from "./task-input.js";
|
|
11
23
|
import { createServiceConfig } from "@forwardimpact/libconfig";
|
|
@@ -15,7 +27,7 @@ import { AGENT_MODEL } from "@forwardimpact/libutil/models";
|
|
|
15
27
|
* Parse and validate run command options from parsed values.
|
|
16
28
|
* @param {object} values - Parsed option values from cli.parse()
|
|
17
29
|
* @param {import("@forwardimpact/libutil/runtime").Runtime} runtime
|
|
18
|
-
* @returns {{ taskContent: string, cwd: string,
|
|
30
|
+
* @returns {{ taskContent: string, taskAmend: string|undefined, cwd: string, agentModel: string, maxTurns: number, outputPath: string|undefined, agentProfile: string|undefined, workTracker: string, allowedTools: string[], mcpServer: string|undefined, advisorModel: string|undefined, advisorMaxUses: number }}
|
|
19
31
|
*/
|
|
20
32
|
export function parseRunOptions(values, runtime) {
|
|
21
33
|
const { task: taskContent, amend: taskAmend } = resolveTaskContent(
|
|
@@ -40,9 +52,148 @@ export function parseRunOptions(values, runtime) {
|
|
|
40
52
|
"Bash,Read,Glob,Grep,Write,Edit,Agent,TodoWrite"
|
|
41
53
|
).split(","),
|
|
42
54
|
mcpServer: values["mcp-server"] || undefined,
|
|
55
|
+
...parseAdvisorOptions(values),
|
|
43
56
|
};
|
|
44
57
|
}
|
|
45
58
|
|
|
59
|
+
const devNull = new Writable({
|
|
60
|
+
write(_chunk, _enc, cb) {
|
|
61
|
+
cb();
|
|
62
|
+
},
|
|
63
|
+
});
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Wire the run-mode agent session: external MCP entry, `LIBHARNESS_*` env
|
|
67
|
+
* writes, system-prompt composition, and — when an advisor model is set —
|
|
68
|
+
* the advisor wiring (budget, recorder, advisor session, dedicated MCP
|
|
69
|
+
* server holding only the `Advisor` tool). Extracted from `runRunCommand`
|
|
70
|
+
* so tests can inject a fake `query`.
|
|
71
|
+
*
|
|
72
|
+
* Run mode has no stop path (the command simply awaits the runner), so the
|
|
73
|
+
* consult timeout is deliberately the advisor's only guard.
|
|
74
|
+
*
|
|
75
|
+
* @param {object} deps
|
|
76
|
+
* @param {ReturnType<typeof parseRunOptions>} deps.opts
|
|
77
|
+
* @param {import("../redaction.js").Redactor} deps.redactor
|
|
78
|
+
* @param {import("stream").Writable} deps.output - Envelope NDJSON sink.
|
|
79
|
+
* @param {SequenceCounter} deps.counter
|
|
80
|
+
* @param {function} deps.query - SDK query function.
|
|
81
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} deps.runtime
|
|
82
|
+
* @returns {Promise<{runner: import("../agent-runner.js").AgentRunner, advisor: object|null}>}
|
|
83
|
+
*/
|
|
84
|
+
export async function wireRunSession({
|
|
85
|
+
opts,
|
|
86
|
+
redactor,
|
|
87
|
+
output,
|
|
88
|
+
counter,
|
|
89
|
+
query,
|
|
90
|
+
runtime,
|
|
91
|
+
}) {
|
|
92
|
+
const emitEnvelope = (source, event) => {
|
|
93
|
+
output.write(
|
|
94
|
+
JSON.stringify(
|
|
95
|
+
redactor.redactValue({ source, seq: counter.next(), event }),
|
|
96
|
+
) + "\n",
|
|
97
|
+
);
|
|
98
|
+
};
|
|
99
|
+
const onLine = (line) => emitEnvelope("agent", JSON.parse(line));
|
|
100
|
+
|
|
101
|
+
let mcpServers = null;
|
|
102
|
+
const allowedTools = opts.allowedTools;
|
|
103
|
+
if (opts.mcpServer) {
|
|
104
|
+
const mcpConfig = await createServiceConfig("mcp");
|
|
105
|
+
mcpServers = {
|
|
106
|
+
[opts.mcpServer]: {
|
|
107
|
+
type: "http",
|
|
108
|
+
url: mcpConfig.url,
|
|
109
|
+
headers: { Authorization: `Bearer ${mcpConfig.mcpToken()}` },
|
|
110
|
+
},
|
|
111
|
+
};
|
|
112
|
+
allowedTools.push(`mcp__${opts.mcpServer}__*`);
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
if (opts.agentProfile) {
|
|
116
|
+
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = opts.agentProfile;
|
|
117
|
+
}
|
|
118
|
+
// Unconditional so the default "github" is observable to the agent's
|
|
119
|
+
// active-tracker resolution, mirroring --agent-profile's env write above.
|
|
120
|
+
runtime.proc.env.LIBHARNESS_WORK_TRACKER = opts.workTracker;
|
|
121
|
+
|
|
122
|
+
// With a profile, the consult guidance rides the profile composer's
|
|
123
|
+
// amendment parameter; with no profile, a preset-append prompt carries
|
|
124
|
+
// the guidance as its only session-protocol fragment. Advisor off and no
|
|
125
|
+
// profile means no system prompt — today's behavior, unchanged.
|
|
126
|
+
let systemPrompt;
|
|
127
|
+
if (opts.agentProfile) {
|
|
128
|
+
systemPrompt = composeProfilePrompt(opts.agentProfile, {
|
|
129
|
+
profilesDir: resolve(opts.cwd, ".claude/agents"),
|
|
130
|
+
runtime,
|
|
131
|
+
...(opts.advisorModel && {
|
|
132
|
+
amend: advisorGuidance(opts.advisorMaxUses),
|
|
133
|
+
}),
|
|
134
|
+
});
|
|
135
|
+
} else if (opts.advisorModel) {
|
|
136
|
+
systemPrompt = composeSystemPrompt({
|
|
137
|
+
role: "agent",
|
|
138
|
+
trailer: advisorGuidance(opts.advisorMaxUses),
|
|
139
|
+
runtime,
|
|
140
|
+
});
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
let advisor = null;
|
|
144
|
+
let recorder = null;
|
|
145
|
+
if (opts.advisorModel) {
|
|
146
|
+
const budget = createAdvisorBudget(opts.advisorMaxUses);
|
|
147
|
+
recorder = createTranscriptRecorder({ systemPrompt, redactor });
|
|
148
|
+
advisor = createAdvisor({
|
|
149
|
+
model: opts.advisorModel,
|
|
150
|
+
cwd: opts.cwd,
|
|
151
|
+
query,
|
|
152
|
+
recorder,
|
|
153
|
+
redactor,
|
|
154
|
+
runtime,
|
|
155
|
+
onLine: (line) => emitEnvelope("advisor", JSON.parse(line)),
|
|
156
|
+
});
|
|
157
|
+
const advTool = advisorTool({
|
|
158
|
+
from: "agent",
|
|
159
|
+
consult: (q) => advisor.consult(q),
|
|
160
|
+
emit: (event) => emitEnvelope("orchestrator", event),
|
|
161
|
+
budget,
|
|
162
|
+
model: opts.advisorModel,
|
|
163
|
+
});
|
|
164
|
+
// No allowlist push: in-process SDK MCP servers work under
|
|
165
|
+
// bypassPermissions without allowlist entries (loop-mode precedent).
|
|
166
|
+
mcpServers = {
|
|
167
|
+
...mcpServers,
|
|
168
|
+
advisor: createSdkMcpServer({ name: "advisor", tools: [advTool] }),
|
|
169
|
+
};
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
const runner = createAgentRunner({
|
|
173
|
+
cwd: opts.cwd,
|
|
174
|
+
query,
|
|
175
|
+
output: devNull,
|
|
176
|
+
model: opts.agentModel,
|
|
177
|
+
maxTurns: opts.maxTurns,
|
|
178
|
+
allowedTools,
|
|
179
|
+
onLine: recorder
|
|
180
|
+
? (line) => {
|
|
181
|
+
onLine(line);
|
|
182
|
+
recorder.recordMessage(line);
|
|
183
|
+
}
|
|
184
|
+
: onLine,
|
|
185
|
+
...(recorder && { onPrompt: (text) => recorder.recordPrompt(text) }),
|
|
186
|
+
settingSources: ["project"],
|
|
187
|
+
systemPrompt,
|
|
188
|
+
taskAmend: opts.taskAmend,
|
|
189
|
+
mcpServers,
|
|
190
|
+
redactor,
|
|
191
|
+
runtime,
|
|
192
|
+
});
|
|
193
|
+
|
|
194
|
+
return { runner, advisor };
|
|
195
|
+
}
|
|
196
|
+
|
|
46
197
|
/**
|
|
47
198
|
* Run command — execute a single agent via the Claude Agent SDK.
|
|
48
199
|
*
|
|
@@ -53,18 +204,7 @@ export function parseRunOptions(values, runtime) {
|
|
|
53
204
|
*/
|
|
54
205
|
export async function runRunCommand(ctx) {
|
|
55
206
|
const runtime = ctx.deps.runtime;
|
|
56
|
-
const
|
|
57
|
-
taskContent,
|
|
58
|
-
taskAmend,
|
|
59
|
-
cwd,
|
|
60
|
-
agentModel,
|
|
61
|
-
maxTurns,
|
|
62
|
-
outputPath,
|
|
63
|
-
agentProfile,
|
|
64
|
-
workTracker,
|
|
65
|
-
allowedTools,
|
|
66
|
-
mcpServer,
|
|
67
|
-
} = parseRunOptions(ctx.options, runtime);
|
|
207
|
+
const opts = parseRunOptions(ctx.options, runtime);
|
|
68
208
|
|
|
69
209
|
// Build the redactor as the first observable side-effect after option
|
|
70
210
|
// parsing — the env snapshot must freeze BEFORE any in-process
|
|
@@ -73,8 +213,8 @@ export async function runRunCommand(ctx) {
|
|
|
73
213
|
|
|
74
214
|
// When --output is specified, stream text to stdout while writing NDJSON to file.
|
|
75
215
|
// Otherwise, write NDJSON directly to stdout (backwards-compatible).
|
|
76
|
-
const fileStream = outputPath
|
|
77
|
-
? runtime.fs.createWriteStream(outputPath)
|
|
216
|
+
const fileStream = opts.outputPath
|
|
217
|
+
? runtime.fs.createWriteStream(opts.outputPath)
|
|
78
218
|
: null;
|
|
79
219
|
const output = fileStream
|
|
80
220
|
? createTeeWriter({
|
|
@@ -86,62 +226,17 @@ export async function runRunCommand(ctx) {
|
|
|
86
226
|
: runtime.proc.stdout;
|
|
87
227
|
|
|
88
228
|
const counter = new SequenceCounter();
|
|
89
|
-
const devNull = new Writable({
|
|
90
|
-
write(_chunk, _enc, cb) {
|
|
91
|
-
cb();
|
|
92
|
-
},
|
|
93
|
-
});
|
|
94
|
-
const onLine = (line) => {
|
|
95
|
-
const event = JSON.parse(line);
|
|
96
|
-
const tagged = { source: "agent", seq: counter.next(), event };
|
|
97
|
-
output.write(JSON.stringify(redactor.redactValue(tagged)) + "\n");
|
|
98
|
-
};
|
|
99
|
-
|
|
100
|
-
let mcpServers = null;
|
|
101
|
-
if (mcpServer) {
|
|
102
|
-
const mcpConfig = await createServiceConfig("mcp");
|
|
103
|
-
mcpServers = {
|
|
104
|
-
[mcpServer]: {
|
|
105
|
-
type: "http",
|
|
106
|
-
url: mcpConfig.url,
|
|
107
|
-
headers: { Authorization: `Bearer ${mcpConfig.mcpToken()}` },
|
|
108
|
-
},
|
|
109
|
-
};
|
|
110
|
-
allowedTools.push(`mcp__${mcpServer}__*`);
|
|
111
|
-
}
|
|
112
|
-
|
|
113
|
-
if (agentProfile) {
|
|
114
|
-
runtime.proc.env.LIBHARNESS_AGENT_PROFILE = agentProfile;
|
|
115
|
-
}
|
|
116
|
-
// Unconditional so the default "github" is observable to the agent's
|
|
117
|
-
// active-tracker resolution, mirroring --agent-profile's env write above.
|
|
118
|
-
runtime.proc.env.LIBHARNESS_WORK_TRACKER = workTracker;
|
|
119
|
-
|
|
120
|
-
const systemPrompt = agentProfile
|
|
121
|
-
? composeProfilePrompt(agentProfile, {
|
|
122
|
-
profilesDir: resolve(cwd, ".claude/agents"),
|
|
123
|
-
runtime,
|
|
124
|
-
})
|
|
125
|
-
: undefined;
|
|
126
|
-
|
|
127
229
|
const { query } = await import("@anthropic-ai/claude-agent-sdk");
|
|
128
|
-
const runner =
|
|
129
|
-
|
|
130
|
-
query,
|
|
131
|
-
output: devNull,
|
|
132
|
-
model: agentModel,
|
|
133
|
-
maxTurns,
|
|
134
|
-
allowedTools,
|
|
135
|
-
onLine,
|
|
136
|
-
settingSources: ["project"],
|
|
137
|
-
systemPrompt,
|
|
138
|
-
taskAmend,
|
|
139
|
-
mcpServers,
|
|
230
|
+
const { runner } = await wireRunSession({
|
|
231
|
+
opts,
|
|
140
232
|
redactor,
|
|
233
|
+
output,
|
|
234
|
+
counter,
|
|
235
|
+
query,
|
|
141
236
|
runtime,
|
|
142
237
|
});
|
|
143
238
|
|
|
144
|
-
const result = await runner.run(taskContent);
|
|
239
|
+
const result = await runner.run(opts.taskContent);
|
|
145
240
|
|
|
146
241
|
if (fileStream) {
|
|
147
242
|
await new Promise((r) => output.end(r));
|
|
@@ -3,6 +3,7 @@ import { isoTimestamp } from "@forwardimpact/libutil";
|
|
|
3
3
|
import { createSupervisor } from "../supervisor.js";
|
|
4
4
|
import { createRedactor } from "../redaction.js";
|
|
5
5
|
import { createTeeWriter } from "../tee-writer.js";
|
|
6
|
+
import { parseAdvisorOptions } from "./advisor-flags.js";
|
|
6
7
|
import { resolveTaskContent } from "./task-input.js";
|
|
7
8
|
import { resolveWorkTracker } from "./work-tracker.js";
|
|
8
9
|
import { createServiceConfig } from "@forwardimpact/libconfig";
|
|
@@ -52,6 +53,7 @@ export async function parseSuperviseOptions(values, runtime) {
|
|
|
52
53
|
? supervisorAllowedToolsRaw.split(",")
|
|
53
54
|
: undefined,
|
|
54
55
|
mcpServer: values["mcp-server"] || undefined,
|
|
56
|
+
...parseAdvisorOptions(values),
|
|
55
57
|
};
|
|
56
58
|
}
|
|
57
59
|
|
|
@@ -125,6 +127,8 @@ export async function runSuperviseCommand(ctx) {
|
|
|
125
127
|
agentMcpServers,
|
|
126
128
|
redactor,
|
|
127
129
|
runtime,
|
|
130
|
+
advisorModel: opts.advisorModel,
|
|
131
|
+
advisorMaxUses: opts.advisorMaxUses,
|
|
128
132
|
});
|
|
129
133
|
|
|
130
134
|
const result = await supervisor.run(opts.taskContent);
|
package/src/discuss-tools.js
CHANGED
|
@@ -107,7 +107,7 @@ const ACKNOWLEDGE_DESC =
|
|
|
107
107
|
"Acknowledge an Ask before starting work. Posts a visible comment on the thread. Does not discharge the Ask — you still owe an Answer.";
|
|
108
108
|
|
|
109
109
|
/** Discuss-mode agent tool server. */
|
|
110
|
-
export function createDiscussAgentToolServer(ctx, { from }) {
|
|
110
|
+
export function createDiscussAgentToolServer(ctx, { from, extraTools = [] }) {
|
|
111
111
|
return orchestrationServer([
|
|
112
112
|
...baseTools(ctx, { from, defaultTo: "lead", broadcast: true }),
|
|
113
113
|
requestForCommentTool(ctx),
|
|
@@ -133,6 +133,7 @@ export function createDiscussAgentToolServer(ctx, { from }) {
|
|
|
133
133
|
return { content: [{ type: "text", text: "Acknowledged." }] };
|
|
134
134
|
},
|
|
135
135
|
),
|
|
136
|
+
...extraTools,
|
|
136
137
|
]);
|
|
137
138
|
}
|
|
138
139
|
|