@forwardimpact/libharness 2.0.0 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +68 -65
- package/package.json +15 -13
- package/src/advisor.js +47 -41
- package/src/agent-runner.js +58 -48
- package/src/benchmark/apm-installer.js +28 -28
- package/src/benchmark/env-loader.js +24 -16
- package/src/benchmark/grade.js +44 -41
- package/src/benchmark/hidden-tests.js +25 -24
- package/src/benchmark/hook-env.js +11 -9
- package/src/benchmark/invariants.js +20 -17
- package/src/benchmark/judge.js +29 -28
- package/src/benchmark/npm-installer.js +9 -8
- package/src/benchmark/report.js +53 -50
- package/src/benchmark/result.js +24 -23
- package/src/benchmark/runner.js +75 -69
- package/src/benchmark/scheduler.js +17 -16
- package/src/benchmark/task-family.js +29 -27
- package/src/benchmark/trace-split.js +9 -8
- package/src/benchmark/workdir.js +27 -25
- package/src/claude-code-executable.js +11 -11
- package/src/commands/advisor-flags.js +8 -7
- package/src/commands/assert.js +16 -15
- package/src/commands/benchmark-definition.js +20 -20
- package/src/commands/benchmark-grade.js +13 -12
- package/src/commands/benchmark-report.js +5 -5
- package/src/commands/benchmark-run.js +31 -28
- package/src/commands/by-discussion.js +11 -11
- package/src/commands/callback.js +11 -11
- package/src/commands/discuss.js +8 -7
- package/src/commands/facilitate.js +16 -14
- package/src/commands/output.js +4 -3
- package/src/commands/run.js +15 -15
- package/src/commands/scan-logs.js +22 -20
- package/src/commands/selfedit.js +124 -0
- package/src/commands/supervise.js +13 -11
- package/src/commands/task-input.js +9 -9
- package/src/commands/tee.js +11 -10
- package/src/commands/trace.js +55 -42
- package/src/commands/work-tracker.js +4 -3
- package/src/cost.js +17 -17
- package/src/discuss-tools.js +16 -16
- package/src/discusser.js +39 -38
- package/src/events/github.js +54 -37
- package/src/facilitator.js +21 -21
- package/src/inbox-poller.js +4 -4
- package/src/judge.js +32 -30
- package/src/message-bus.js +12 -11
- package/src/orchestration-loop.js +35 -36
- package/src/orchestration-toolkit.js +58 -53
- package/src/orchestrator-helpers.js +2 -2
- package/src/profile-prompt.js +54 -53
- package/src/redaction.js +63 -57
- package/src/render/line-renderer.js +5 -5
- package/src/render/orchestrator-filter.js +3 -3
- package/src/render/palette.js +11 -9
- package/src/render/tool-hints.js +18 -15
- package/src/render/turn-renderer.js +4 -4
- package/src/reply-emitter.js +2 -2
- package/src/sequence-counter.js +4 -3
- package/src/signature-filter.js +7 -6
- package/src/supervisor.js +19 -18
- package/src/tee-writer.js +25 -25
- package/src/trace-collector.js +53 -48
- package/src/trace-github.js +53 -44
- package/src/trace-multi.js +16 -14
- package/src/trace-query.js +61 -52
- package/src/trace-render.js +19 -19
- package/src/trace-usage.js +31 -28
- package/src/transcript-recorder.js +24 -20
- package/bin/fit-benchmark.js +0 -44
- package/bin/fit-harness.js +0 -412
- package/bin/fit-selfedit.js +0 -165
- package/bin/fit-trace.js +0 -520
package/src/redaction.js
CHANGED
|
@@ -1,19 +1,20 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Redactor — replaces secrets in JSON-serialisable values before they reach
|
|
3
|
-
* the trace artifact.
|
|
4
|
-
* set of credential-shape regexes. Both run on every primitive string.
|
|
3
|
+
* the trace artifact. It composes two layers: an env-var value allowlist and
|
|
4
|
+
* a set of credential-shape regexes. Both run on every primitive string.
|
|
5
5
|
*
|
|
6
|
-
* Coverage includes encoded credential forms
|
|
7
|
-
* layer matches each allowlisted secret
|
|
8
|
-
* base64** form at any byte offset within the encoded
|
|
9
|
-
* pattern layer covers the git `extraheader` basic-auth
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
6
|
+
* Coverage includes encoded credential forms as well as raw bytes. The env
|
|
7
|
+
* layer matches each allowlisted secret raw. It also matches the secret in
|
|
8
|
+
* its **standard base64** form at any byte offset within the encoded
|
|
9
|
+
* plaintext. The pattern layer covers the git `extraheader` basic-auth
|
|
10
|
+
* wrapper. Two limits apply. The redactor covers **standard base64 only**,
|
|
11
|
+
* so it does not cover URL-safe base64, hex, or percent-encoding. It also
|
|
12
|
+
* covers the **trace-write sink only**. Content an agent authors into a wiki
|
|
13
|
+
* commit never passes through this redactor.
|
|
13
14
|
*
|
|
14
|
-
*
|
|
15
|
-
* `process.env` writes (e.g. agent-runner.js LIBHARNESS_SKILL,
|
|
16
|
-
* LIBHARNESS_AGENT_PROFILE) cannot smuggle a value past
|
|
15
|
+
* The redactor is stateless after construction. It captures `env` once, so
|
|
16
|
+
* in-process `process.env` writes (e.g. agent-runner.js LIBHARNESS_SKILL,
|
|
17
|
+
* commands/run.js LIBHARNESS_AGENT_PROFILE) cannot smuggle a value past it.
|
|
17
18
|
*/
|
|
18
19
|
|
|
19
20
|
export const DEFAULT_ENV_ALLOWLIST = Object.freeze([
|
|
@@ -36,7 +37,7 @@ export const DEFAULT_ENV_ALLOWLIST = Object.freeze([
|
|
|
36
37
|
|
|
37
38
|
// Anchored prefixes per
|
|
38
39
|
// https://github.blog/security/application-security/behind-githubs-new-authentication-token-formats/
|
|
39
|
-
// Anthropic prefix is heuristic
|
|
40
|
+
// The Anthropic prefix is heuristic. The env-allowlist layer is the primary
|
|
40
41
|
// defence for Anthropic keys.
|
|
41
42
|
export const DEFAULT_PATTERNS = Object.freeze([
|
|
42
43
|
{ kind: "anthropic", regex: /sk-ant-[A-Za-z0-9_-]{80,}/g },
|
|
@@ -45,11 +46,11 @@ export const DEFAULT_PATTERNS = Object.freeze([
|
|
|
45
46
|
{ kind: "gh-oauth", regex: /\bgho_[A-Za-z0-9]{36}\b/g },
|
|
46
47
|
{ kind: "gh-fine-grained", regex: /\bgithub_pat_[A-Za-z0-9_]{82}\b/g },
|
|
47
48
|
// git persists HTTP basic-auth credentials base64-encoded in
|
|
48
|
-
// `http.<url>.extraheader` as `AUTHORIZATION: basic <b64
|
|
49
|
-
// plaintext is `x-access-token:<token>` (actions/checkout form)
|
|
50
|
-
//
|
|
51
|
-
//
|
|
52
|
-
// 20 chars
|
|
49
|
+
// `http.<url>.extraheader` as `AUTHORIZATION: basic <b64>`. There the
|
|
50
|
+
// plaintext is `x-access-token:<token>` (actions/checkout form). The
|
|
51
|
+
// raw-byte layers above cannot see that shape. The plaintext prefix is 15
|
|
52
|
+
// bytes, which is five whole base64 triplets. So every encoded form starts
|
|
53
|
+
// with the same 20 chars, whatever token follows.
|
|
53
54
|
{
|
|
54
55
|
kind: "gh-b64-basic-credential",
|
|
55
56
|
regex: /\beC1hY2Nlc3MtdG9rZW46[A-Za-z0-9+/]{8,}={0,2}/g,
|
|
@@ -60,27 +61,29 @@ const ENV_PLACEHOLDER = (name) => `[REDACTED:env:${name}]`;
|
|
|
60
61
|
const PATTERN_PLACEHOLDER = (kind) => `[REDACTED:pattern:${kind}]`;
|
|
61
62
|
|
|
62
63
|
/**
|
|
63
|
-
*
|
|
64
|
-
* shortest offset core is exactly 8 chars
|
|
65
|
-
*
|
|
66
|
-
* safety, false positives).
|
|
67
|
-
* password) far exceeds it.
|
|
64
|
+
* The minimum byte length a secret needs before the redactor matches its
|
|
65
|
+
* encoded form. At 9 bytes the shortest offset core is exactly 8 chars.
|
|
66
|
+
* Below 9 bytes it drops under 8 chars. That is too short for a sound needle
|
|
67
|
+
* against ordinary base64 trace content (margin of safety, false positives).
|
|
68
|
+
* Every DEFAULT_ENV_ALLOWLIST value (token, key, password) far exceeds it.
|
|
68
69
|
*/
|
|
69
70
|
const MIN_ENCODED_SECRET_BYTES = 9;
|
|
70
71
|
|
|
71
|
-
//
|
|
72
|
+
// The k filler bytes contaminate this many base64 chars at the start, per
|
|
73
|
+
// alignment.
|
|
72
74
|
const ENCODED_LEAD_STRIP = [0, 2, 3];
|
|
73
75
|
|
|
74
76
|
/**
|
|
75
|
-
*
|
|
76
|
-
*
|
|
77
|
-
* independently
|
|
78
|
-
* on the secret's bytes
|
|
79
|
-
* groups at each edge
|
|
80
|
-
*
|
|
81
|
-
*
|
|
82
|
-
*
|
|
83
|
-
*
|
|
77
|
+
* Return the three standard-base64 core substrings of `secret`, one per byte
|
|
78
|
+
* alignment (k = 0/1/2). Each core is offset-invariant. base64 maps disjoint
|
|
79
|
+
* 3-byte groups to 4 chars independently. So the chars that cover a secret's
|
|
80
|
+
* interior groups depend only on the secret's bytes. They never depend on the
|
|
81
|
+
* bytes around it. Only the partial groups at each edge depend on the
|
|
82
|
+
* neighbours. This function strips those groups. The core that remains
|
|
83
|
+
* appears in the base64 of any plaintext that puts `secret` at that
|
|
84
|
+
* alignment. Padding lives only in the final partial group, and this function
|
|
85
|
+
* strips that group. So each core is padding-free. One needle matches padded
|
|
86
|
+
* and unpadded haystack content. Returns [] below MIN_ENCODED_SECRET_BYTES.
|
|
84
87
|
* @param {string} secret
|
|
85
88
|
* @returns {string[]}
|
|
86
89
|
*/
|
|
@@ -98,9 +101,10 @@ function encodedNeedles(secret) {
|
|
|
98
101
|
|
|
99
102
|
/**
|
|
100
103
|
* Build a frozen { name → { secret, needles } } snapshot of the requested env
|
|
101
|
-
* vars.
|
|
102
|
-
*
|
|
103
|
-
* precomputed standard-base64 cores (empty for sub-floor
|
|
104
|
+
* vars. This function skips empty strings. A leaked empty env var would
|
|
105
|
+
* otherwise make the redactor replace every empty string in the trace.
|
|
106
|
+
* `needles` are the precomputed standard-base64 cores (empty for sub-floor
|
|
107
|
+
* secrets).
|
|
104
108
|
*/
|
|
105
109
|
function snapshotEnv(env, allowlist) {
|
|
106
110
|
const snap = {};
|
|
@@ -129,8 +133,8 @@ function walk(value, redactString) {
|
|
|
129
133
|
export class Redactor {
|
|
130
134
|
/**
|
|
131
135
|
* @param {object} deps
|
|
132
|
-
* @param {Readonly<Record<string, {secret: string, needles: string[]}>>} deps.envSnapshot - Frozen { name → { secret, needles } } map captured at construction time
|
|
133
|
-
* @param {ReadonlyArray<{kind: string, regex: RegExp}>} deps.patterns - Credential-shape regexes
|
|
136
|
+
* @param {Readonly<Record<string, {secret: string, needles: string[]}>>} deps.envSnapshot - Frozen { name → { secret, needles } } map captured at construction time. `needles` are the precomputed standard-base64 cores of `secret`.
|
|
137
|
+
* @param {ReadonlyArray<{kind: string, regex: RegExp}>} deps.patterns - Credential-shape regexes. Each match becomes `[REDACTED:pattern:KIND]`.
|
|
134
138
|
* @param {boolean} deps.enabled - When false, `redactValue` returns its input by reference.
|
|
135
139
|
*/
|
|
136
140
|
constructor({ envSnapshot, patterns, enabled }) {
|
|
@@ -140,8 +144,9 @@ export class Redactor {
|
|
|
140
144
|
}
|
|
141
145
|
|
|
142
146
|
/**
|
|
143
|
-
* Redact any JSON-serialisable value
|
|
144
|
-
* in every primitive string.
|
|
147
|
+
* Redact any JSON-serialisable value. This method deep-walks the value and
|
|
148
|
+
* replaces secrets in every primitive string. When disabled, it returns its
|
|
149
|
+
* input by reference.
|
|
145
150
|
* @param {unknown} value
|
|
146
151
|
* @returns {unknown}
|
|
147
152
|
*/
|
|
@@ -163,10 +168,11 @@ export class Redactor {
|
|
|
163
168
|
if (out.includes(secret)) {
|
|
164
169
|
out = out.split(secret).join(ENV_PLACEHOLDER(name));
|
|
165
170
|
}
|
|
166
|
-
// Standard-base64 form at any byte offset.
|
|
167
|
-
//
|
|
168
|
-
//
|
|
169
|
-
// needle cannot re-match them. The
|
|
171
|
+
// Standard-base64 form at any byte offset. The order among the three
|
|
172
|
+
// needles does not matter. The placeholder shares no base64 run with
|
|
173
|
+
// any needle. Once a replacement puts the placeholder over a region,
|
|
174
|
+
// those bytes are gone, so a later needle cannot re-match them. The
|
|
175
|
+
// floor keeps every needle ≥ 8 chars.
|
|
170
176
|
for (const needle of needles) {
|
|
171
177
|
if (out.includes(needle)) {
|
|
172
178
|
out = out.split(needle).join(ENV_PLACEHOLDER(name));
|
|
@@ -181,20 +187,20 @@ export class Redactor {
|
|
|
181
187
|
}
|
|
182
188
|
|
|
183
189
|
/**
|
|
184
|
-
* Build a redactor.
|
|
185
|
-
* `LIBHARNESS_REDACTION_ENV_VARS` from the supplied env.
|
|
186
|
-
*
|
|
187
|
-
* `runtime.proc.stderr`)
|
|
188
|
-
*
|
|
189
|
-
* override still wins for the snapshot.
|
|
190
|
-
*
|
|
191
|
-
* fixtures.
|
|
190
|
+
* Build a redactor. It reads `LIBHARNESS_REDACTION_DISABLED` and
|
|
191
|
+
* `LIBHARNESS_REDACTION_ENV_VARS` from the supplied env. An injected
|
|
192
|
+
* `runtime` supplies the env and the stderr sink (`runtime.proc.env` /
|
|
193
|
+
* `runtime.proc.stderr`). When a caller supplies no runtime, the function
|
|
194
|
+
* constructs a default one so current callers keep working. An explicit
|
|
195
|
+
* `opts.env` override still wins for the snapshot. The function fires a
|
|
196
|
+
* one-shot stderr warning when a caller constructs it disabled. Use
|
|
197
|
+
* `createNoopRedactor()` for silent fixtures to bypass that warning.
|
|
192
198
|
* @param {object} [opts]
|
|
193
|
-
* @param {import("@forwardimpact/libutil/runtime").Runtime} [opts.runtime] - Ambient collaborators
|
|
199
|
+
* @param {import("@forwardimpact/libutil/runtime").Runtime} [opts.runtime] - Ambient collaborators. The factory uses `proc.env` and `proc.stderr`.
|
|
194
200
|
* @param {Record<string, string|undefined>} [opts.env] - Environment to snapshot. Defaults to `runtime.proc.env`.
|
|
195
201
|
* @param {string[]} [opts.allowlist] - Override the env-var name list. Defaults to `DEFAULT_ENV_ALLOWLIST` or the parsed `LIBHARNESS_REDACTION_ENV_VARS` value.
|
|
196
202
|
* @param {ReadonlyArray<{kind: string, regex: RegExp}>} [opts.patterns] - Credential-shape regexes. Defaults to `DEFAULT_PATTERNS`.
|
|
197
|
-
* @param {boolean} [opts.enabled] - Force enabled
|
|
203
|
+
* @param {boolean} [opts.enabled] - Force enabled or disabled. It bypasses `LIBHARNESS_REDACTION_DISABLED`.
|
|
198
204
|
* @returns {Redactor}
|
|
199
205
|
*/
|
|
200
206
|
export function createRedactor({
|
|
@@ -215,7 +221,7 @@ export function createRedactor({
|
|
|
215
221
|
: Object.freeze({});
|
|
216
222
|
if (!resolvedEnabled) {
|
|
217
223
|
proc.stderr.write(
|
|
218
|
-
"libharness: trace redaction DISABLED
|
|
224
|
+
"libharness: trace redaction DISABLED through LIBHARNESS_REDACTION_DISABLED. Secrets may appear in the trace artifact\n",
|
|
219
225
|
);
|
|
220
226
|
}
|
|
221
227
|
return new Redactor({ envSnapshot, patterns, enabled: resolvedEnabled });
|
|
@@ -240,8 +246,8 @@ function resolveAllowlistFromEnv(env) {
|
|
|
240
246
|
|
|
241
247
|
/**
|
|
242
248
|
* Build a disabled redactor whose `redactValue` is the identity function.
|
|
243
|
-
*
|
|
244
|
-
* fires
|
|
249
|
+
* Use this form in test fixtures. It bypasses `createRedactor`, so no stderr
|
|
250
|
+
* warning fires whatever the env state.
|
|
245
251
|
* @returns {Redactor}
|
|
246
252
|
*/
|
|
247
253
|
export function createNoopRedactor() {
|
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Line renderer — composes prefix + color + body + reset into a single
|
|
3
|
-
* terminal line. Pure
|
|
3
|
+
* terminal line. Pure, with no side effects.
|
|
4
4
|
*
|
|
5
5
|
* Every renderer returns a `\n`-terminated string:
|
|
6
6
|
* <source>: <ESC><color><body><RESET>\n
|
|
7
7
|
*
|
|
8
|
-
* The `<source>: ` prefix lives outside the color escape so grep and
|
|
9
|
-
* color
|
|
10
|
-
* the source label and the kind label (`Bash:`, `Result:`, `Error:`)
|
|
11
|
-
*
|
|
8
|
+
* The `<source>: ` prefix lives outside the color escape, so grep and
|
|
9
|
+
* terminals that strip color preserve the participant tag. Colons separate
|
|
10
|
+
* the source label and the kind label (`Bash:`, `Result:`, `Error:`). The
|
|
11
|
+
* line then stays tight on narrow viewports and keeps its structure.
|
|
12
12
|
*/
|
|
13
13
|
|
|
14
14
|
import { colorForSource, ERROR_COLOR, RESET } from "./palette.js";
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Orchestrator filter — predicate
|
|
3
|
-
*
|
|
2
|
+
* Orchestrator filter — predicate that names the orchestrator lifecycle
|
|
3
|
+
* events to suppress from the human-readable log.
|
|
4
4
|
*
|
|
5
|
-
* NDJSON artifacts still carry every orchestrator event
|
|
5
|
+
* NDJSON artifacts still carry every orchestrator event. This module only
|
|
6
6
|
* controls what the live `textStream` and offline `toText()` show.
|
|
7
7
|
*/
|
|
8
8
|
|
package/src/render/palette.js
CHANGED
|
@@ -1,15 +1,16 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Palette — pure profile-name → ANSI SGR foreground color function.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
* the same name maps to the same color in every
|
|
6
|
-
* tool-result errors
|
|
4
|
+
* The module assigns a color from a FNV-1a hash of the source name modulo
|
|
5
|
+
* the palette size. So the same name maps to the same color in every
|
|
6
|
+
* process. The module reserves red for tool-result errors. The palette never
|
|
7
|
+
* contains red.
|
|
7
8
|
*
|
|
8
9
|
* Colors use the 24-bit truecolor SGR escape (`ESC[38;2;R;G;Bm`) rather than
|
|
9
10
|
* the 16-color table. GitHub Actions' log viewer and most modern terminals
|
|
10
|
-
* render truecolor as the exact hex requested
|
|
11
|
+
* render truecolor as the exact hex requested. This avoids the washed-out
|
|
11
12
|
* mustard/olive tones GHA applies to `ESC[93m` etc. Eight slots cover the
|
|
12
|
-
* largest concurrent cast in any
|
|
13
|
+
* largest concurrent cast in any current workflow (five domain agents plus
|
|
13
14
|
* the facilitator) with headroom.
|
|
14
15
|
*/
|
|
15
16
|
|
|
@@ -33,9 +34,10 @@ export const RESET = "\u001b[0m";
|
|
|
33
34
|
/**
|
|
34
35
|
* Map a source name to a stable ANSI foreground color.
|
|
35
36
|
*
|
|
36
|
-
*
|
|
37
|
-
* name
|
|
38
|
-
*
|
|
37
|
+
* This function is pure. A FNV-1a 32-bit hash of the name decides the color.
|
|
38
|
+
* The same name gives the same color on every call, in every process.
|
|
39
|
+
* Returns `RESET` for absent or empty names so callers never emit a stray
|
|
40
|
+
* escape.
|
|
39
41
|
*
|
|
40
42
|
* @param {string|null|undefined} name
|
|
41
43
|
* @returns {string} ANSI SGR escape, never equal to `ERROR_COLOR`
|
|
@@ -47,7 +49,7 @@ export function colorForSource(name) {
|
|
|
47
49
|
h ^= name.charCodeAt(i);
|
|
48
50
|
h = Math.imul(h, 0x01000193) >>> 0;
|
|
49
51
|
}
|
|
50
|
-
// Length mixer
|
|
52
|
+
// Length mixer. It reduces FNV's intrinsic birthday collisions on short
|
|
51
53
|
// names with shared affixes (e.g. `staff-engineer`/`facilitator`).
|
|
52
54
|
h ^= name.length;
|
|
53
55
|
h = Math.imul(h, 0x01000193) >>> 0;
|
package/src/render/tool-hints.js
CHANGED
|
@@ -3,24 +3,25 @@
|
|
|
3
3
|
* tool-result previews.
|
|
4
4
|
*
|
|
5
5
|
* `hintForCall(name, input)` renders the human-meaningful field for each
|
|
6
|
-
* tool (file path, command, pattern, …)
|
|
7
|
-
* (`{`, `}`, `"`)
|
|
6
|
+
* tool (file path, command, pattern, …). It strips JSON punctuation
|
|
7
|
+
* (`{`, `}`, `"`) from that field. It collapses the field to a single line
|
|
8
|
+
* ≤ 80 chars.
|
|
8
9
|
*
|
|
9
|
-
* MCP-prefixed tools (`mcp__*`) are an intentional carve-out
|
|
10
|
+
* MCP-prefixed tools (`mcp__*`) are an intentional carve-out. Their hint is
|
|
10
11
|
* the full input rendered as compact single-line JSON, so `{` and `"` do
|
|
11
12
|
* appear on those lines. Readers of GitHub workflow logs need the full MCP
|
|
12
|
-
* payload to know what
|
|
13
|
+
* payload to know what the caller actually sent across the protocol.
|
|
13
14
|
*
|
|
14
15
|
* `previewForResult(content, isError)` collapses a tool result to a single
|
|
15
|
-
* line ≤ 80 chars
|
|
16
|
-
* error color and the `Error:` label.
|
|
16
|
+
* line ≤ 80 chars. It also flags errors. The flag lets the renderer apply
|
|
17
|
+
* the reserved error color and the `Error:` label.
|
|
17
18
|
*/
|
|
18
19
|
|
|
19
20
|
const MAX_HINT_CHARS = 80;
|
|
20
21
|
|
|
21
22
|
/**
|
|
22
23
|
* Strip `{`, `}`, `"`, collapse whitespace, and truncate to MAX_HINT_CHARS.
|
|
23
|
-
*
|
|
24
|
+
* Use the first line only. Drop anything past a newline. Always returns a
|
|
24
25
|
* string, never null/undefined.
|
|
25
26
|
* @param {unknown} raw
|
|
26
27
|
* @returns {string}
|
|
@@ -36,9 +37,10 @@ function sanitize(raw) {
|
|
|
36
37
|
}
|
|
37
38
|
|
|
38
39
|
/**
|
|
39
|
-
* Truncate an already-sanitized string to MAX_HINT_CHARS with
|
|
40
|
-
*
|
|
41
|
-
*
|
|
40
|
+
* Truncate an already-sanitized string to MAX_HINT_CHARS, with an ellipsis
|
|
41
|
+
* at the end when it overflows. The few handlers that concatenate multiple
|
|
42
|
+
* sanitized pieces share this helper. They concatenate first, then decide
|
|
43
|
+
* whether to truncate.
|
|
42
44
|
* @param {string} str
|
|
43
45
|
* @returns {string}
|
|
44
46
|
*/
|
|
@@ -50,8 +52,9 @@ function truncate(str) {
|
|
|
50
52
|
|
|
51
53
|
/**
|
|
52
54
|
* Per-tool hint handlers. Each entry takes the sanitized input object
|
|
53
|
-
* (never null) and returns the hint string.
|
|
54
|
-
*
|
|
55
|
+
* (never null) and returns the hint string. This table stays flat. A new
|
|
56
|
+
* tool then needs one entry. It does not need a new branch in a switch that
|
|
57
|
+
* grows.
|
|
55
58
|
*/
|
|
56
59
|
const HINT_HANDLERS = {
|
|
57
60
|
Bash: (i) => sanitize(i.command),
|
|
@@ -104,7 +107,7 @@ export function simplifyToolName(name) {
|
|
|
104
107
|
* `{` / `"` from the input (built-in tool hints stay free of JSON
|
|
105
108
|
* punctuation so readers see clean one-liners).
|
|
106
109
|
* - An MCP-prefixed tool (`mcp__*`) → full input rendered as compact
|
|
107
|
-
* single-line JSON
|
|
110
|
+
* single-line JSON. `{` and `"` intentionally appear so readers see
|
|
108
111
|
* the actual MCP payload.
|
|
109
112
|
* - Anything else → "" (the caller still shows the bare tool name).
|
|
110
113
|
*
|
|
@@ -126,8 +129,8 @@ export function hintForCall(name, input) {
|
|
|
126
129
|
|
|
127
130
|
/**
|
|
128
131
|
* Render a tool result as a single preview line plus an `isError` flag.
|
|
129
|
-
* The flag lets the line-renderer pick the reserved error color
|
|
130
|
-
* re-
|
|
132
|
+
* The flag lets the line-renderer pick the reserved error color. The
|
|
133
|
+
* line-renderer does not re-inspect the content.
|
|
131
134
|
*
|
|
132
135
|
* @param {string|object|null|undefined} content - Tool result content
|
|
133
136
|
* @param {boolean} isError - Whether the tool call failed
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Turn renderer — maps a structured turn into formatted text lines.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
4
|
+
* `TeeWriter.flushTurns()` (live stream) and `TraceCollector.toText()`
|
|
5
|
+
* (offline replay) share it, so both emit identical output.
|
|
6
6
|
*/
|
|
7
7
|
|
|
8
8
|
import {
|
|
@@ -51,8 +51,8 @@ function renderAssistantTurn(turn, withPrefix) {
|
|
|
51
51
|
|
|
52
52
|
/** @param {object} turn @param {boolean} withPrefix @returns {string[]} */
|
|
53
53
|
function renderToolResultTurn(turn, withPrefix) {
|
|
54
|
-
// Successful tool results emit no preview line
|
|
55
|
-
// the structured turn
|
|
54
|
+
// Successful tool results emit no preview line. The trace document keeps
|
|
55
|
+
// the structured turn. Readers of the streamed log see errors only.
|
|
56
56
|
if (!turn.isError) return [];
|
|
57
57
|
return [
|
|
58
58
|
renderToolResultLine({
|
package/src/reply-emitter.js
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* ReplyEmitter — POST reply/ack events to the callback URL as they
|
|
3
|
-
* happen. Each emission is fire-and-forget so
|
|
4
|
-
*
|
|
3
|
+
* happen. Each emission is fire-and-forget, so network I/O never blocks
|
|
4
|
+
* the message bus.
|
|
5
5
|
*/
|
|
6
6
|
export class ReplyEmitter {
|
|
7
7
|
#callbackUrl;
|
package/src/sequence-counter.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* SequenceCounter — global monotonic counter
|
|
3
|
-
*
|
|
2
|
+
* SequenceCounter — global monotonic counter. All participants in a session
|
|
3
|
+
* share one counter. Single-threaded JS means the counter needs no
|
|
4
|
+
* synchronization.
|
|
4
5
|
*/
|
|
5
6
|
/** Monotonic counter that assigns globally ordered sequence numbers within a session. */
|
|
6
7
|
export class SequenceCounter {
|
|
@@ -15,7 +16,7 @@ export class SequenceCounter {
|
|
|
15
16
|
}
|
|
16
17
|
}
|
|
17
18
|
|
|
18
|
-
/** Create a new SequenceCounter
|
|
19
|
+
/** Create a new SequenceCounter that starts at zero. */
|
|
19
20
|
export function createSequenceCounter() {
|
|
20
21
|
return new SequenceCounter();
|
|
21
22
|
}
|
package/src/signature-filter.js
CHANGED
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* Strip `thinking.signature` base64 blobs from a JSON-serializable value.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
* signatures intact (lossless storage)
|
|
6
|
-
* by default because they dominate output
|
|
4
|
+
* The CLI applies this filter at the output boundary. The stored structured
|
|
5
|
+
* trace keeps signatures intact (lossless storage). The display filter drops
|
|
6
|
+
* them by default, because they dominate the output and do not help
|
|
7
|
+
* analysis.
|
|
7
8
|
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
* any other type
|
|
9
|
+
* This function walks the input recursively. For any object whose
|
|
10
|
+
* `type === "thinking"`, it copies the object and then removes the
|
|
11
|
+
* `signature` field. It keeps signatures on objects of any other type.
|
|
11
12
|
*
|
|
12
13
|
* @param {*} value - Any JSON-serializable value
|
|
13
14
|
* @returns {*} A deep-copy with thinking signatures removed
|
package/src/supervisor.js
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Supervisor — supervise-mode wrapper around `OrchestrationLoop`.
|
|
3
|
-
*
|
|
4
|
-
* (`"
|
|
5
|
-
*
|
|
6
|
-
*
|
|
2
|
+
* Supervisor — supervise-mode wrapper around `OrchestrationLoop`. A lead
|
|
3
|
+
* participant (`"supervisor"`) coordinates one named participant
|
|
4
|
+
* (`"agent"`). The structure is the same as `Facilitator` with a single
|
|
5
|
+
* agent. Only the role names, the prompts, and the pass-through accessors
|
|
6
|
+
* differ.
|
|
7
7
|
*
|
|
8
|
-
* Ask is async
|
|
9
|
-
* `{askIds:[N]}` immediately
|
|
8
|
+
* Ask is async, with the same contract as facilitate and discuss. It
|
|
9
|
+
* returns `{askIds:[N]}` immediately. The agent's reply arrives on the
|
|
10
10
|
* supervisor's next turn as `[answer#N] agent: <text>`. The supervisor
|
|
11
|
-
* sees the agent at each Ask boundary
|
|
11
|
+
* sees the agent at each Ask boundary. It plans the next step. It
|
|
12
12
|
* eventually calls Conclude.
|
|
13
13
|
*
|
|
14
14
|
* For tighter feedback loops, size the agent's per-turn budget down
|
|
@@ -34,7 +34,7 @@ import {
|
|
|
34
34
|
} from "./advisor.js";
|
|
35
35
|
import { createTranscriptRecorder } from "./transcript-recorder.js";
|
|
36
36
|
|
|
37
|
-
/** System prompt for the supervisor lead. L0 mechanics only per
|
|
37
|
+
/** System prompt for the supervisor lead. L0 mechanics only per JIDOKA. */
|
|
38
38
|
export const SUPERVISOR_SYSTEM_PROMPT =
|
|
39
39
|
"You supervise one agent.\n" +
|
|
40
40
|
"Use `Ask` to delegate the agent's task to the agent.\n" +
|
|
@@ -42,9 +42,9 @@ export const SUPERVISOR_SYSTEM_PROMPT =
|
|
|
42
42
|
"The reply arrives on your next turn as `[answer#N] agent: <text>` in your inbox.\n" +
|
|
43
43
|
"End your turn while Asks are pending. The system resumes you when an answer arrives.\n" +
|
|
44
44
|
"If the agent goes off-track, send a corrective `Ask`.\n" +
|
|
45
|
-
"
|
|
45
|
+
"Call `Conclude` with a verdict and summary to end every session.";
|
|
46
46
|
|
|
47
|
-
/** System prompt for the supervised agent. L0 mechanics only per
|
|
47
|
+
/** System prompt for the supervised agent. L0 mechanics only per JIDOKA. */
|
|
48
48
|
export const AGENT_SYSTEM_PROMPT =
|
|
49
49
|
"A supervisor directs your work.\n" +
|
|
50
50
|
"Each question arrives as `[ask#N] supervisor: <text>` in your inbox.\n" +
|
|
@@ -54,7 +54,8 @@ export const AGENT_SYSTEM_PROMPT =
|
|
|
54
54
|
|
|
55
55
|
/**
|
|
56
56
|
* Supervise-mode wrapper around `OrchestrationLoop`. The lead is
|
|
57
|
-
* `"supervisor"
|
|
57
|
+
* `"supervisor"`. One participant is `"agent"`. The mode tag is
|
|
58
|
+
* `"supervised"`.
|
|
58
59
|
*/
|
|
59
60
|
export class Supervisor extends OrchestrationLoop {
|
|
60
61
|
/**
|
|
@@ -135,7 +136,7 @@ const devNull = new Writable({
|
|
|
135
136
|
* @param {string} [deps.profilesDir]
|
|
136
137
|
* @param {string} [deps.taskAmend]
|
|
137
138
|
* @param {Record<string, object>} [deps.agentMcpServers]
|
|
138
|
-
* @param {string} [deps.advisorModel] - Claude model for advisor consults
|
|
139
|
+
* @param {string} [deps.advisorModel] - Claude model for advisor consults. When absent, the factory offers no Advisor tool.
|
|
139
140
|
* @param {number} [deps.advisorMaxUses] - Session-wide consult budget (default 3).
|
|
140
141
|
* @returns {Supervisor}
|
|
141
142
|
*/
|
|
@@ -181,9 +182,9 @@ export function createSupervisor({
|
|
|
181
182
|
const perRunBudget = maxTurns ?? 200;
|
|
182
183
|
const abortController = new AbortController();
|
|
183
184
|
|
|
184
|
-
//
|
|
185
|
-
//
|
|
186
|
-
// to today's.
|
|
185
|
+
// Everything below wires the advisor. It runs only when advisorModel is
|
|
186
|
+
// set. When advisorModel is unset, the composed prompt and the tool
|
|
187
|
+
// surface stay byte-identical to today's.
|
|
187
188
|
const budget = advisorModel ? createAdvisorBudget(advisorMaxUses ?? 3) : null;
|
|
188
189
|
const agentSystemPrompt = composeSystemPrompt({
|
|
189
190
|
role: "agent",
|
|
@@ -201,8 +202,8 @@ export function createSupervisor({
|
|
|
201
202
|
systemPrompt: agentSystemPrompt,
|
|
202
203
|
redactor,
|
|
203
204
|
});
|
|
204
|
-
//
|
|
205
|
-
//
|
|
205
|
+
// The `let supervisor` closure binds this late. The instance does not
|
|
206
|
+
// exist yet when the factory builds the advisor and the tool.
|
|
206
207
|
const advisor = createAdvisor({
|
|
207
208
|
model: advisorModel,
|
|
208
209
|
cwd: agentCwd,
|