pi-daddy 0.15.0 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. package/CHANGELOG.md +91 -0
  2. package/README.md +27 -12
  3. package/dist/cli.js +0 -0
  4. package/dist/executor.d.ts +38 -0
  5. package/dist/executor.d.ts.map +1 -0
  6. package/dist/executor.js +93 -0
  7. package/dist/executor.js.map +1 -0
  8. package/dist/herdr-cli.d.ts +78 -0
  9. package/dist/herdr-cli.d.ts.map +1 -0
  10. package/dist/herdr-cli.js +113 -0
  11. package/dist/herdr-cli.js.map +1 -0
  12. package/dist/herdr-name.d.ts +37 -0
  13. package/dist/herdr-name.d.ts.map +1 -0
  14. package/dist/herdr-name.js +59 -0
  15. package/dist/herdr-name.js.map +1 -0
  16. package/dist/herdr-poll.d.ts +104 -0
  17. package/dist/herdr-poll.d.ts.map +1 -0
  18. package/dist/herdr-poll.js +150 -0
  19. package/dist/herdr-poll.js.map +1 -0
  20. package/dist/herdr-stage.d.ts +40 -0
  21. package/dist/herdr-stage.d.ts.map +1 -0
  22. package/dist/herdr-stage.js +54 -0
  23. package/dist/herdr-stage.js.map +1 -0
  24. package/dist/ledger-report.d.ts +18 -0
  25. package/dist/ledger-report.d.ts.map +1 -1
  26. package/dist/ledger-report.js +10 -0
  27. package/dist/ledger-report.js.map +1 -1
  28. package/dist/ledger.d.ts +17 -0
  29. package/dist/ledger.d.ts.map +1 -1
  30. package/dist/ledger.js +1 -0
  31. package/dist/ledger.js.map +1 -1
  32. package/dist/pane-reaper.d.ts +66 -4
  33. package/dist/pane-reaper.d.ts.map +1 -1
  34. package/dist/pane-reaper.js +131 -9
  35. package/dist/pane-reaper.js.map +1 -1
  36. package/dist/progress.d.ts +96 -0
  37. package/dist/progress.d.ts.map +1 -0
  38. package/dist/progress.js +167 -0
  39. package/dist/progress.js.map +1 -0
  40. package/dist/run-child.d.ts +27 -0
  41. package/dist/run-child.d.ts.map +1 -1
  42. package/dist/run-child.js +84 -7
  43. package/dist/run-child.js.map +1 -1
  44. package/dist/run-herdr.d.ts +41 -28
  45. package/dist/run-herdr.d.ts.map +1 -1
  46. package/dist/run-herdr.js +150 -167
  47. package/dist/run-herdr.js.map +1 -1
  48. package/extensions/delegation.ts +94 -2
  49. package/extensions/grants-command.ts +26 -1
  50. package/extensions/grants.ts +85 -163
  51. package/extensions/run-delegation.ts +70 -8
  52. package/extensions/session-report.ts +231 -0
  53. package/extensions/session.ts +63 -11
  54. package/extensions/tripwire.ts +44 -0
  55. package/package.json +17 -1
  56. package/src/executor.ts +122 -0
  57. package/src/herdr-cli.ts +125 -0
  58. package/src/herdr-name.ts +61 -0
  59. package/src/herdr-poll.ts +185 -0
  60. package/src/herdr-stage.ts +55 -0
  61. package/src/ledger-report.ts +21 -0
  62. package/src/ledger.ts +18 -0
  63. package/src/pane-reaper.ts +147 -9
  64. package/src/progress.ts +206 -0
  65. package/src/run-child.ts +96 -7
  66. package/src/run-herdr.ts +170 -174
@@ -0,0 +1,125 @@
1
+ /**
2
+ * Talking to herdr: one command, one JSON envelope, plus the two questions ADR-0031 needs answered.
3
+ *
4
+ * Lifted out of `src/run-herdr.ts`, which was at 357 of the 400-line ceiling and gains output polling under
5
+ * ADR-0032. But the split is not only about lines: **the probe is not an executor concern**. It runs at session
6
+ * start, before any delegation exists, to decide *which* executor a session will use — so leaving it inside the
7
+ * herdr executor would mean the session imported the thing it was deciding whether to use.
8
+ *
9
+ * Every rule here is tested against an injected `exec`, so the suite stays fast, pi-free and herdr-free. The
10
+ * facts the fakes reproduce were measured against real herdr 0.7.5 (`docs/probes/g16-herdr`).
11
+ */
12
+
13
+ import { execFile } from "node:child_process";
14
+
15
+ /** One herdr CLI invocation. Injectable so every rule below is testable without herdr installed. */
16
+ export type HerdrExec = (args: string[]) => Promise<{ code: number | null; stdout: string; stderr: string }>;
17
+
18
+ export const defaultExec: HerdrExec = (args) =>
19
+ new Promise((settle) => {
20
+ execFile("herdr", args, { maxBuffer: 32 * 1024 * 1024 }, (error, stdout, stderr) => {
21
+ const raw = (error as { code?: unknown } | null)?.code;
22
+ const code = typeof raw === "number" ? raw : error ? 1 : 0;
23
+ // **A string `code` is a spawn failure, and it used to be thrown away.** `ENOENT` — herdr not installed —
24
+ // arrives as `code: "ENOENT"`, so the numeric test failed, the message was dropped, and an operator with
25
+ // `PI_GRANTS_HERDR=1` on a machine without herdr was told *"herdr is not answering (unparseable herdr
26
+ // reply: (no output))"* rather than that the binary is missing. Rule 8 wants the loud version, and this is
27
+ // the first diagnostic such an operator meets.
28
+ const spawnFailure = typeof raw === "string" ? `herdr could not be run (${raw}): ${error?.message ?? ""}` : "";
29
+ settle({ code, stdout: String(stdout), stderr: spawnFailure || String(stderr) });
30
+ });
31
+ });
32
+
33
+ /**
34
+ * Parse herdr's JSON envelope. Every command replies `{id, result}` or `{id, error:{code,message}}`.
35
+ *
36
+ * `stderr` is folded into the message because the first end-to-end run failed with an EMPTY stdout and the
37
+ * real reason on stderr, producing the useless diagnostic "unparseable herdr reply: ". A wrapper that
38
+ * hides the substrate's own error message costs more time than it saves.
39
+ */
40
+ export function parseReply(reply: { stdout: string; stderr: string }): { result?: Record<string, unknown>; error?: string } {
41
+ try {
42
+ const parsed = JSON.parse(reply.stdout) as { result?: Record<string, unknown>; error?: { message?: string; code?: string } };
43
+ if (parsed.error) return { error: parsed.error.message ?? parsed.error.code ?? "herdr reported an error" };
44
+ return { result: parsed.result };
45
+ } catch {
46
+ // A non-JSON reply is a herdr-version or PATH problem, not a governance decision. Surfaced as a spawn
47
+ // error so the caller reports "could not start" rather than "the child produced nothing".
48
+ const detail = [reply.stdout.trim(), reply.stderr.trim()].filter((t) => t.length > 0).join(" | ");
49
+ return { error: `unparseable herdr reply: ${detail.slice(0, 300) || "(no output)"}` };
50
+ }
51
+ }
52
+
53
+ /** Bound on the session-start probe. Short: it sits in front of the operator's first prompt. */
54
+ export const PROBE_TIMEOUT_MS = 2000;
55
+
56
+ export interface HerdrProbe {
57
+ ok: boolean;
58
+ /** herdr's own words when it is not reachable. Carried so the disclosure line can name the reason. */
59
+ error?: string;
60
+ }
61
+
62
+ /**
63
+ * Is there a herdr server that will answer right now? — ADR-0031's selection input.
64
+ *
65
+ * **`tab list`, not `which herdr`.** ADR-0031 rejects `PATH` detection as option C by name: a binary on `PATH`
66
+ * with no server behind it would make every delegation fail at `tab create`, on a path the operator never
67
+ * chose, and the diagnostic would arrive at the first delegation rather than at startup. Only a parsed
68
+ * `result` envelope counts as reachable; an `error` envelope, a non-JSON reply, a timeout and a throwing
69
+ * `exec` are all "not reachable" with the reason preserved.
70
+ *
71
+ * **Zero tabs is a successful answer**, deliberately: a fresh herdr with nothing open is reachable.
72
+ *
73
+ * Never throws. A probe that threw out of `session_start` would cancel every control after it, which is
74
+ * R-60's shape exactly — and this one runs *before* the line that discloses what it decided.
75
+ */
76
+ export async function probeHerdr(options: { exec?: HerdrExec; timeoutMs?: number } = {}): Promise<HerdrProbe> {
77
+ const exec = options.exec ?? defaultExec;
78
+ const timeoutMs = options.timeoutMs ?? PROBE_TIMEOUT_MS;
79
+
80
+ let timer: NodeJS.Timeout | undefined;
81
+ try {
82
+ return await Promise.race<HerdrProbe>([
83
+ exec(["tab", "list"]).then((reply) => {
84
+ const parsed = parseReply(reply);
85
+ return parsed.error ? { ok: false, error: parsed.error } : { ok: true };
86
+ }),
87
+ new Promise<HerdrProbe>((settle) => {
88
+ timer = setTimeout(() => settle({ ok: false, error: `probe timed out after ${timeoutMs}ms` }), timeoutMs);
89
+ }),
90
+ ]);
91
+ } catch (error) {
92
+ return { ok: false, error: String(error) };
93
+ } finally {
94
+ // Cleared whichever branch won, so a fast probe does not hold the event loop open for the timeout's
95
+ // remainder — which would add up to two seconds to every `node --test` run of this file.
96
+ if (timer) clearTimeout(timer);
97
+ }
98
+ }
99
+
100
+ /** herdr's own variable, set in every pane it creates. Measured 2026-08-17; documented nowhere. */
101
+ export const ENV_PARENT_WORKSPACE = "HERDR_WORKSPACE_ID";
102
+
103
+ /** The operator's explicit override. Defined here because this is the only module that reads it. */
104
+ export const ENV_HERDR_WORKSPACE = "PI_GRANTS_HERDR_WORKSPACE";
105
+
106
+ /**
107
+ * Which herdr workspace a governed child's pane belongs in.
108
+ *
109
+ * **Defaults to the parent's own workspace.** herdr tells a pane which workspace it is in
110
+ * (`HERDR_WORKSPACE_ID`, alongside `HERDR_TAB_ID` and `HERDR_PANE_ID`), and a child placed in a *different*
111
+ * workspace from the pi session that spawned it turns "switch between them" into a workspace hop — which is
112
+ * the entire feature ADR-0032 exists to deliver. The previous behaviour was "omitted lets herdr choose",
113
+ * which is that failure by default on any machine with more than one workspace.
114
+ *
115
+ * `PI_GRANTS_HERDR_WORKSPACE` still wins: it is the operator saying so explicitly, and an explicit answer
116
+ * beating an inference is this package's standing rule (ADR-0030 says it about the grant itself).
117
+ *
118
+ * Blank is treated as absent rather than passed through — `--workspace ""` is not a workspace, and it would
119
+ * fail `tab create` on a path nobody chose.
120
+ */
121
+ export function resolveWorkspace(env: NodeJS.ProcessEnv): string | undefined {
122
+ const explicit = env[ENV_HERDR_WORKSPACE]?.trim();
123
+ if (explicit) return explicit;
124
+ return env[ENV_PARENT_WORKSPACE]?.trim() || undefined;
125
+ }
@@ -0,0 +1,61 @@
1
+ /**
2
+ * Naming a herdr agent: the grammar herdr enforces, and uniqueness it does not.
3
+ *
4
+ * Split out of `src/run-herdr.ts` at the 400-line ceiling, and a real seam: both rules below come from **herdr's
5
+ * own validation and lifecycle**, not from anything this package decides. Two shipping defects lived here, and
6
+ * both were invisible to every test because the unit fake accepts whatever name it is handed and the integration
7
+ * suite never reaches a real herdr spawn. They surfaced from two real spawns against the live daemon.
8
+ */
9
+
10
+ /** Monotonic within this process. See `uniqueAgentName`. */
11
+ let spawnSeq = 0;
12
+
13
+ /**
14
+ * herdr's agent-name grammar, measured from its own rejection message.
15
+ *
16
+ * `agent name must start with a lowercase letter and contain only lowercase letters, digits, '-' or '_'
17
+ * (1-32 characters)`.
18
+ */
19
+ const AGENT_NAME_MAX = 32;
20
+
21
+ /**
22
+ * Make a herdr agent name that is **valid** and cannot collide with a live one.
23
+ *
24
+ * **Validity is a separate, PRE-EXISTING defect, and it is the more serious half.** Callers build a name as
25
+ * `${definition}-${childId}`, and a ledger child id is hierarchical — `d0.1`, `d0.1.2` (ADR-0008/F8). Those dots
26
+ * are **not in herdr's grammar**, so `agent start review-d0.1 …` is rejected with `invalid_agent_name`. Every
27
+ * `delegate({agent})` on the herdr path has therefore failed at `agent start` since the executor was written.
28
+ *
29
+ * Nothing could see it. The unit fake accepts any name it is handed, and the integration suite never reaches a
30
+ * real herdr spawn — so both were green while the feature could not work. It surfaced only by running two real
31
+ * spawns against the live daemon, which is the argument for doing that at all.
32
+ *
33
+ * **Measured, and a shipping defect without it.** herdr binds an agent name to its **tab**, and only closing
34
+ * the tab frees the name: a second `agent start` with a name still held returns
35
+ * `agent_name_taken: agent <name> is already used; … tab_id=…`. `herdr agent stop` does not exist (see
36
+ * `cleanup`), so nothing else releases it.
37
+ *
38
+ * Callers build a name from the definition and the ledger child id — and for a plain blocking `delegate` that
39
+ * id is **constant** (`d0.1`, index 0 of the session), so every delegation in a session asked for the same
40
+ * name. That was harmless while the pane closed at the end of each call. Once ADR-0032 kept panes alive to
41
+ * `agent_settled`, the **first** delegation of a turn worked and every later one failed with
42
+ * `agent_name_taken`, on the executor ADR-0031 had just made the default.
43
+ *
44
+ * Uniquified HERE rather than at the call site, so no caller can forget: the constraint belongs to herdr, and
45
+ * this module is the only thing that talks to herdr. The suffix is a counter rather than a random token so a
46
+ * pane label stays readable and reproducible within a run.
47
+ */
48
+ export function uniqueAgentName(base: string): string {
49
+ spawnSeq += 1;
50
+ const suffix = `-${spawnSeq}`;
51
+ const cleaned = base
52
+ .toLowerCase()
53
+ .replace(/[^a-z0-9_-]+/g, "-") // dots from a child id, and anything else outside the grammar
54
+ .replace(/-{2,}/g, "-")
55
+ .replace(/^[^a-z]+/, ""); // must START with a lowercase letter, so a leading digit or dash goes
56
+ // Truncated so the whole name fits, and trimmed of a trailing separator so the join stays readable. The
57
+ // fallback covers a base that sanitises to nothing at all (a definition named entirely in non-Latin script).
58
+ const room = AGENT_NAME_MAX - suffix.length;
59
+ const head = cleaned.slice(0, room).replace(/[-_]+$/, "") || "agent";
60
+ return `${head}${suffix}`;
61
+ }
@@ -0,0 +1,185 @@
1
+ /**
2
+ * Waiting for a herdr agent to settle, and reading what it printed on the way.
3
+ *
4
+ * Split out of `src/run-herdr.ts` when that file hit **404 of the 400-line ceiling** adding ADR-0032's output
5
+ * polling. The seam was named in the plan before it was needed, and it is a real one: this module is about
6
+ * *observing* an agent, `run-herdr.ts` is about *starting and cleaning up after* one. Nothing here creates or
7
+ * destroys anything.
8
+ *
9
+ * The two facts it is built on were measured against real herdr 0.7.5 (`docs/probes/g16-herdr`) and both are
10
+ * counter-intuitive enough to be worth the module comment: `agent wait --until idle` matches the state the
11
+ * agent was **already** in, and `agent read` is the one command that does **not** return a JSON envelope.
12
+ */
13
+
14
+ import { parseReply, type HerdrExec } from "./herdr-cli.ts";
15
+
16
+ /**
17
+ * What `waitForSettled` needs from a run request.
18
+ *
19
+ * Declared here rather than importing `HerdrRunRequest`, which would make the two modules mutually dependent
20
+ * for no benefit. `HerdrRunRequest` satisfies it structurally, so the call site needs no adapter.
21
+ */
22
+ export interface PollTarget {
23
+ /** herdr agent name. */
24
+ name: string;
25
+ signal?: AbortSignal;
26
+ /**
27
+ * The pane's last few lines, re-reported on every poll — a SNAPSHOT, not a stream (ADR-0032).
28
+ *
29
+ * The consumer must **replace** what it holds rather than append. `agent read` returns a snapshot of a
30
+ * bounded terminal, and treating it as append-only is what produced an 89,000× amplification; see
31
+ * `tailLines`.
32
+ */
33
+ onSnapshot?: (lines: string[]) => void;
34
+ /** How many lines the display wants. Bounds the per-poll cost regardless of how big the pane is. */
35
+ snapshotLines?: number;
36
+ /** Poll cadence override. Exists so tests do not wait `POLL_INTERVAL_MS` per state transition. */
37
+ pollIntervalMs?: number;
38
+ }
39
+
40
+ /** Statuses herdr reports for a settled agent. `blocked` counts: it is waiting for a human, not working. */
41
+ const TERMINAL = new Set(["idle", "done", "blocked"]);
42
+
43
+ /** How often to poll `agent get` while waiting for the child to settle. */
44
+ export const POLL_INTERVAL_MS = 750;
45
+ /** Lines of pane tail reported per poll. Matches the status block's own tail, so nothing is fetched unused. */
46
+ export const DEFAULT_SNAPSHOT_LINES = 3;
47
+
48
+ /**
49
+ * Wait for the child to settle, without accepting the state it was already in.
50
+ *
51
+ * **R-33, measured.** `herdr agent wait --until idle` called right after `agent prompt` returned
52
+ * *immediately*, matching the agent's **pre-existing** idle state with `state_change_seq` unchanged — a
53
+ * reply indistinguishable from a completed run. For fan-out that is not an inconvenience but a
54
+ * correctness bug: an orchestrator would "collect" N children that never ran and merge N empty results
55
+ * into a confident summary (R-03 with a new cause).
56
+ *
57
+ * So this polls `agent get` and requires **both** that the status is terminal **and** that
58
+ * `state_change_seq` has advanced past the value observed before prompting. `agent wait` is deliberately
59
+ * not used at all: its contract cannot express "settled *after* this point".
60
+ */
61
+ export async function waitForSettled(
62
+ exec: HerdrExec,
63
+ request: PollTarget,
64
+ before: number,
65
+ deadline: number,
66
+ maxOutputBytes: number,
67
+ ): Promise<{ status?: string; timedOut?: boolean; aborted?: boolean; spawnError?: string }> {
68
+ const interval = request.pollIntervalMs ?? POLL_INTERVAL_MS;
69
+ // ADR-0032. The pane is read on every poll so the parent can show what the child is doing, rather than the
70
+ // one word `delegate` for up to ten minutes.
71
+ //
72
+ // No cross-poll state is kept, deliberately: the previous design remembered what it had reported so it could
73
+ // send a diff, and that is what broke — see `tailLines`. A snapshot needs no memory.
74
+ const keep = request.snapshotLines ?? DEFAULT_SNAPSHOT_LINES;
75
+
76
+ for (;;) {
77
+ if (request.signal?.aborted) return { aborted: true };
78
+ if (Date.now() >= deadline) return { timedOut: true };
79
+
80
+ const reply = parseReply(await exec(["agent", "get", request.name]));
81
+ if (reply.error) return { spawnError: `herdr agent get failed: ${reply.error}` };
82
+
83
+ const agent = (reply.result?.agent ?? reply.result ?? {}) as { agent_status?: string; state_change_seq?: number };
84
+ const status = agent.agent_status;
85
+ const seq = typeof agent.state_change_seq === "number" ? agent.state_change_seq : -1;
86
+
87
+ if (request.onSnapshot) {
88
+ // Read BEFORE the terminal check returns, so a child that settles on this iteration still has its last
89
+ // output shown. `readFailed` is passed so a failed read renders as such instead of silently freezing the
90
+ // block on the previous frame — and, crucially, is never mistaken for the child's output.
91
+ const read = await readPane(exec, request.name, maxOutputBytes);
92
+ try {
93
+ request.onSnapshot(read.readFailed ? ["[pane could not be read]"] : tailLines(read.text, keep));
94
+ } catch {
95
+ /* display only — a renderer must not break a governed run */
96
+ }
97
+ }
98
+
99
+ if (status && TERMINAL.has(status) && seq > before) return { status };
100
+
101
+ await new Promise((r) => setTimeout(r, Math.min(interval, Math.max(0, deadline - Date.now()))));
102
+ }
103
+ }
104
+
105
+ /**
106
+ * The last `keep` non-blank lines of a pane snapshot — what the display actually needs.
107
+ *
108
+ * **This replaces a `newSuffix` diff, and the replacement is a correction rather than a tune-up.** The old
109
+ * design treated `agent read` as a *stream* and tried to report only what was new, by testing whether the new
110
+ * text extended the old. That is wrong about the substrate: `agent read` returns a **snapshot of a bounded
111
+ * terminal**, and a snapshot is not an append-only log. Two ordinary things break the prefix test forever —
112
+ * the pane **scrolling** (its top lines are gone, so the new text is not an extension of the old) and
113
+ * `readPane` **truncating to the tail** past `maxOutputBytes` (each read is a different window of a growing
114
+ * buffer). Once either happens, every poll reported the whole buffer.
115
+ *
116
+ * Measured before the fix: **51 MiB streamed for ~600 bytes of real output — 89,000× amplification** in 37
117
+ * seconds, per child, with a scrolling pane also delivering the same real lines three times each. The old
118
+ * docstring named that exact failure as the thing it prevented.
119
+ *
120
+ * So the herdr path now reports a **bounded snapshot** and the consumer *replaces* rather than appends. There
121
+ * is no diff to get wrong, the per-poll cost is `keep` lines regardless of buffer size, and a scrolling pane
122
+ * simply shows its current tail — which is what a human looking at that pane would see.
123
+ */
124
+ export function tailLines(snapshot: string, keep: number): string[] {
125
+ if (keep <= 0) return [];
126
+ const lines: string[] = [];
127
+ // Walked from the END, so a 1 MiB buffer costs the last few lines rather than a full split.
128
+ let end = snapshot.length;
129
+ while (end > 0 && lines.length < keep) {
130
+ const start = snapshot.lastIndexOf("\n", end - 1);
131
+ const line = snapshot.slice(start + 1, end).replace(/\r/g, "").trimEnd();
132
+ if (line.length > 0) lines.unshift(line);
133
+ if (start === -1) break;
134
+ end = start;
135
+ }
136
+ return lines;
137
+ }
138
+
139
+ /**
140
+ * Read the pane's contents.
141
+ *
142
+ * `agent read` is the ONE command that does not return herdr's JSON envelope — it writes the terminal's
143
+ * text straight to stdout. Running it through `parseReply` turned every successful read into
144
+ * "unparseable herdr reply", i.e. reported the child's actual answer as a failure to read it. Found by the
145
+ * end-to-end run; the unit fake had been written to the envelope shape and so agreed with the bug.
146
+ *
147
+ * A JSON envelope is still accepted first, because an `error` reply here IS JSON and must not be mistaken
148
+ * for terminal output.
149
+ *
150
+ * **`readFailed` is separate from `text`, and that separation is the fix for an R-03 defect.** A failed read
151
+ * used to return its own diagnostic *as* `text` — so `runHerdrPane` returned
152
+ * `[grants] could not read the agent pane: pane is gone` **as the child's answer, with `code: 0`**, and the
153
+ * orchestrator read a failure message as a completed sub-agent's report. Measured. It mattered little when this
154
+ * ran once per child; ADR-0032 made it run on every poll, up to 800 times for a ten-minute child, so a
155
+ * transient failure went from unlikely to expected. The caller must now decide, and it cannot do so by
156
+ * inspecting a string.
157
+ */
158
+ export async function readPane(
159
+ exec: HerdrExec,
160
+ name: string,
161
+ maxOutputBytes: number,
162
+ ): Promise<{ text: string; truncated: boolean; readFailed?: string }> {
163
+ const reply = await exec(["agent", "read", name]);
164
+ let text: string;
165
+ try {
166
+ const parsed = JSON.parse(reply.stdout) as { result?: Record<string, unknown>; error?: { message?: string } };
167
+ if (parsed.error) {
168
+ return { text: "", truncated: false, readFailed: parsed.error.message ?? "unknown error" };
169
+ }
170
+ const raw = parsed.result?.output ?? parsed.result?.text ?? parsed.result?.content ?? "";
171
+ text = typeof raw === "string" ? raw : JSON.stringify(raw);
172
+ } catch {
173
+ text = reply.stdout;
174
+ }
175
+ if (Buffer.byteLength(text) <= maxOutputBytes) return { text, truncated: false };
176
+ // Keep the TAIL, not the head: a terminal's useful content is its most recent output, and the head is
177
+ // the startup banner. `runChild` keeps the head because it streams and must stop a runaway producer;
178
+ // here the output is already complete, so the choice is free and the tail is the answer.
179
+ //
180
+ // Sliced by BYTES via a Buffer round trip rather than by code units, for `takeBytes`'s reason in
181
+ // `run-child.ts`: `slice(-maxOutputBytes)` on a CJK pane overran the cap by three times. `toString` repairs
182
+ // a character split at the cut by replacing it, which is the right trade for a tail — one replacement
183
+ // character at the boundary, versus a cap that does not hold.
184
+ return { text: Buffer.from(text, "utf8").subarray(-maxOutputBytes).toString("utf8"), truncated: true };
185
+ }
@@ -0,0 +1,55 @@
1
+ /**
2
+ * Staging a definition's instructions to a file, because herdr cannot carry them in argv.
3
+ *
4
+ * Split out of `src/run-herdr.ts` at the 400-line ceiling. A real seam rather than an arbitrary cut: this is
5
+ * the one place that works around a **herdr encoding limit**, and it is the only part of the herdr path that
6
+ * touches the filesystem. `run-herdr.ts` starts and reaps agents; this prepares one argument.
7
+ */
8
+
9
+ import { mkdtemp, writeFile } from "node:fs/promises";
10
+ import { tmpdir } from "node:os";
11
+ import { join } from "node:path";
12
+
13
+ /**
14
+ * Move a multi-line `--append-system-prompt` out of argv, because herdr cannot encode it.
15
+ *
16
+ * **Measured.** `herdr agent start` types the argv into the pane's shell, so a value containing newlines
17
+ * is rejected outright: `invalid_agent_argument — agent arguments cannot be encoded safely for the target
18
+ * shell`. A definition's `SKILL.md` body is always multi-line, so every `delegate({agent})` spawn would
19
+ * fail on this path.
20
+ *
21
+ * pi accepts a **file path** there as readily as literal text (`resolvePromptInput` + `existsSync` in
22
+ * `dist/core/resource-loader.js`), so the fix is to write the body to a temp file and pass its path — one
23
+ * short, shell-safe argument.
24
+ *
25
+ * The split lives here rather than in `planSpawn` because the constraint is **herdr's**, not pi's: the
26
+ * direct executor passes the same text inline with no trouble, and a plan builder that pre-emptively wrote
27
+ * temp files for everybody would be paying one executor's tax on both paths.
28
+ */
29
+ export function splitSystemPrompt(args: string[]): { args: string[]; systemPrompt?: string } {
30
+ const at = args.indexOf("--append-system-prompt");
31
+ if (at === -1 || at + 1 >= args.length) return { args };
32
+ return { args: [...args.slice(0, at), ...args.slice(at + 2)], systemPrompt: args[at + 1] };
33
+ }
34
+
35
+ /**
36
+ * Move a multi-line system prompt into a temp file and return argv pointing at it.
37
+ *
38
+ * Returns the directory so the caller can remove it — and **every** early return on the herdr path must, because
39
+ * nothing else can: a reviewer measured a permanent `/tmp/grants-herdr-*` per failed `tab create`, unreachable by
40
+ * either pane sweep because the pane it belonged to never existed.
41
+ */
42
+ export async function stageSystemPrompt(
43
+ args: string[],
44
+ ): Promise<{ args: string[]; promptDir?: string; error?: string }> {
45
+ const split = splitSystemPrompt(args);
46
+ if (split.systemPrompt === undefined) return { args: split.args };
47
+ try {
48
+ const promptDir = await mkdtemp(join(tmpdir(), "grants-herdr-"));
49
+ const file = join(promptDir, "system-prompt.md");
50
+ await writeFile(file, split.systemPrompt, "utf8");
51
+ return { args: [...split.args, "--append-system-prompt", file], promptDir };
52
+ } catch (error) {
53
+ return { args: split.args, error: `could not stage the system prompt for herdr: ${String(error)}` };
54
+ }
55
+ }
@@ -27,6 +27,20 @@ export interface LedgerReport {
27
27
  corrupt: Array<{ line: number; text: string }>;
28
28
  /** Records where an agent asked for more than it held — ADR-0008's designated signal. */
29
29
  escalationAttempts: number;
30
+ /**
31
+ * How many records ran under each executor, plus how many name none — ADR-0031.
32
+ *
33
+ * **Added because the field was written and never read, which is R-51's shape exactly.** R-51 was
34
+ * `definitionDigest`: recorded from the start, absent from every report, so the questions ADR-0018 advertised
35
+ * needed hand-written `jq`. `executor` arrived the same way — `src/ledger.ts` justifies making it *required*
36
+ * with "reading it back is the only reason it exists", and nothing read it back. `docs/SPEC.md` claims the
37
+ * executor is "announced three times… per child in the ledger"; without this the third announcement was to
38
+ * `jq` only.
39
+ *
40
+ * `unknown` counts pre-0.16 lines, which have no such field. Reported rather than folded into `process`,
41
+ * because "written before the executor was recorded" and "ran as a subprocess" are different facts.
42
+ */
43
+ executors: { herdr: number; process: number; unknown: number };
30
44
  /**
31
45
  * Every distinct set of instructions this ledger saw run, with how many spawns used it (R-51).
32
46
  *
@@ -113,6 +127,7 @@ export async function verifyLedger(path: string): Promise<LedgerReport> {
113
127
  records: 0,
114
128
  corrupt: [],
115
129
  escalationAttempts: 0,
130
+ executors: { herdr: 0, process: 0, unknown: 0 },
116
131
  definitions: [],
117
132
  approvals: {
118
133
  bySource: { prompt: 0, session: 0, persisted: 0, inherited: 0 },
@@ -132,6 +147,7 @@ export async function verifyLedger(path: string): Promise<LedgerReport> {
132
147
  const digests = new Map<string, { name: string; source: string; sha256: string; spawns: number }>();
133
148
  let records = 0;
134
149
  let escalationAttempts = 0;
150
+ const executors = { herdr: 0, process: 0, unknown: 0 };
135
151
  const bySource: Record<ApprovalSource, number> = { prompt: 0, session: 0, persisted: 0, inherited: 0 };
136
152
  // `capability@subject` seen per source, so the report can state a bound as well as a raw count.
137
153
  const distinct: Record<ApprovalSource, Set<string>> = {
@@ -153,6 +169,10 @@ export async function verifyLedger(path: string): Promise<LedgerReport> {
153
169
  if (!Array.isArray(parsed.denied)) throw new Error("not a grant record");
154
170
  records += 1;
155
171
  if (isEscalationAttempt(parsed)) escalationAttempts += 1;
172
+ const executor = (parsed as { executor?: unknown }).executor;
173
+ if (executor === "herdr") executors.herdr += 1;
174
+ else if (executor === "process") executors.process += 1;
175
+ else executors.unknown += 1;
156
176
  if (parsed.humanDenied) {
157
177
  humanDenied += 1;
158
178
  const subject = parsed.agentType === undefined || parsed.agentType === "delegate" ? DELEGATE_SUBJECT : parsed.agentType;
@@ -209,6 +229,7 @@ export async function verifyLedger(path: string): Promise<LedgerReport> {
209
229
  records,
210
230
  corrupt,
211
231
  escalationAttempts,
232
+ executors,
212
233
  definitions: [...digests.values()].sort((a, b) => a.name.localeCompare(b.name) || a.sha256.localeCompare(b.sha256)),
213
234
  approvals: {
214
235
  bySource,
package/src/ledger.ts CHANGED
@@ -23,6 +23,7 @@ import { appendFile, mkdir, readFile } from "node:fs/promises";
23
23
  import { withFileLock } from "./file-lock.ts";
24
24
  import { dirname } from "node:path";
25
25
  import type { Capability, ResolveResult } from "./resolve.ts";
26
+ import type { ExecutorKind } from "./executor.ts";
26
27
  import type { DefinitionDigest } from "./definitions.ts";
27
28
  import { DELEGATE_SUBJECT } from "./approval.ts";
28
29
  import type { ApprovalScope, ApprovalSource } from "./approval.ts";
@@ -107,6 +108,20 @@ export interface GrantRecord {
107
108
  * it. Absent for a `tools:`-style delegation, which has no definition.
108
109
  */
109
110
  definitionDigest?: DefinitionDigest;
111
+ /**
112
+ * WHERE this child ran — ADR-0031.
113
+ *
114
+ * **Required rather than optional**, which is unusual in this record and deliberate. Before ADR-0031 the
115
+ * executor was a variable an operator set, so "which one ran?" was answerable from configuration after the
116
+ * fact. It is now decided by a **runtime probe** at session start, so nothing outside the record preserves
117
+ * the answer — and the two paths do not produce the same argv, because the herdr plan withholds `--print`
118
+ * (`delegationContext.interactive`). A trail that cannot say where a child ran cannot be read back
119
+ * reliably, and reading it back is the only reason it exists.
120
+ *
121
+ * Written on refusals too, including the tripwire's: the honest value there is the executor the session
122
+ * *would* have used, because a refused spawn has no executor of its own.
123
+ */
124
+ executor: ExecutorKind;
110
125
  }
111
126
 
112
127
  export interface LedgerOptions {
@@ -137,6 +152,8 @@ export function buildRecord(args: {
137
152
  humanDenied?: boolean;
138
153
  gateOutcome?: PromptOutcomeKind;
139
154
  definitionDigest?: DefinitionDigest;
155
+ /** Where the child ran (ADR-0031). Required: the probe's answer survives nowhere else. */
156
+ executor: ExecutorKind;
140
157
  now: Date;
141
158
  }): GrantRecord {
142
159
  // R-46: the scalar is a SUMMARY, emitted only when it cannot mislead. `buildRecord` derives it rather
@@ -151,6 +168,7 @@ export function buildRecord(args: {
151
168
  childId: args.childId,
152
169
  depth: args.depth,
153
170
  agentType: args.agentType,
171
+ executor: args.executor,
154
172
  requested: args.requested,
155
173
  parentGrant: args.parentGrant,
156
174
  effective: args.result.effective,