pi-daddy 0.15.0 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +91 -0
- package/README.md +27 -12
- package/dist/cli.js +0 -0
- package/dist/executor.d.ts +38 -0
- package/dist/executor.d.ts.map +1 -0
- package/dist/executor.js +93 -0
- package/dist/executor.js.map +1 -0
- package/dist/herdr-cli.d.ts +78 -0
- package/dist/herdr-cli.d.ts.map +1 -0
- package/dist/herdr-cli.js +113 -0
- package/dist/herdr-cli.js.map +1 -0
- package/dist/herdr-name.d.ts +37 -0
- package/dist/herdr-name.d.ts.map +1 -0
- package/dist/herdr-name.js +59 -0
- package/dist/herdr-name.js.map +1 -0
- package/dist/herdr-poll.d.ts +104 -0
- package/dist/herdr-poll.d.ts.map +1 -0
- package/dist/herdr-poll.js +150 -0
- package/dist/herdr-poll.js.map +1 -0
- package/dist/herdr-stage.d.ts +40 -0
- package/dist/herdr-stage.d.ts.map +1 -0
- package/dist/herdr-stage.js +54 -0
- package/dist/herdr-stage.js.map +1 -0
- package/dist/ledger-report.d.ts +18 -0
- package/dist/ledger-report.d.ts.map +1 -1
- package/dist/ledger-report.js +10 -0
- package/dist/ledger-report.js.map +1 -1
- package/dist/ledger.d.ts +17 -0
- package/dist/ledger.d.ts.map +1 -1
- package/dist/ledger.js +1 -0
- package/dist/ledger.js.map +1 -1
- package/dist/pane-reaper.d.ts +66 -4
- package/dist/pane-reaper.d.ts.map +1 -1
- package/dist/pane-reaper.js +131 -9
- package/dist/pane-reaper.js.map +1 -1
- package/dist/progress.d.ts +96 -0
- package/dist/progress.d.ts.map +1 -0
- package/dist/progress.js +167 -0
- package/dist/progress.js.map +1 -0
- package/dist/run-child.d.ts +27 -0
- package/dist/run-child.d.ts.map +1 -1
- package/dist/run-child.js +84 -7
- package/dist/run-child.js.map +1 -1
- package/dist/run-herdr.d.ts +41 -28
- package/dist/run-herdr.d.ts.map +1 -1
- package/dist/run-herdr.js +150 -167
- package/dist/run-herdr.js.map +1 -1
- package/extensions/delegation.ts +94 -2
- package/extensions/grants-command.ts +26 -1
- package/extensions/grants.ts +85 -163
- package/extensions/run-delegation.ts +70 -8
- package/extensions/session-report.ts +231 -0
- package/extensions/session.ts +63 -11
- package/extensions/tripwire.ts +44 -0
- package/package.json +17 -1
- package/src/executor.ts +122 -0
- package/src/herdr-cli.ts +125 -0
- package/src/herdr-name.ts +61 -0
- package/src/herdr-poll.ts +185 -0
- package/src/herdr-stage.ts +55 -0
- package/src/ledger-report.ts +21 -0
- package/src/ledger.ts +18 -0
- package/src/pane-reaper.ts +147 -9
- package/src/progress.ts +206 -0
- package/src/run-child.ts +96 -7
- package/src/run-herdr.ts +170 -174
package/src/herdr-cli.ts
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Talking to herdr: one command, one JSON envelope, plus the two questions ADR-0031 needs answered.
|
|
3
|
+
*
|
|
4
|
+
* Lifted out of `src/run-herdr.ts`, which was at 357 of the 400-line ceiling and gains output polling under
|
|
5
|
+
* ADR-0032. But the split is not only about lines: **the probe is not an executor concern**. It runs at session
|
|
6
|
+
* start, before any delegation exists, to decide *which* executor a session will use — so leaving it inside the
|
|
7
|
+
* herdr executor would mean the session imported the thing it was deciding whether to use.
|
|
8
|
+
*
|
|
9
|
+
* Every rule here is tested against an injected `exec`, so the suite stays fast, pi-free and herdr-free. The
|
|
10
|
+
* facts the fakes reproduce were measured against real herdr 0.7.5 (`docs/probes/g16-herdr`).
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { execFile } from "node:child_process";
|
|
14
|
+
|
|
15
|
+
/** One herdr CLI invocation. Injectable so every rule below is testable without herdr installed. */
|
|
16
|
+
export type HerdrExec = (args: string[]) => Promise<{ code: number | null; stdout: string; stderr: string }>;
|
|
17
|
+
|
|
18
|
+
export const defaultExec: HerdrExec = (args) =>
|
|
19
|
+
new Promise((settle) => {
|
|
20
|
+
execFile("herdr", args, { maxBuffer: 32 * 1024 * 1024 }, (error, stdout, stderr) => {
|
|
21
|
+
const raw = (error as { code?: unknown } | null)?.code;
|
|
22
|
+
const code = typeof raw === "number" ? raw : error ? 1 : 0;
|
|
23
|
+
// **A string `code` is a spawn failure, and it used to be thrown away.** `ENOENT` — herdr not installed —
|
|
24
|
+
// arrives as `code: "ENOENT"`, so the numeric test failed, the message was dropped, and an operator with
|
|
25
|
+
// `PI_GRANTS_HERDR=1` on a machine without herdr was told *"herdr is not answering (unparseable herdr
|
|
26
|
+
// reply: (no output))"* rather than that the binary is missing. Rule 8 wants the loud version, and this is
|
|
27
|
+
// the first diagnostic such an operator meets.
|
|
28
|
+
const spawnFailure = typeof raw === "string" ? `herdr could not be run (${raw}): ${error?.message ?? ""}` : "";
|
|
29
|
+
settle({ code, stdout: String(stdout), stderr: spawnFailure || String(stderr) });
|
|
30
|
+
});
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Parse herdr's JSON envelope. Every command replies `{id, result}` or `{id, error:{code,message}}`.
|
|
35
|
+
*
|
|
36
|
+
* `stderr` is folded into the message because the first end-to-end run failed with an EMPTY stdout and the
|
|
37
|
+
* real reason on stderr, producing the useless diagnostic "unparseable herdr reply: ". A wrapper that
|
|
38
|
+
* hides the substrate's own error message costs more time than it saves.
|
|
39
|
+
*/
|
|
40
|
+
export function parseReply(reply: { stdout: string; stderr: string }): { result?: Record<string, unknown>; error?: string } {
|
|
41
|
+
try {
|
|
42
|
+
const parsed = JSON.parse(reply.stdout) as { result?: Record<string, unknown>; error?: { message?: string; code?: string } };
|
|
43
|
+
if (parsed.error) return { error: parsed.error.message ?? parsed.error.code ?? "herdr reported an error" };
|
|
44
|
+
return { result: parsed.result };
|
|
45
|
+
} catch {
|
|
46
|
+
// A non-JSON reply is a herdr-version or PATH problem, not a governance decision. Surfaced as a spawn
|
|
47
|
+
// error so the caller reports "could not start" rather than "the child produced nothing".
|
|
48
|
+
const detail = [reply.stdout.trim(), reply.stderr.trim()].filter((t) => t.length > 0).join(" | ");
|
|
49
|
+
return { error: `unparseable herdr reply: ${detail.slice(0, 300) || "(no output)"}` };
|
|
50
|
+
}
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
/** Bound on the session-start probe. Short: it sits in front of the operator's first prompt. */
|
|
54
|
+
export const PROBE_TIMEOUT_MS = 2000;
|
|
55
|
+
|
|
56
|
+
export interface HerdrProbe {
|
|
57
|
+
ok: boolean;
|
|
58
|
+
/** herdr's own words when it is not reachable. Carried so the disclosure line can name the reason. */
|
|
59
|
+
error?: string;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Is there a herdr server that will answer right now? — ADR-0031's selection input.
|
|
64
|
+
*
|
|
65
|
+
* **`tab list`, not `which herdr`.** ADR-0031 rejects `PATH` detection as option C by name: a binary on `PATH`
|
|
66
|
+
* with no server behind it would make every delegation fail at `tab create`, on a path the operator never
|
|
67
|
+
* chose, and the diagnostic would arrive at the first delegation rather than at startup. Only a parsed
|
|
68
|
+
* `result` envelope counts as reachable; an `error` envelope, a non-JSON reply, a timeout and a throwing
|
|
69
|
+
* `exec` are all "not reachable" with the reason preserved.
|
|
70
|
+
*
|
|
71
|
+
* **Zero tabs is a successful answer**, deliberately: a fresh herdr with nothing open is reachable.
|
|
72
|
+
*
|
|
73
|
+
* Never throws. A probe that threw out of `session_start` would cancel every control after it, which is
|
|
74
|
+
* R-60's shape exactly — and this one runs *before* the line that discloses what it decided.
|
|
75
|
+
*/
|
|
76
|
+
export async function probeHerdr(options: { exec?: HerdrExec; timeoutMs?: number } = {}): Promise<HerdrProbe> {
|
|
77
|
+
const exec = options.exec ?? defaultExec;
|
|
78
|
+
const timeoutMs = options.timeoutMs ?? PROBE_TIMEOUT_MS;
|
|
79
|
+
|
|
80
|
+
let timer: NodeJS.Timeout | undefined;
|
|
81
|
+
try {
|
|
82
|
+
return await Promise.race<HerdrProbe>([
|
|
83
|
+
exec(["tab", "list"]).then((reply) => {
|
|
84
|
+
const parsed = parseReply(reply);
|
|
85
|
+
return parsed.error ? { ok: false, error: parsed.error } : { ok: true };
|
|
86
|
+
}),
|
|
87
|
+
new Promise<HerdrProbe>((settle) => {
|
|
88
|
+
timer = setTimeout(() => settle({ ok: false, error: `probe timed out after ${timeoutMs}ms` }), timeoutMs);
|
|
89
|
+
}),
|
|
90
|
+
]);
|
|
91
|
+
} catch (error) {
|
|
92
|
+
return { ok: false, error: String(error) };
|
|
93
|
+
} finally {
|
|
94
|
+
// Cleared whichever branch won, so a fast probe does not hold the event loop open for the timeout's
|
|
95
|
+
// remainder — which would add up to two seconds to every `node --test` run of this file.
|
|
96
|
+
if (timer) clearTimeout(timer);
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/** herdr's own variable, set in every pane it creates. Measured 2026-08-17; documented nowhere. */
|
|
101
|
+
export const ENV_PARENT_WORKSPACE = "HERDR_WORKSPACE_ID";
|
|
102
|
+
|
|
103
|
+
/** The operator's explicit override. Defined here because this is the only module that reads it. */
|
|
104
|
+
export const ENV_HERDR_WORKSPACE = "PI_GRANTS_HERDR_WORKSPACE";
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Which herdr workspace a governed child's pane belongs in.
|
|
108
|
+
*
|
|
109
|
+
* **Defaults to the parent's own workspace.** herdr tells a pane which workspace it is in
|
|
110
|
+
* (`HERDR_WORKSPACE_ID`, alongside `HERDR_TAB_ID` and `HERDR_PANE_ID`), and a child placed in a *different*
|
|
111
|
+
* workspace from the pi session that spawned it turns "switch between them" into a workspace hop — which is
|
|
112
|
+
* the entire feature ADR-0032 exists to deliver. The previous behaviour was "omitted lets herdr choose",
|
|
113
|
+
* which is that failure by default on any machine with more than one workspace.
|
|
114
|
+
*
|
|
115
|
+
* `PI_GRANTS_HERDR_WORKSPACE` still wins: it is the operator saying so explicitly, and an explicit answer
|
|
116
|
+
* beating an inference is this package's standing rule (ADR-0030 says it about the grant itself).
|
|
117
|
+
*
|
|
118
|
+
* Blank is treated as absent rather than passed through — `--workspace ""` is not a workspace, and it would
|
|
119
|
+
* fail `tab create` on a path nobody chose.
|
|
120
|
+
*/
|
|
121
|
+
export function resolveWorkspace(env: NodeJS.ProcessEnv): string | undefined {
|
|
122
|
+
const explicit = env[ENV_HERDR_WORKSPACE]?.trim();
|
|
123
|
+
if (explicit) return explicit;
|
|
124
|
+
return env[ENV_PARENT_WORKSPACE]?.trim() || undefined;
|
|
125
|
+
}
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Naming a herdr agent: the grammar herdr enforces, and uniqueness it does not.
|
|
3
|
+
*
|
|
4
|
+
* Split out of `src/run-herdr.ts` at the 400-line ceiling, and a real seam: both rules below come from **herdr's
|
|
5
|
+
* own validation and lifecycle**, not from anything this package decides. Two shipping defects lived here, and
|
|
6
|
+
* both were invisible to every test because the unit fake accepts whatever name it is handed and the integration
|
|
7
|
+
* suite never reaches a real herdr spawn. They surfaced from two real spawns against the live daemon.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
/** Monotonic within this process. See `uniqueAgentName`. */
|
|
11
|
+
let spawnSeq = 0;
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* herdr's agent-name grammar, measured from its own rejection message.
|
|
15
|
+
*
|
|
16
|
+
* `agent name must start with a lowercase letter and contain only lowercase letters, digits, '-' or '_'
|
|
17
|
+
* (1-32 characters)`.
|
|
18
|
+
*/
|
|
19
|
+
const AGENT_NAME_MAX = 32;
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* Make a herdr agent name that is **valid** and cannot collide with a live one.
|
|
23
|
+
*
|
|
24
|
+
* **Validity is a separate, PRE-EXISTING defect, and it is the more serious half.** Callers build a name as
|
|
25
|
+
* `${definition}-${childId}`, and a ledger child id is hierarchical — `d0.1`, `d0.1.2` (ADR-0008/F8). Those dots
|
|
26
|
+
* are **not in herdr's grammar**, so `agent start review-d0.1 …` is rejected with `invalid_agent_name`. Every
|
|
27
|
+
* `delegate({agent})` on the herdr path has therefore failed at `agent start` since the executor was written.
|
|
28
|
+
*
|
|
29
|
+
* Nothing could see it. The unit fake accepts any name it is handed, and the integration suite never reaches a
|
|
30
|
+
* real herdr spawn — so both were green while the feature could not work. It surfaced only by running two real
|
|
31
|
+
* spawns against the live daemon, which is the argument for doing that at all.
|
|
32
|
+
*
|
|
33
|
+
* **Measured, and a shipping defect without it.** herdr binds an agent name to its **tab**, and only closing
|
|
34
|
+
* the tab frees the name: a second `agent start` with a name still held returns
|
|
35
|
+
* `agent_name_taken: agent <name> is already used; … tab_id=…`. `herdr agent stop` does not exist (see
|
|
36
|
+
* `cleanup`), so nothing else releases it.
|
|
37
|
+
*
|
|
38
|
+
* Callers build a name from the definition and the ledger child id — and for a plain blocking `delegate` that
|
|
39
|
+
* id is **constant** (`d0.1`, index 0 of the session), so every delegation in a session asked for the same
|
|
40
|
+
* name. That was harmless while the pane closed at the end of each call. Once ADR-0032 kept panes alive to
|
|
41
|
+
* `agent_settled`, the **first** delegation of a turn worked and every later one failed with
|
|
42
|
+
* `agent_name_taken`, on the executor ADR-0031 had just made the default.
|
|
43
|
+
*
|
|
44
|
+
* Uniquified HERE rather than at the call site, so no caller can forget: the constraint belongs to herdr, and
|
|
45
|
+
* this module is the only thing that talks to herdr. The suffix is a counter rather than a random token so a
|
|
46
|
+
* pane label stays readable and reproducible within a run.
|
|
47
|
+
*/
|
|
48
|
+
export function uniqueAgentName(base: string): string {
|
|
49
|
+
spawnSeq += 1;
|
|
50
|
+
const suffix = `-${spawnSeq}`;
|
|
51
|
+
const cleaned = base
|
|
52
|
+
.toLowerCase()
|
|
53
|
+
.replace(/[^a-z0-9_-]+/g, "-") // dots from a child id, and anything else outside the grammar
|
|
54
|
+
.replace(/-{2,}/g, "-")
|
|
55
|
+
.replace(/^[^a-z]+/, ""); // must START with a lowercase letter, so a leading digit or dash goes
|
|
56
|
+
// Truncated so the whole name fits, and trimmed of a trailing separator so the join stays readable. The
|
|
57
|
+
// fallback covers a base that sanitises to nothing at all (a definition named entirely in non-Latin script).
|
|
58
|
+
const room = AGENT_NAME_MAX - suffix.length;
|
|
59
|
+
const head = cleaned.slice(0, room).replace(/[-_]+$/, "") || "agent";
|
|
60
|
+
return `${head}${suffix}`;
|
|
61
|
+
}
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Waiting for a herdr agent to settle, and reading what it printed on the way.
|
|
3
|
+
*
|
|
4
|
+
* Split out of `src/run-herdr.ts` when that file hit **404 of the 400-line ceiling** adding ADR-0032's output
|
|
5
|
+
* polling. The seam was named in the plan before it was needed, and it is a real one: this module is about
|
|
6
|
+
* *observing* an agent, `run-herdr.ts` is about *starting and cleaning up after* one. Nothing here creates or
|
|
7
|
+
* destroys anything.
|
|
8
|
+
*
|
|
9
|
+
* The two facts it is built on were measured against real herdr 0.7.5 (`docs/probes/g16-herdr`) and both are
|
|
10
|
+
* counter-intuitive enough to be worth the module comment: `agent wait --until idle` matches the state the
|
|
11
|
+
* agent was **already** in, and `agent read` is the one command that does **not** return a JSON envelope.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
import { parseReply, type HerdrExec } from "./herdr-cli.ts";
|
|
15
|
+
|
|
16
|
+
/**
|
|
17
|
+
* What `waitForSettled` needs from a run request.
|
|
18
|
+
*
|
|
19
|
+
* Declared here rather than importing `HerdrRunRequest`, which would make the two modules mutually dependent
|
|
20
|
+
* for no benefit. `HerdrRunRequest` satisfies it structurally, so the call site needs no adapter.
|
|
21
|
+
*/
|
|
22
|
+
export interface PollTarget {
|
|
23
|
+
/** herdr agent name. */
|
|
24
|
+
name: string;
|
|
25
|
+
signal?: AbortSignal;
|
|
26
|
+
/**
|
|
27
|
+
* The pane's last few lines, re-reported on every poll — a SNAPSHOT, not a stream (ADR-0032).
|
|
28
|
+
*
|
|
29
|
+
* The consumer must **replace** what it holds rather than append. `agent read` returns a snapshot of a
|
|
30
|
+
* bounded terminal, and treating it as append-only is what produced an 89,000× amplification; see
|
|
31
|
+
* `tailLines`.
|
|
32
|
+
*/
|
|
33
|
+
onSnapshot?: (lines: string[]) => void;
|
|
34
|
+
/** How many lines the display wants. Bounds the per-poll cost regardless of how big the pane is. */
|
|
35
|
+
snapshotLines?: number;
|
|
36
|
+
/** Poll cadence override. Exists so tests do not wait `POLL_INTERVAL_MS` per state transition. */
|
|
37
|
+
pollIntervalMs?: number;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
/** Statuses herdr reports for a settled agent. `blocked` counts: it is waiting for a human, not working. */
|
|
41
|
+
const TERMINAL = new Set(["idle", "done", "blocked"]);
|
|
42
|
+
|
|
43
|
+
/** How often to poll `agent get` while waiting for the child to settle. */
|
|
44
|
+
export const POLL_INTERVAL_MS = 750;
|
|
45
|
+
/** Lines of pane tail reported per poll. Matches the status block's own tail, so nothing is fetched unused. */
|
|
46
|
+
export const DEFAULT_SNAPSHOT_LINES = 3;
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Wait for the child to settle, without accepting the state it was already in.
|
|
50
|
+
*
|
|
51
|
+
* **R-33, measured.** `herdr agent wait --until idle` called right after `agent prompt` returned
|
|
52
|
+
* *immediately*, matching the agent's **pre-existing** idle state with `state_change_seq` unchanged — a
|
|
53
|
+
* reply indistinguishable from a completed run. For fan-out that is not an inconvenience but a
|
|
54
|
+
* correctness bug: an orchestrator would "collect" N children that never ran and merge N empty results
|
|
55
|
+
* into a confident summary (R-03 with a new cause).
|
|
56
|
+
*
|
|
57
|
+
* So this polls `agent get` and requires **both** that the status is terminal **and** that
|
|
58
|
+
* `state_change_seq` has advanced past the value observed before prompting. `agent wait` is deliberately
|
|
59
|
+
* not used at all: its contract cannot express "settled *after* this point".
|
|
60
|
+
*/
|
|
61
|
+
export async function waitForSettled(
|
|
62
|
+
exec: HerdrExec,
|
|
63
|
+
request: PollTarget,
|
|
64
|
+
before: number,
|
|
65
|
+
deadline: number,
|
|
66
|
+
maxOutputBytes: number,
|
|
67
|
+
): Promise<{ status?: string; timedOut?: boolean; aborted?: boolean; spawnError?: string }> {
|
|
68
|
+
const interval = request.pollIntervalMs ?? POLL_INTERVAL_MS;
|
|
69
|
+
// ADR-0032. The pane is read on every poll so the parent can show what the child is doing, rather than the
|
|
70
|
+
// one word `delegate` for up to ten minutes.
|
|
71
|
+
//
|
|
72
|
+
// No cross-poll state is kept, deliberately: the previous design remembered what it had reported so it could
|
|
73
|
+
// send a diff, and that is what broke — see `tailLines`. A snapshot needs no memory.
|
|
74
|
+
const keep = request.snapshotLines ?? DEFAULT_SNAPSHOT_LINES;
|
|
75
|
+
|
|
76
|
+
for (;;) {
|
|
77
|
+
if (request.signal?.aborted) return { aborted: true };
|
|
78
|
+
if (Date.now() >= deadline) return { timedOut: true };
|
|
79
|
+
|
|
80
|
+
const reply = parseReply(await exec(["agent", "get", request.name]));
|
|
81
|
+
if (reply.error) return { spawnError: `herdr agent get failed: ${reply.error}` };
|
|
82
|
+
|
|
83
|
+
const agent = (reply.result?.agent ?? reply.result ?? {}) as { agent_status?: string; state_change_seq?: number };
|
|
84
|
+
const status = agent.agent_status;
|
|
85
|
+
const seq = typeof agent.state_change_seq === "number" ? agent.state_change_seq : -1;
|
|
86
|
+
|
|
87
|
+
if (request.onSnapshot) {
|
|
88
|
+
// Read BEFORE the terminal check returns, so a child that settles on this iteration still has its last
|
|
89
|
+
// output shown. `readFailed` is passed so a failed read renders as such instead of silently freezing the
|
|
90
|
+
// block on the previous frame — and, crucially, is never mistaken for the child's output.
|
|
91
|
+
const read = await readPane(exec, request.name, maxOutputBytes);
|
|
92
|
+
try {
|
|
93
|
+
request.onSnapshot(read.readFailed ? ["[pane could not be read]"] : tailLines(read.text, keep));
|
|
94
|
+
} catch {
|
|
95
|
+
/* display only — a renderer must not break a governed run */
|
|
96
|
+
}
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
if (status && TERMINAL.has(status) && seq > before) return { status };
|
|
100
|
+
|
|
101
|
+
await new Promise((r) => setTimeout(r, Math.min(interval, Math.max(0, deadline - Date.now()))));
|
|
102
|
+
}
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* The last `keep` non-blank lines of a pane snapshot — what the display actually needs.
|
|
107
|
+
*
|
|
108
|
+
* **This replaces a `newSuffix` diff, and the replacement is a correction rather than a tune-up.** The old
|
|
109
|
+
* design treated `agent read` as a *stream* and tried to report only what was new, by testing whether the new
|
|
110
|
+
* text extended the old. That is wrong about the substrate: `agent read` returns a **snapshot of a bounded
|
|
111
|
+
* terminal**, and a snapshot is not an append-only log. Two ordinary things break the prefix test forever —
|
|
112
|
+
* the pane **scrolling** (its top lines are gone, so the new text is not an extension of the old) and
|
|
113
|
+
* `readPane` **truncating to the tail** past `maxOutputBytes` (each read is a different window of a growing
|
|
114
|
+
* buffer). Once either happens, every poll reported the whole buffer.
|
|
115
|
+
*
|
|
116
|
+
* Measured before the fix: **51 MiB streamed for ~600 bytes of real output — 89,000× amplification** in 37
|
|
117
|
+
* seconds, per child, with a scrolling pane also delivering the same real lines three times each. The old
|
|
118
|
+
* docstring named that exact failure as the thing it prevented.
|
|
119
|
+
*
|
|
120
|
+
* So the herdr path now reports a **bounded snapshot** and the consumer *replaces* rather than appends. There
|
|
121
|
+
* is no diff to get wrong, the per-poll cost is `keep` lines regardless of buffer size, and a scrolling pane
|
|
122
|
+
* simply shows its current tail — which is what a human looking at that pane would see.
|
|
123
|
+
*/
|
|
124
|
+
export function tailLines(snapshot: string, keep: number): string[] {
|
|
125
|
+
if (keep <= 0) return [];
|
|
126
|
+
const lines: string[] = [];
|
|
127
|
+
// Walked from the END, so a 1 MiB buffer costs the last few lines rather than a full split.
|
|
128
|
+
let end = snapshot.length;
|
|
129
|
+
while (end > 0 && lines.length < keep) {
|
|
130
|
+
const start = snapshot.lastIndexOf("\n", end - 1);
|
|
131
|
+
const line = snapshot.slice(start + 1, end).replace(/\r/g, "").trimEnd();
|
|
132
|
+
if (line.length > 0) lines.unshift(line);
|
|
133
|
+
if (start === -1) break;
|
|
134
|
+
end = start;
|
|
135
|
+
}
|
|
136
|
+
return lines;
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* Read the pane's contents.
|
|
141
|
+
*
|
|
142
|
+
* `agent read` is the ONE command that does not return herdr's JSON envelope — it writes the terminal's
|
|
143
|
+
* text straight to stdout. Running it through `parseReply` turned every successful read into
|
|
144
|
+
* "unparseable herdr reply", i.e. reported the child's actual answer as a failure to read it. Found by the
|
|
145
|
+
* end-to-end run; the unit fake had been written to the envelope shape and so agreed with the bug.
|
|
146
|
+
*
|
|
147
|
+
* A JSON envelope is still accepted first, because an `error` reply here IS JSON and must not be mistaken
|
|
148
|
+
* for terminal output.
|
|
149
|
+
*
|
|
150
|
+
* **`readFailed` is separate from `text`, and that separation is the fix for an R-03 defect.** A failed read
|
|
151
|
+
* used to return its own diagnostic *as* `text` — so `runHerdrPane` returned
|
|
152
|
+
* `[grants] could not read the agent pane: pane is gone` **as the child's answer, with `code: 0`**, and the
|
|
153
|
+
* orchestrator read a failure message as a completed sub-agent's report. Measured. It mattered little when this
|
|
154
|
+
* ran once per child; ADR-0032 made it run on every poll, up to 800 times for a ten-minute child, so a
|
|
155
|
+
* transient failure went from unlikely to expected. The caller must now decide, and it cannot do so by
|
|
156
|
+
* inspecting a string.
|
|
157
|
+
*/
|
|
158
|
+
export async function readPane(
|
|
159
|
+
exec: HerdrExec,
|
|
160
|
+
name: string,
|
|
161
|
+
maxOutputBytes: number,
|
|
162
|
+
): Promise<{ text: string; truncated: boolean; readFailed?: string }> {
|
|
163
|
+
const reply = await exec(["agent", "read", name]);
|
|
164
|
+
let text: string;
|
|
165
|
+
try {
|
|
166
|
+
const parsed = JSON.parse(reply.stdout) as { result?: Record<string, unknown>; error?: { message?: string } };
|
|
167
|
+
if (parsed.error) {
|
|
168
|
+
return { text: "", truncated: false, readFailed: parsed.error.message ?? "unknown error" };
|
|
169
|
+
}
|
|
170
|
+
const raw = parsed.result?.output ?? parsed.result?.text ?? parsed.result?.content ?? "";
|
|
171
|
+
text = typeof raw === "string" ? raw : JSON.stringify(raw);
|
|
172
|
+
} catch {
|
|
173
|
+
text = reply.stdout;
|
|
174
|
+
}
|
|
175
|
+
if (Buffer.byteLength(text) <= maxOutputBytes) return { text, truncated: false };
|
|
176
|
+
// Keep the TAIL, not the head: a terminal's useful content is its most recent output, and the head is
|
|
177
|
+
// the startup banner. `runChild` keeps the head because it streams and must stop a runaway producer;
|
|
178
|
+
// here the output is already complete, so the choice is free and the tail is the answer.
|
|
179
|
+
//
|
|
180
|
+
// Sliced by BYTES via a Buffer round trip rather than by code units, for `takeBytes`'s reason in
|
|
181
|
+
// `run-child.ts`: `slice(-maxOutputBytes)` on a CJK pane overran the cap by three times. `toString` repairs
|
|
182
|
+
// a character split at the cut by replacing it, which is the right trade for a tail — one replacement
|
|
183
|
+
// character at the boundary, versus a cap that does not hold.
|
|
184
|
+
return { text: Buffer.from(text, "utf8").subarray(-maxOutputBytes).toString("utf8"), truncated: true };
|
|
185
|
+
}
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Staging a definition's instructions to a file, because herdr cannot carry them in argv.
|
|
3
|
+
*
|
|
4
|
+
* Split out of `src/run-herdr.ts` at the 400-line ceiling. A real seam rather than an arbitrary cut: this is
|
|
5
|
+
* the one place that works around a **herdr encoding limit**, and it is the only part of the herdr path that
|
|
6
|
+
* touches the filesystem. `run-herdr.ts` starts and reaps agents; this prepares one argument.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { mkdtemp, writeFile } from "node:fs/promises";
|
|
10
|
+
import { tmpdir } from "node:os";
|
|
11
|
+
import { join } from "node:path";
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Move a multi-line `--append-system-prompt` out of argv, because herdr cannot encode it.
|
|
15
|
+
*
|
|
16
|
+
* **Measured.** `herdr agent start` types the argv into the pane's shell, so a value containing newlines
|
|
17
|
+
* is rejected outright: `invalid_agent_argument — agent arguments cannot be encoded safely for the target
|
|
18
|
+
* shell`. A definition's `SKILL.md` body is always multi-line, so every `delegate({agent})` spawn would
|
|
19
|
+
* fail on this path.
|
|
20
|
+
*
|
|
21
|
+
* pi accepts a **file path** there as readily as literal text (`resolvePromptInput` + `existsSync` in
|
|
22
|
+
* `dist/core/resource-loader.js`), so the fix is to write the body to a temp file and pass its path — one
|
|
23
|
+
* short, shell-safe argument.
|
|
24
|
+
*
|
|
25
|
+
* The split lives here rather than in `planSpawn` because the constraint is **herdr's**, not pi's: the
|
|
26
|
+
* direct executor passes the same text inline with no trouble, and a plan builder that pre-emptively wrote
|
|
27
|
+
* temp files for everybody would be paying one executor's tax on both paths.
|
|
28
|
+
*/
|
|
29
|
+
export function splitSystemPrompt(args: string[]): { args: string[]; systemPrompt?: string } {
|
|
30
|
+
const at = args.indexOf("--append-system-prompt");
|
|
31
|
+
if (at === -1 || at + 1 >= args.length) return { args };
|
|
32
|
+
return { args: [...args.slice(0, at), ...args.slice(at + 2)], systemPrompt: args[at + 1] };
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Move a multi-line system prompt into a temp file and return argv pointing at it.
|
|
37
|
+
*
|
|
38
|
+
* Returns the directory so the caller can remove it — and **every** early return on the herdr path must, because
|
|
39
|
+
* nothing else can: a reviewer measured a permanent `/tmp/grants-herdr-*` per failed `tab create`, unreachable by
|
|
40
|
+
* either pane sweep because the pane it belonged to never existed.
|
|
41
|
+
*/
|
|
42
|
+
export async function stageSystemPrompt(
|
|
43
|
+
args: string[],
|
|
44
|
+
): Promise<{ args: string[]; promptDir?: string; error?: string }> {
|
|
45
|
+
const split = splitSystemPrompt(args);
|
|
46
|
+
if (split.systemPrompt === undefined) return { args: split.args };
|
|
47
|
+
try {
|
|
48
|
+
const promptDir = await mkdtemp(join(tmpdir(), "grants-herdr-"));
|
|
49
|
+
const file = join(promptDir, "system-prompt.md");
|
|
50
|
+
await writeFile(file, split.systemPrompt, "utf8");
|
|
51
|
+
return { args: [...split.args, "--append-system-prompt", file], promptDir };
|
|
52
|
+
} catch (error) {
|
|
53
|
+
return { args: split.args, error: `could not stage the system prompt for herdr: ${String(error)}` };
|
|
54
|
+
}
|
|
55
|
+
}
|
package/src/ledger-report.ts
CHANGED
|
@@ -27,6 +27,20 @@ export interface LedgerReport {
|
|
|
27
27
|
corrupt: Array<{ line: number; text: string }>;
|
|
28
28
|
/** Records where an agent asked for more than it held — ADR-0008's designated signal. */
|
|
29
29
|
escalationAttempts: number;
|
|
30
|
+
/**
|
|
31
|
+
* How many records ran under each executor, plus how many name none — ADR-0031.
|
|
32
|
+
*
|
|
33
|
+
* **Added because the field was written and never read, which is R-51's shape exactly.** R-51 was
|
|
34
|
+
* `definitionDigest`: recorded from the start, absent from every report, so the questions ADR-0018 advertised
|
|
35
|
+
* needed hand-written `jq`. `executor` arrived the same way — `src/ledger.ts` justifies making it *required*
|
|
36
|
+
* with "reading it back is the only reason it exists", and nothing read it back. `docs/SPEC.md` claims the
|
|
37
|
+
* executor is "announced three times… per child in the ledger"; without this the third announcement was to
|
|
38
|
+
* `jq` only.
|
|
39
|
+
*
|
|
40
|
+
* `unknown` counts pre-0.16 lines, which have no such field. Reported rather than folded into `process`,
|
|
41
|
+
* because "written before the executor was recorded" and "ran as a subprocess" are different facts.
|
|
42
|
+
*/
|
|
43
|
+
executors: { herdr: number; process: number; unknown: number };
|
|
30
44
|
/**
|
|
31
45
|
* Every distinct set of instructions this ledger saw run, with how many spawns used it (R-51).
|
|
32
46
|
*
|
|
@@ -113,6 +127,7 @@ export async function verifyLedger(path: string): Promise<LedgerReport> {
|
|
|
113
127
|
records: 0,
|
|
114
128
|
corrupt: [],
|
|
115
129
|
escalationAttempts: 0,
|
|
130
|
+
executors: { herdr: 0, process: 0, unknown: 0 },
|
|
116
131
|
definitions: [],
|
|
117
132
|
approvals: {
|
|
118
133
|
bySource: { prompt: 0, session: 0, persisted: 0, inherited: 0 },
|
|
@@ -132,6 +147,7 @@ export async function verifyLedger(path: string): Promise<LedgerReport> {
|
|
|
132
147
|
const digests = new Map<string, { name: string; source: string; sha256: string; spawns: number }>();
|
|
133
148
|
let records = 0;
|
|
134
149
|
let escalationAttempts = 0;
|
|
150
|
+
const executors = { herdr: 0, process: 0, unknown: 0 };
|
|
135
151
|
const bySource: Record<ApprovalSource, number> = { prompt: 0, session: 0, persisted: 0, inherited: 0 };
|
|
136
152
|
// `capability@subject` seen per source, so the report can state a bound as well as a raw count.
|
|
137
153
|
const distinct: Record<ApprovalSource, Set<string>> = {
|
|
@@ -153,6 +169,10 @@ export async function verifyLedger(path: string): Promise<LedgerReport> {
|
|
|
153
169
|
if (!Array.isArray(parsed.denied)) throw new Error("not a grant record");
|
|
154
170
|
records += 1;
|
|
155
171
|
if (isEscalationAttempt(parsed)) escalationAttempts += 1;
|
|
172
|
+
const executor = (parsed as { executor?: unknown }).executor;
|
|
173
|
+
if (executor === "herdr") executors.herdr += 1;
|
|
174
|
+
else if (executor === "process") executors.process += 1;
|
|
175
|
+
else executors.unknown += 1;
|
|
156
176
|
if (parsed.humanDenied) {
|
|
157
177
|
humanDenied += 1;
|
|
158
178
|
const subject = parsed.agentType === undefined || parsed.agentType === "delegate" ? DELEGATE_SUBJECT : parsed.agentType;
|
|
@@ -209,6 +229,7 @@ export async function verifyLedger(path: string): Promise<LedgerReport> {
|
|
|
209
229
|
records,
|
|
210
230
|
corrupt,
|
|
211
231
|
escalationAttempts,
|
|
232
|
+
executors,
|
|
212
233
|
definitions: [...digests.values()].sort((a, b) => a.name.localeCompare(b.name) || a.sha256.localeCompare(b.sha256)),
|
|
213
234
|
approvals: {
|
|
214
235
|
bySource,
|
package/src/ledger.ts
CHANGED
|
@@ -23,6 +23,7 @@ import { appendFile, mkdir, readFile } from "node:fs/promises";
|
|
|
23
23
|
import { withFileLock } from "./file-lock.ts";
|
|
24
24
|
import { dirname } from "node:path";
|
|
25
25
|
import type { Capability, ResolveResult } from "./resolve.ts";
|
|
26
|
+
import type { ExecutorKind } from "./executor.ts";
|
|
26
27
|
import type { DefinitionDigest } from "./definitions.ts";
|
|
27
28
|
import { DELEGATE_SUBJECT } from "./approval.ts";
|
|
28
29
|
import type { ApprovalScope, ApprovalSource } from "./approval.ts";
|
|
@@ -107,6 +108,20 @@ export interface GrantRecord {
|
|
|
107
108
|
* it. Absent for a `tools:`-style delegation, which has no definition.
|
|
108
109
|
*/
|
|
109
110
|
definitionDigest?: DefinitionDigest;
|
|
111
|
+
/**
|
|
112
|
+
* WHERE this child ran — ADR-0031.
|
|
113
|
+
*
|
|
114
|
+
* **Required rather than optional**, which is unusual in this record and deliberate. Before ADR-0031 the
|
|
115
|
+
* executor was a variable an operator set, so "which one ran?" was answerable from configuration after the
|
|
116
|
+
* fact. It is now decided by a **runtime probe** at session start, so nothing outside the record preserves
|
|
117
|
+
* the answer — and the two paths do not produce the same argv, because the herdr plan withholds `--print`
|
|
118
|
+
* (`delegationContext.interactive`). A trail that cannot say where a child ran cannot be read back
|
|
119
|
+
* reliably, and reading it back is the only reason it exists.
|
|
120
|
+
*
|
|
121
|
+
* Written on refusals too, including the tripwire's: the honest value there is the executor the session
|
|
122
|
+
* *would* have used, because a refused spawn has no executor of its own.
|
|
123
|
+
*/
|
|
124
|
+
executor: ExecutorKind;
|
|
110
125
|
}
|
|
111
126
|
|
|
112
127
|
export interface LedgerOptions {
|
|
@@ -137,6 +152,8 @@ export function buildRecord(args: {
|
|
|
137
152
|
humanDenied?: boolean;
|
|
138
153
|
gateOutcome?: PromptOutcomeKind;
|
|
139
154
|
definitionDigest?: DefinitionDigest;
|
|
155
|
+
/** Where the child ran (ADR-0031). Required: the probe's answer survives nowhere else. */
|
|
156
|
+
executor: ExecutorKind;
|
|
140
157
|
now: Date;
|
|
141
158
|
}): GrantRecord {
|
|
142
159
|
// R-46: the scalar is a SUMMARY, emitted only when it cannot mislead. `buildRecord` derives it rather
|
|
@@ -151,6 +168,7 @@ export function buildRecord(args: {
|
|
|
151
168
|
childId: args.childId,
|
|
152
169
|
depth: args.depth,
|
|
153
170
|
agentType: args.agentType,
|
|
171
|
+
executor: args.executor,
|
|
154
172
|
requested: args.requested,
|
|
155
173
|
parentGrant: args.parentGrant,
|
|
156
174
|
effective: args.result.effective,
|