gentle-pi 2.3.0 → 2.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (153) hide show
  1. package/README.md +195 -11
  2. package/assets/agents/gentle-ai-worker.md +9 -0
  3. package/assets/agents/sdd-explore.md +1 -0
  4. package/assets/orchestrator-delegation.md +21 -10
  5. package/assets/orchestrator.md +8 -12
  6. package/contracts/review-provider-contract-mirror/provider-contract.lock.json +8 -7
  7. package/contracts/review-provider-contract-mirror/{v1.1.0 → v1.2.0}/bundle/README.md +10 -0
  8. package/contracts/review-provider-contract-mirror/v1.2.0/bundle/manifest.json +74 -0
  9. package/contracts/review-provider-contract-mirror/v1.2.0/bundle/orchestration/pi.md +53 -0
  10. package/contracts/review-provider-contract-mirror/v1.2.0/bundle/schemas/targeted-validator.schema.json +1 -0
  11. package/contracts/review-provider-contract-mirror/{v1.1.0 → v1.2.0}/generated/provider-capabilities.baseline.json +9 -2
  12. package/contracts/review-provider-contract-mirror/{v1.1.0 → v1.2.0}/generated/provider-roles.baseline.json +2 -2
  13. package/docs/delegated-verification.md +25 -0
  14. package/docs/review-integration.md +1 -1
  15. package/docs/telemetry.md +38 -0
  16. package/extensions/ask-user-choice.ts +26 -20
  17. package/extensions/codegraph-tools.ts +94 -5
  18. package/extensions/gentle-agents.ts +588 -0
  19. package/extensions/gentle-ai.ts +1421 -143
  20. package/extensions/gentle-shell.ts +547 -0
  21. package/extensions/gentle-todo.ts +199 -0
  22. package/extensions/quiet-tools.ts +1 -1
  23. package/lib/agent-home.ts +8 -0
  24. package/lib/agents-config.ts +318 -0
  25. package/lib/agents-history.ts +80 -0
  26. package/lib/agents-protocol.ts +429 -0
  27. package/lib/agents-runner.ts +490 -0
  28. package/lib/agents-transcript.ts +87 -0
  29. package/lib/agents-view.ts +557 -0
  30. package/lib/agents-widget.ts +222 -0
  31. package/lib/gentle-ai-renderer.ts +142 -26
  32. package/lib/native-choice-list.ts +194 -0
  33. package/lib/native-fullscreen-interaction.ts +47 -0
  34. package/lib/native-pointer-region.ts +164 -0
  35. package/lib/native-review-cli.ts +103 -12
  36. package/lib/provider-contract-bundle.ts +88 -6
  37. package/lib/review-candidate-view-owner.ts +177 -0
  38. package/lib/review-candidate-view.ts +127 -35
  39. package/lib/review-consent-ui.ts +65 -0
  40. package/lib/review-host-relay.ts +146 -60
  41. package/lib/review-integration-v2.ts +92 -13
  42. package/lib/review-last-event-controller.ts +1 -0
  43. package/lib/review-relay-contract.ts +11 -0
  44. package/lib/review-repository.ts +2 -2
  45. package/lib/review-risk-assessment.ts +339 -0
  46. package/lib/review-session-standing-permission-ipc.ts +309 -0
  47. package/lib/review-session-standing-permission.ts +219 -0
  48. package/lib/sdd-preflight.ts +2 -2
  49. package/lib/shell-bar.ts +138 -0
  50. package/lib/shell-card.ts +136 -0
  51. package/lib/shell-changes-view.ts +205 -0
  52. package/lib/shell-changes.ts +210 -0
  53. package/lib/shell-gauge.ts +40 -0
  54. package/lib/shell-prompt.ts +119 -0
  55. package/lib/shell-todo.ts +280 -0
  56. package/lib/shell-usage-view.ts +76 -0
  57. package/lib/shell-usage.ts +246 -0
  58. package/lib/telemetry-trigger.ts +151 -0
  59. package/package.json +4 -4
  60. package/runtime/native-review-cli.mjs +102 -11
  61. package/runtime/review-integration-v2.mjs +92 -13
  62. package/runtime/review-relay-contract.mjs +11 -0
  63. package/runtime/review-risk-assessment.mjs +340 -0
  64. package/runtime/telemetry-trigger.mjs +152 -0
  65. package/scripts/build-runtime-modules.mjs +2 -0
  66. package/scripts/gentle-ai-installer.mjs +10 -10
  67. package/scripts/test-packed-runner.mjs +22 -0
  68. package/scripts/verify-package-files.mjs +18 -13
  69. package/skills/_shared/review-ledger-contract.md +9 -1
  70. package/skills/issue-creation/SKILL.md +53 -93
  71. package/tests/agents-config.test.ts +143 -0
  72. package/tests/agents-fake-child.ts +52 -0
  73. package/tests/agents-history.test.ts +54 -0
  74. package/tests/agents-protocol.test.ts +153 -0
  75. package/tests/agents-runner-process.test.ts +111 -0
  76. package/tests/agents-runner.test.ts +402 -0
  77. package/tests/agents-transcript.test.ts +30 -0
  78. package/tests/agents-view.test.ts +274 -0
  79. package/tests/agents-widget.test.ts +111 -0
  80. package/tests/ask-user-choice.test.ts +157 -3
  81. package/tests/codegraph-tools.test.ts +110 -1
  82. package/tests/devbinary/native-review-parity.devtest.ts +108 -0
  83. package/tests/fixtures/agents-process-child.mjs +23 -0
  84. package/tests/fixtures/provider-contract-bundle/v1.2.0/README.md +22 -0
  85. package/{contracts/review-provider-contract-mirror/v1.1.0/bundle → tests/fixtures/provider-contract-bundle/v1.2.0}/manifest.json +11 -2
  86. package/tests/fixtures/provider-contract-bundle/v1.2.0/orchestration/pi.md +97 -0
  87. package/tests/fixtures/provider-contract-bundle/v1.2.0/schemas/lens.schema.json +16 -0
  88. package/tests/fixtures/provider-contract-bundle/v1.2.0/schemas/refuter.schema.json +1 -0
  89. package/tests/fixtures/provider-contract-bundle/v1.2.0/vectors/lens.json +1 -0
  90. package/tests/fixtures/provider-contract-bundle/v1.2.0/vectors/refuter.json +1 -0
  91. package/tests/fixtures/provider-contract-bundle/v1.2.0/vectors/targeted-validator.json +1 -0
  92. package/tests/gentle-agents.test.ts +741 -0
  93. package/tests/gentle-ai-binary.test.ts +1 -1
  94. package/tests/gentle-ai-installer.test.ts +47 -47
  95. package/tests/gentle-ai-renderer.test.ts +65 -0
  96. package/tests/gentle-ai.test.ts +31 -14
  97. package/tests/gentle-card-text.ts +35 -0
  98. package/tests/gentle-shell.test.ts +527 -0
  99. package/tests/gentle-todo.test.ts +182 -0
  100. package/tests/issue-creation-skill.test.ts +103 -0
  101. package/tests/native-choice-list.test.ts +202 -0
  102. package/tests/native-fullscreen-interaction.test.ts +125 -0
  103. package/tests/native-pointer-region.test.ts +245 -0
  104. package/tests/native-review-capability-contract.test.ts +33 -1
  105. package/tests/native-review-cli.test.ts +40 -0
  106. package/tests/native-review-consent.test.ts +91 -0
  107. package/tests/native-review-parity-runtime.test.ts +8 -2
  108. package/tests/native-review-parity.test.ts +29 -22
  109. package/tests/orchestrator-budget.test.ts +71 -2
  110. package/tests/orchestrator-rdd-ownership.test.ts +10 -1
  111. package/tests/package-manifest.test.ts +134 -9
  112. package/tests/provider-contract-bundle.test.ts +76 -0
  113. package/tests/provider-contract-mirror.test.ts +19 -0
  114. package/tests/quiet-tool-rendering.test.ts +96 -37
  115. package/tests/rdd-aware-verification-contract.test.ts +216 -0
  116. package/tests/rdd-status-line.test.ts +286 -0
  117. package/tests/review-agent-end-preflight.test.ts +408 -0
  118. package/tests/review-candidate-view.test.ts +452 -6
  119. package/tests/review-contract-prompt.test.ts +142 -0
  120. package/tests/review-controller-native-recovery.test.ts +29 -4
  121. package/tests/review-controller-native-routing.test.ts +321 -4
  122. package/tests/review-controller-workspace-root.test.ts +45 -2
  123. package/tests/review-controller.test.ts +26 -1
  124. package/tests/review-host-relay-routing.test.ts +229 -11
  125. package/tests/review-host-relay.test.ts +195 -7
  126. package/tests/review-integration-v2-forward.test.ts +47 -0
  127. package/tests/review-integration-v2.test.ts +112 -0
  128. package/tests/review-last-event-closure.test.ts +7 -2
  129. package/tests/review-ledger-contract.test.ts +1 -1
  130. package/tests/review-relay-contract.test.ts +26 -0
  131. package/tests/review-repository.test.ts +28 -1
  132. package/tests/review-risk-assessment.test.ts +626 -0
  133. package/tests/review-session-standing-permission-controller.test.ts +608 -0
  134. package/tests/review-session-standing-permission-ipc.test.ts +233 -0
  135. package/tests/review-session-standing-permission-runtime.test.ts +212 -0
  136. package/tests/review-session-standing-permission.test.ts +126 -0
  137. package/tests/runtime-harness.mjs +1 -0
  138. package/tests/shell-bar.test.ts +176 -0
  139. package/tests/shell-card.test.ts +118 -0
  140. package/tests/shell-changes-view.test.ts +146 -0
  141. package/tests/shell-changes.test.ts +182 -0
  142. package/tests/shell-prompt.test.ts +118 -0
  143. package/tests/shell-todo.test.ts +170 -0
  144. package/tests/shell-usage-view.test.ts +62 -0
  145. package/tests/shell-usage.test.ts +197 -0
  146. package/tests/telemetry-trigger.test.ts +349 -0
  147. package/tests/writer-edit-surface-scope.test.ts +153 -17
  148. /package/contracts/review-provider-contract-mirror/{v1.1.0 → v1.2.0}/bundle/schemas/lens.schema.json +0 -0
  149. /package/contracts/review-provider-contract-mirror/{v1.1.0 → v1.2.0}/bundle/schemas/refuter.schema.json +0 -0
  150. /package/contracts/review-provider-contract-mirror/{v1.1.0 → v1.2.0}/bundle/vectors/lens.json +0 -0
  151. /package/contracts/review-provider-contract-mirror/{v1.1.0 → v1.2.0}/bundle/vectors/refuter.json +0 -0
  152. /package/contracts/review-provider-contract-mirror/{v1.1.0 → v1.2.0}/bundle/vectors/targeted-validator.json +0 -0
  153. /package/{contracts/review-provider-contract-mirror/v1.1.0/bundle → tests/fixtures/provider-contract-bundle/v1.2.0}/schemas/targeted-validator.schema.json +0 -0
@@ -0,0 +1,402 @@
1
+ import assert from "node:assert/strict";
2
+ import test from "node:test";
3
+ import { AGENT_MODE, type AgentDefinition } from "../lib/agents-config.ts";
4
+ import { TASK_STATUS, TaskStore } from "../lib/agents-protocol.ts";
5
+ import { AgentRunner, childArguments, JsonLines, piCommand, type RunnerDeps, type TaskRequest } from "../lib/agents-runner.ts";
6
+ import { fakeChild, type FakeChild } from "./agents-fake-child.ts";
7
+
8
+ // Gentle Agents runner: every subagent is a child `pi --mode rpc` process.
9
+ // The host only parses JSON lines, applies deltas to the store, answers
10
+ // dialogs, and enforces its inactivity watchdog. These tests drive a fake child.
11
+
12
+ const explorer: AgentDefinition = { name: "explore", description: "maps", filePath: "/a/explore.md", scope: "global", instructions: "You map things.", model: undefined, thinking: undefined, mode: undefined, tools: ["read", "grep"] };
13
+
14
+ function request(overrides: Partial<TaskRequest> = {}): TaskRequest {
15
+ return { agent: explorer, prompt: "Map the repo", label: undefined, context: undefined, mode: AGENT_MODE.TASK, cwd: "/repo", parentSessionId: "s1", model: { provider: "openai-codex", id: "gpt-5.6-terra" }, thinking: "high", sessionDir: "/sessions", resumeSessionPath: undefined, env: {}, ...overrides };
16
+ }
17
+
18
+ interface Harness {
19
+ store: TaskStore;
20
+ runner: AgentRunner;
21
+ children: FakeChild[];
22
+ timers: Array<{ fn: () => void; ms: number; cancelled: boolean }>;
23
+ asks: Array<{ taskId: string; method: string }>;
24
+ finishes: string[];
25
+ spawnOptions: Array<{ stdio?: string[] }>;
26
+ }
27
+
28
+ function harness(options: { maxConcurrency?: number; answer?: Record<string, unknown>; exitOnKill?: boolean } = {}): Harness {
29
+ const children: FakeChild[] = [];
30
+ const timers: Harness["timers"] = [];
31
+ const asks: Harness["asks"] = [];
32
+ const finishes: string[] = [];
33
+ const spawnOptions: Harness["spawnOptions"] = [];
34
+ let clock = 1000;
35
+ const deps: RunnerDeps = {
36
+ spawn: (_command, _args, launchOptions) => {
37
+ spawnOptions.push({ stdio: launchOptions.stdio });
38
+ const fake = fakeChild({ exitOnKill: options.exitOnKill });
39
+ children.push(fake);
40
+ return fake.child;
41
+ },
42
+ now: () => (clock += 1),
43
+ schedule: (fn, ms) => {
44
+ const timer = { fn, ms, cancelled: false };
45
+ timers.push(timer);
46
+ return () => {
47
+ timer.cancelled = true;
48
+ };
49
+ },
50
+ pi: { command: "pi", args: [] },
51
+ };
52
+ const store = new TaskStore();
53
+ const runner = new AgentRunner(store, { maxConcurrency: options.maxConcurrency ?? 2, stallTimeoutMs: 10_000 }, deps, {
54
+ askUser: async (taskId, ask) => {
55
+ asks.push({ taskId, method: ask.method });
56
+ return options.answer ?? { value: "yes" };
57
+ },
58
+ onFinish: (task) => finishes.push(task.id),
59
+ });
60
+ return { store, runner, children, timers, asks, finishes, spawnOptions };
61
+ }
62
+
63
+ const tick = () => new Promise((resolve) => setImmediate(resolve));
64
+
65
+ test("childArguments builds an rpc launch with model, thinking, tools, session dir, and instructions", () => {
66
+ const args = childArguments(request());
67
+ assert.deepEqual(args.slice(0, 2), ["--mode", "rpc"]);
68
+ assert.ok(args.includes("--session-dir") && args[args.indexOf("--session-dir") + 1] === "/sessions");
69
+ assert.equal(args[args.indexOf("--model") + 1], "openai-codex/gpt-5.6-terra:high");
70
+ assert.equal(args[args.indexOf("--tools") + 1], "read,grep");
71
+ assert.equal(args[args.indexOf("--append-system-prompt") + 1], "You map things.");
72
+ assert.ok(!args.includes("--session"));
73
+ const resumed = childArguments(request({ resumeSessionPath: "/sessions/old.jsonl", model: undefined, thinking: undefined, agent: { ...explorer, tools: [] } }));
74
+ assert.equal(resumed[resumed.indexOf("--session") + 1], "/sessions/old.jsonl");
75
+ assert.ok(!resumed.includes("--model") && !resumed.includes("--tools"));
76
+ });
77
+
78
+ test("piCommand reuses the running pi entry point and honors the override", () => {
79
+ assert.deepEqual(piCommand({ execPath: "/bin/node", argv: ["/bin/node", "/x/dist/cli.js"], env: {} }), { command: "/bin/node", args: ["/x/dist/cli.js"] });
80
+ assert.deepEqual(piCommand({ execPath: "/bin/node", argv: ["/bin/node", "/x/other.js"], env: {} }), { command: "pi", args: [] });
81
+ assert.deepEqual(piCommand({ execPath: "/bin/node", argv: [], env: { GENTLE_PI_AGENTS_PI: "/opt/pi --flag" } }), { command: "/opt/pi", args: ["--flag"] });
82
+ });
83
+
84
+ test("JsonLines splits on LF only, tolerates CRLF, and skips lines that are not JSON", () => {
85
+ const seen: unknown[] = [];
86
+ const lines = new JsonLines((value) => seen.push(value));
87
+ lines.push('{"a":1}\r\n{"b":"x
y"}\nnot json\n{"c":');
88
+ lines.push("3}\n");
89
+ assert.deepEqual(seen, [{ a: 1 }, { b: "x
y" }, { c: 3 }]);
90
+ });
91
+
92
+ test("AgentRunner runs a task end to end: prompt, deltas into the store, completion with the last answer", async () => {
93
+ const { store, runner, children } = harness();
94
+ const task = runner.run(request());
95
+ assert.equal(task.status, TASK_STATUS.QUEUED);
96
+ await tick();
97
+ assert.equal(store.get(task.id)?.status, TASK_STATUS.RUNNING);
98
+ const [child] = children;
99
+ await tick();
100
+ assert.deepEqual(children[0].written.map((command) => command.type), ["get_state", "prompt"]);
101
+ assert.equal(children[0].written[1].message, "Map the repo");
102
+ child.emit({ type: "tool_execution_start", toolCallId: "c1", toolName: "grep", args: {} });
103
+ child.emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "Found it" } });
104
+ child.emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "Found it" }] }] });
105
+ await tick();
106
+ assert.equal(store.get(task.id)?.status, TASK_STATUS.RUNNING, "agent_end retains the latest answer while queued follow-up may still run");
107
+ assert.equal(children[0].killed.length, 0, "the child remains available until Pi reports settlement");
108
+ child.emit({ type: "agent_settled" });
109
+ await tick();
110
+ const finished = store.get(task.id);
111
+ assert.equal(finished?.status, TASK_STATUS.COMPLETED);
112
+ assert.equal(finished?.result, "Found it");
113
+ assert.equal(finished?.toolCalls, 1);
114
+ assert.equal(finished?.sessionPath, "/sessions/child.jsonl");
115
+ assert.equal(finished?.label, "Map the repo");
116
+ assert.ok(children[0].killed.length > 0, "the child is stopped once the answer is in");
117
+ assert.equal(store.thread(task.id).items.length, 2);
118
+ assert.equal((await runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED);
119
+ });
120
+
121
+ test("AgentRunner waits for child exit after settlement before releasing its queue slot or finishing twice", async () => {
122
+ const { store, runner, children, finishes } = harness({ maxConcurrency: 1, exitOnKill: false });
123
+ const first = runner.run(request());
124
+ const second = runner.run(request({ prompt: "Second" }));
125
+ await tick();
126
+ assert.equal(children.length, 1);
127
+ children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "Answer" }] }] });
128
+ await tick();
129
+ assert.equal(store.get(first.id)?.status, TASK_STATUS.RUNNING, "agent_end ends one run, not the session");
130
+ assert.equal(store.get(first.id)?.result, "Answer", "agent_end retains the final run output");
131
+ assert.equal(store.get(second.id)?.status, TASK_STATUS.QUEUED, "the slot stays occupied until settlement");
132
+ assert.deepEqual(finishes, []);
133
+ children[0].emit({ type: "agent_settled" });
134
+ await tick();
135
+ assert.equal(store.get(first.id)?.status, TASK_STATUS.RUNNING, "terminal RPC state does not release a live process");
136
+ assert.equal(store.get(second.id)?.status, TASK_STATUS.QUEUED);
137
+ children[0].exit(0);
138
+ await tick();
139
+ await tick();
140
+ assert.equal(store.get(first.id)?.status, TASK_STATUS.COMPLETED);
141
+ assert.deepEqual(finishes, [first.id], "settlement delivers completion once");
142
+ assert.equal(children.length, 2, "child exit releases the queue slot");
143
+ children[0].emit({ type: "agent_settled" });
144
+ await tick();
145
+ assert.deepEqual(finishes, [first.id], "duplicate terminal events do not finalize twice");
146
+ });
147
+
148
+ test("AgentRunner queues beyond max concurrency and starts the next task when one finishes", async () => {
149
+ const { store, runner, children } = harness({ maxConcurrency: 1 });
150
+ const first = runner.run(request());
151
+ const second = runner.run(request({ prompt: "Second" }));
152
+ await tick();
153
+ assert.equal(children.length, 1);
154
+ assert.equal(store.get(second.id)?.status, TASK_STATUS.QUEUED);
155
+ children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "First complete." }], stopReason: "stop" }] });
156
+ await tick();
157
+ assert.equal(store.get(first.id)?.status, TASK_STATUS.RUNNING, "the concurrency slot remains held through a queued follow-up");
158
+ children[0].emit({ type: "agent_settled" });
159
+ await tick();
160
+ await tick();
161
+ assert.equal(store.get(first.id)?.status, TASK_STATUS.COMPLETED);
162
+ assert.equal(children.length, 2);
163
+ assert.equal(store.get(second.id)?.status, TASK_STATUS.RUNNING);
164
+ });
165
+
166
+ test("AgentRunner classifies terminal assistant outcomes only after settlement", async () => {
167
+ const scenarios = [
168
+ { name: "error", messages: [{ role: "assistant", content: [], stopReason: "error", errorMessage: "WebSocket error: secret=never-copy" }], status: TASK_STATUS.FAILED, error: /assistant reported an error/ },
169
+ { name: "aborted", messages: [{ role: "assistant", content: [], stopReason: "aborted" }], status: TASK_STATUS.FAILED, error: /assistant aborted/ },
170
+ { name: "empty", messages: [{ role: "assistant", content: [], stopReason: "stop" }], status: TASK_STATUS.FAILED, error: /no final report/ },
171
+ { name: "success", messages: [{ role: "assistant", content: [{ type: "text", text: "final report" }], stopReason: "stop" }], status: TASK_STATUS.COMPLETED, error: null },
172
+ ] as const;
173
+ for (const scenario of scenarios) {
174
+ const { store, runner, children } = harness();
175
+ const task = runner.run(request());
176
+ await tick();
177
+ children[0].emit({ type: "agent_end", messages: scenario.messages });
178
+ assert.equal(store.get(task.id)?.status, TASK_STATUS.RUNNING, `${scenario.name} stays running until settlement`);
179
+ children[0].emit({ type: "agent_settled" });
180
+ const finished = await runner.waitFor(task.id);
181
+ assert.equal(finished.status, scenario.status, scenario.name);
182
+ if (scenario.error) assert.match(finished.error ?? "", scenario.error);
183
+ else assert.equal(finished.result, "final report");
184
+ }
185
+ });
186
+
187
+ test("AgentRunner clears an earlier answer after a later error, but permits a successful retry before settlement", async () => {
188
+ const first = harness();
189
+ const failedTask = first.runner.run(request());
190
+ await tick();
191
+ first.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "stale success" }], stopReason: "stop" }] });
192
+ first.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [], stopReason: "error", errorMessage: "provider detail must not persist" }] });
193
+ first.children[0].emit({ type: "agent_settled" });
194
+ const failed = await first.runner.waitFor(failedTask.id);
195
+ assert.equal(failed.status, TASK_STATUS.FAILED);
196
+ assert.equal(failed.result, null, "a later error must not report stale successful text");
197
+
198
+ const retry = harness();
199
+ const retryTask = retry.runner.run(request());
200
+ await tick();
201
+ retry.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [], stopReason: "error" }] });
202
+ retry.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "retry report" }], stopReason: "stop" }] });
203
+ retry.children[0].emit({ type: "agent_settled" });
204
+ const recovered = await retry.runner.waitFor(retryTask.id);
205
+ assert.equal(recovered.status, TASK_STATUS.COMPLETED);
206
+ assert.equal(recovered.result, "retry report");
207
+ });
208
+
209
+ test("AgentRunner fails if the child exits after agent_end but before agent_settled", async () => {
210
+ const { store, runner, children } = harness();
211
+ const task = runner.run(request());
212
+ await tick();
213
+ children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "partial answer" }] }] });
214
+ await tick();
215
+ children[0].exit(0);
216
+ await tick();
217
+ assert.equal(store.get(task.id)?.status, TASK_STATUS.FAILED);
218
+ assert.match(store.get(task.id)?.error ?? "", /before agent_settled/);
219
+ assert.equal(store.get(task.id)?.result, "partial answer", "the final observed answer remains available for diagnostics");
220
+ });
221
+
222
+ test("AgentRunner reserves a parent-owned fourth stdio fd only for package-child authorization", async () => {
223
+ const { runner, children, spawnOptions } = harness();
224
+ const task = runner.run(request({ authorizeParentStandingReviewPermission: () => true }));
225
+ await tick();
226
+ assert.deepEqual(spawnOptions[0]?.stdio, ["pipe", "pipe", "pipe", "pipe"]);
227
+ assert.equal((spawnOptions[0] as { stdio?: string[] } | undefined)?.stdio?.length, 4);
228
+ children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "channel checked" }], stopReason: "stop" }] });
229
+ children[0].emit({ type: "agent_settled" });
230
+ assert.equal((await runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED);
231
+ });
232
+
233
+ test("AgentRunner answers dialogs through askUser in task mode and cancels them in background mode", async () => {
234
+ const { store, runner, children, asks } = harness({ answer: { confirmed: true } });
235
+ const task = runner.run(request());
236
+ const background = runner.run(request({ mode: AGENT_MODE.BACKGROUND }));
237
+ await tick();
238
+ children[0].emit({ type: "extension_ui_request", id: "u1", method: "confirm", title: "Delete?" });
239
+ children[1].emit({ type: "extension_ui_request", id: "u2", method: "select", title: "Pick", options: ["a"] });
240
+ children[1].emit({ type: "extension_ui_request", id: "u3", method: "notify", message: "hi" });
241
+ await tick();
242
+ await tick();
243
+ assert.deepEqual(asks, [{ taskId: task.id, method: "confirm" }]);
244
+ assert.deepEqual(children[0].written.at(-1), { type: "extension_ui_response", id: "u1", confirmed: true });
245
+ assert.deepEqual(children[1].written.at(-1), { type: "extension_ui_response", id: "u2", cancelled: true });
246
+ assert.equal(store.get(background.id)?.status, TASK_STATUS.RUNNING);
247
+ assert.equal(store.get(task.id)?.status, TASK_STATUS.RUNNING, "answered questions do not leave the task waiting");
248
+ });
249
+
250
+ test("AgentRunner cancels and fails when the child exits early", async () => {
251
+ const { store, runner, children } = harness({ maxConcurrency: 3 });
252
+ const cancelled = runner.run(request());
253
+ const crashed = runner.run(request());
254
+ await tick();
255
+ runner.cancel(cancelled.id);
256
+ await tick();
257
+ assert.equal(store.get(cancelled.id)?.status, TASK_STATUS.CANCELLED);
258
+ assert.ok(children[0].written.some((command) => command.type === "abort"));
259
+ children[1].exit(1);
260
+ await tick();
261
+ assert.equal(store.get(crashed.id)?.status, TASK_STATUS.FAILED);
262
+ assert.match(store.get(crashed.id)?.error ?? "", /exited with code 1/);
263
+ assert.ok(runner.steer(cancelled.id, "x") === false, "a finished task cannot be steered");
264
+ });
265
+
266
+ test("AgentRunner has no total-duration watchdog but keeps active work alive and times out true silence", async () => {
267
+ const { store, runner, children, timers } = harness();
268
+ const task = runner.run(request({ mode: AGENT_MODE.BACKGROUND }));
269
+ await tick();
270
+ assert.deepEqual(timers.filter((timer) => !timer.cancelled).map((timer) => timer.ms), [10_000], "only the inactivity watchdog is scheduled");
271
+ const initialStall = timers[0];
272
+ children[0].emit({ type: "response", id: "r1", success: true });
273
+ await tick();
274
+ assert.equal(initialStall.cancelled, true, "every child RPC event, including a response, re-arms the inactivity watchdog");
275
+ children[0].emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "still working" } });
276
+ await tick();
277
+ assert.equal(store.get(task.id)?.status, TASK_STATUS.RUNNING, "ongoing RPC activity keeps a long-running task active");
278
+ const stall = timers.filter((timer) => timer.ms === 10_000 && !timer.cancelled).at(-1);
279
+ assert.ok(stall);
280
+ stall.fn();
281
+ await tick();
282
+ assert.equal(store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
283
+ assert.match(store.get(task.id)?.error ?? "", /stalled/);
284
+ });
285
+
286
+ test("AgentRunner.cancelAll stops every queued and running task", async () => {
287
+ const { store, runner, children } = harness({ maxConcurrency: 1 });
288
+ const running = runner.run(request());
289
+ const queued = runner.run(request());
290
+ await tick();
291
+ assert.equal(runner.cancelAll(), 2);
292
+ await tick();
293
+ assert.equal(store.get(running.id)?.status, TASK_STATUS.CANCELLED);
294
+ assert.equal(store.get(queued.id)?.status, TASK_STATUS.CANCELLED);
295
+ assert.deepEqual(children[0].killed, ["SIGTERM"]);
296
+ assert.equal(children.length, 1, "nothing else starts after cancelAll");
297
+ });
298
+
299
+ test("AgentRunner fails only the task when the child cannot start, and the queue moves on", async () => {
300
+ const { store, runner, children, timers } = harness({ maxConcurrency: 1 });
301
+ const broken = runner.run(request());
302
+ const next = runner.run(request({ prompt: "After" }));
303
+ await tick();
304
+ children[0].fail("spawn pi ENOENT");
305
+ await tick();
306
+ assert.equal(store.get(broken.id)?.status, TASK_STATUS.FAILED);
307
+ assert.match(store.get(broken.id)?.error ?? "", /could not start pi: spawn pi ENOENT/);
308
+ assert.equal((await runner.waitFor(broken.id)).status, TASK_STATUS.FAILED, "waiters settle");
309
+ await tick();
310
+ assert.equal(children.length, 2, "the next queued task starts");
311
+ assert.equal(store.get(next.id)?.status, TASK_STATUS.RUNNING);
312
+ assert.ok(timers.filter((timer) => timer.ms === 10_000).some((timer) => timer.cancelled), "the failed task's inactivity watchdog is cancelled");
313
+ });
314
+
315
+ test("AgentRunner turns a synchronous spawn exception into a failed task", async () => {
316
+ const store = new TaskStore();
317
+ const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 1000 }, {
318
+ spawn: () => {
319
+ throw new Error("ENOENT: pi not found");
320
+ },
321
+ now: () => 1,
322
+ schedule: () => () => {},
323
+ pi: { command: "missing-pi", args: [] },
324
+ }, { askUser: async () => ({ cancelled: true }) });
325
+ const task = runner.run(request());
326
+ const finished = await runner.waitFor(task.id);
327
+ assert.equal(finished.status, TASK_STATUS.FAILED);
328
+ assert.match(finished.error ?? "", /could not start pi: ENOENT/);
329
+ });
330
+
331
+ for (const lateEvents of [false, true]) test(`AgentRunner releases quarantined capacity only on proven exit (late events: ${lateEvents})`, async () => {
332
+ const store = new TaskStore();
333
+ const timers: Array<{ fn: () => void; ms: number; cancelled: boolean }> = [];
334
+ let now = 0;
335
+ let groupGone = false;
336
+ let launches = 0;
337
+ let asks = 0;
338
+ const finishes: string[] = [];
339
+ let resolveAnswer!: (answer: { value: string }) => void;
340
+ const answer = new Promise<{ value: string }>((resolve) => { resolveAnswer = resolve; });
341
+ const child = fakeChild({ exitOnKill: false, pid: 71 });
342
+ const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 10_000 }, {
343
+ spawn: () => { launches += 1; return launches === 1 ? child.child : fakeChild().child; },
344
+ now: () => now,
345
+ schedule: (fn, ms) => {
346
+ const timer = { fn, ms, cancelled: false };
347
+ timers.push(timer);
348
+ return () => { timer.cancelled = true; };
349
+ },
350
+ pi: { command: "pi", args: [] },
351
+ process: { platform: "linux", kill: (_pid, signal) => {
352
+ if (signal === 0) throw Object.assign(new Error("group probe"), { code: groupGone ? "ESRCH" : "EPERM" });
353
+ } },
354
+ }, { askUser: async () => { asks += 1; return answer; }, onFinish: (task) => finishes.push(task.id) });
355
+ const first = runner.run(request());
356
+ const second = runner.run(request({ prompt: "queued" }));
357
+ await tick();
358
+ const waiter = runner.waitFor(first.id);
359
+ if (lateEvents) child.emit({ type: "extension_ui_request", id: "early", method: "input", title: "Pending?" });
360
+ runner.cancel(first.id);
361
+ const grace = timers.find((timer) => timer.ms === 250);
362
+ assert.ok(grace);
363
+ grace.fn();
364
+ now = 2_000;
365
+ const check = timers.filter((timer) => timer.ms === 25).at(-1);
366
+ assert.ok(check);
367
+ check.fn();
368
+ await tick();
369
+ assert.equal(store.get(first.id)?.status, TASK_STATUS.FAILED);
370
+ assert.equal((await waiter).status, TASK_STATUS.FAILED);
371
+ assert.match(store.get(first.id)?.error ?? "", /cleanup unconfirmed/);
372
+ assert.equal(store.get(second.id)?.status, TASK_STATUS.QUEUED, "the unconfirmed group retains its capacity");
373
+ assert.equal(timers.filter((timer) => timer.ms === 25 && !timer.cancelled).length, 0, "confirmation polling stops at its deadline");
374
+ const finished = structuredClone(store.get(first.id));
375
+ if (lateEvents) {
376
+ const thread = structuredClone(store.thread(first.id));
377
+ const timerCount = timers.length;
378
+ const writes = child.written.length;
379
+ resolveAnswer({ value: "too late" });
380
+ await tick();
381
+ child.emit({ type: "extension_ui_request", id: "late", method: "input", title: "Reopen?" });
382
+ child.emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "late result" }] }] });
383
+ child.emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "late" } });
384
+ child.emit({ type: "agent_settled" });
385
+ await tick();
386
+ assert.equal(asks, 1, "late dialogs must not reopen");
387
+ assert.equal(child.written.length, writes, "pending answers must not reach a terminal child");
388
+ assert.equal(timers.length, timerCount, "late activity must not rearm the stall watchdog");
389
+ assert.deepEqual(store.get(first.id), finished);
390
+ assert.deepEqual(store.thread(first.id), thread);
391
+ assert.equal(launches, 1, "late events are not process-exit proof");
392
+ }
393
+ groupGone = true;
394
+ child.exit(0);
395
+ await tick();
396
+ assert.equal(launches, 2, "proven late exit must pump queued work");
397
+ assert.equal(store.get(second.id)?.status, TASK_STATUS.RUNNING);
398
+ child.exit(0);
399
+ await tick();
400
+ assert.deepEqual(finishes, [first.id], "cleanup must not finish the quarantined task twice");
401
+ assert.deepEqual(store.get(first.id), finished);
402
+ });
@@ -0,0 +1,30 @@
1
+ import assert from "node:assert/strict";
2
+ import test from "node:test";
3
+ import { sessionToMarkdown } from "../lib/agents-transcript.ts";
4
+
5
+ // Gentle Agents transcript: session JSONL in, readable markdown out.
6
+
7
+ const session = [
8
+ JSON.stringify({ type: "session", version: 3, id: "x" }),
9
+ JSON.stringify({ type: "model_change", provider: "openai-codex", modelId: "gpt-5.6-terra" }),
10
+ JSON.stringify({ type: "message", message: { role: "user", content: [{ type: "text", text: "Map lib/." }] } }),
11
+ JSON.stringify({ type: "message", message: { role: "assistant", content: [{ type: "thinking", thinking: "secret" }, { type: "toolCall", id: "c1", name: "bash", arguments: { command: "ls lib" } }] } }),
12
+ JSON.stringify({ type: "message", message: { role: "toolResult", toolCallId: "c1", toolName: "bash", isError: false, content: [{ type: "text", text: Array.from({ length: 45 }, (_, index) => `f${index}.ts`).join("\n") }] } }),
13
+ JSON.stringify({ type: "message", message: { role: "toolResult", toolCallId: "c2", toolName: "grep", isError: true, content: [{ type: "text", text: "\x1b[31mno matches\x1b[0m" }] } }),
14
+ "not json",
15
+ JSON.stringify({ type: "message", message: { role: "assistant", content: [{ type: "text", text: "Three files.\n" }] } }),
16
+ ].join("\n");
17
+
18
+ test("sessionToMarkdown writes user, tool calls with results, and assistant text, dropping thinking and noise", () => {
19
+ const markdown = sessionToMarkdown(session, { title: "explore · map lib", maxOutputLines: 40 });
20
+ assert.match(markdown, /^# explore · map lib\n\n## User\n\nMap lib\/\.\n\n### ▸ bash ls lib\n\n<!-- bash -->\n```\n… 5 earlier lines omitted\nf5\.ts\n/);
21
+ assert.match(markdown, /<!-- grep error -->\n```\nno matches\n```/);
22
+ assert.match(markdown, /## Assistant\n\nThree files\.\n$/);
23
+ assert.doesNotMatch(markdown, /secret/);
24
+ assert.equal(sessionToMarkdown(""), "# Subagent transcript\n");
25
+ });
26
+
27
+ test("sessionToMarkdown widens the fence when the output itself contains backticks", () => {
28
+ const line = JSON.stringify({ type: "message", message: { role: "toolResult", toolName: "read", content: [{ type: "text", text: "```ts\nlet a = 1\n```" }] } });
29
+ assert.match(sessionToMarkdown(line), /````\n```ts\nlet a = 1\n```\n````/);
30
+ });