gentle-pi 2.6.4 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (128) hide show
  1. package/README.md +25 -7
  2. package/assets/agents/gentle-ai-worker.md +5 -1
  3. package/assets/agents/sdd-apply.md +9 -7
  4. package/assets/agents/sdd-archive.md +42 -23
  5. package/assets/agents/sdd-proposal.md +2 -2
  6. package/assets/agents/sdd-remediate.md +4 -4
  7. package/assets/agents/sdd-research.md +20 -48
  8. package/assets/agents/sdd-tasks.md +5 -5
  9. package/assets/agents/sdd-verify.md +6 -28
  10. package/assets/chains/sdd-full.chain.md +4 -22
  11. package/assets/chains/sdd-verify.chain.md +3 -12
  12. package/assets/orchestrator-delegation.md +33 -3
  13. package/assets/orchestrator-memory.md +20 -7
  14. package/assets/orchestrator.md +5 -3
  15. package/assets/sdd-orchestrator-workflow.md +25 -58
  16. package/assets/support/sdd-status-contract.md +9 -12
  17. package/docs/gentle-shell.md +41 -19
  18. package/docs/readme-reference.md +175 -36
  19. package/extensions/codegraph-tools.ts +2 -0
  20. package/extensions/gentle-agents.ts +289 -361
  21. package/extensions/gentle-ai.ts +588 -117
  22. package/extensions/gentle-shell.ts +123 -97
  23. package/extensions/pi-pretty.ts +63 -14
  24. package/extensions/quiet-tools.ts +1 -2
  25. package/extensions/startup-banner.ts +10 -9
  26. package/lib/agent-home.ts +8 -0
  27. package/lib/agent-profile-pin.ts +336 -0
  28. package/lib/agent-profiles.ts +28 -8
  29. package/lib/agents-config.ts +24 -2
  30. package/lib/agents-history.ts +3 -97
  31. package/lib/agents-keys.ts +27 -0
  32. package/lib/agents-protocol.ts +2 -15
  33. package/lib/agents-runner.ts +74 -113
  34. package/lib/agents-session-transport.ts +691 -0
  35. package/lib/command-palette-catalog.ts +87 -0
  36. package/lib/command-palette.ts +346 -0
  37. package/lib/native-choice-list.ts +5 -0
  38. package/lib/native-review-cli.ts +19 -97
  39. package/lib/review-publication-gate.ts +11 -1
  40. package/lib/review-repository.ts +1 -1
  41. package/lib/review-snapshot.ts +1 -0
  42. package/lib/review-transaction.ts +4 -2
  43. package/lib/sdd-preflight.ts +2 -1
  44. package/lib/sdd-research-capabilities.ts +18 -152
  45. package/lib/sdd-status.ts +7 -779
  46. package/lib/session-change-capture.ts +88 -0
  47. package/lib/session-changes.ts +147 -0
  48. package/lib/shell-bar.ts +24 -10
  49. package/lib/shell-card.ts +8 -12
  50. package/lib/shell-changes-view.ts +2 -1
  51. package/lib/shell-changes.ts +5 -2
  52. package/lib/shell-prompt.ts +25 -8
  53. package/lib/shell-sidebar-banner.ts +2 -2
  54. package/lib/shell-sidebar-layout.ts +5 -2
  55. package/lib/windows-session-transport.ts +877 -0
  56. package/package.json +3 -3
  57. package/runtime/native-review-cli.mjs +18 -96
  58. package/runtime/windows-session-transport.ps1 +791 -0
  59. package/scripts/gentle-ai-installer.mjs +10 -10
  60. package/scripts/test-packed-runner.mjs +1668 -20
  61. package/scripts/verify-package-files.mjs +2 -3
  62. package/tests/agent-home.test.ts +52 -0
  63. package/tests/agent-profiles.test.ts +30 -1
  64. package/tests/agents-config.test.ts +44 -0
  65. package/tests/agents-history.test.ts +12 -24
  66. package/tests/agents-runner.test.ts +321 -58
  67. package/tests/agents-session-transport-process.test.ts +249 -0
  68. package/tests/agents-session-transport.test.ts +823 -0
  69. package/tests/artifact-language.test.ts +10 -7
  70. package/tests/command-palette.test.ts +378 -0
  71. package/tests/delegated-key-learnings-contract.test.ts +2 -2
  72. package/tests/fixtures/agents-session-transport-process.mjs +108 -0
  73. package/tests/fixtures/legacy/sdd-research-v2.5.0.md +54 -0
  74. package/tests/fixtures/windows-session-bootstrap.ps1 +129 -0
  75. package/tests/fixtures/windows-session-compile.ps1 +110 -0
  76. package/tests/gentle-agents.test.ts +883 -356
  77. package/tests/gentle-ai-binary.test.ts +1 -1
  78. package/tests/gentle-ai-installer.test.ts +47 -47
  79. package/tests/gentle-ai.test.ts +472 -4
  80. package/tests/gentle-shell.test.ts +338 -205
  81. package/tests/native-choice-list.test.ts +13 -0
  82. package/tests/native-review-capability-contract.test.ts +15 -1
  83. package/tests/native-review-cli.test.ts +0 -33
  84. package/tests/odd-routing-contract.test.ts +208 -0
  85. package/tests/orchestrator-budget.test.ts +17 -2
  86. package/tests/package-manifest.test.ts +119 -31
  87. package/tests/persona-single-channel.test.ts +3 -3
  88. package/tests/pi-pretty.test.ts +45 -0
  89. package/tests/profile-pin.test.ts +370 -0
  90. package/tests/quiet-tool-rendering.test.ts +32 -5
  91. package/tests/review-contract-prompt.test.ts +9 -0
  92. package/tests/review-controller.test.ts +0 -44
  93. package/tests/review-session-standing-permission-ipc.test.ts +427 -13
  94. package/tests/runtime-harness.mjs +4 -4
  95. package/tests/sdd-agent-tools.test.ts +15 -36
  96. package/tests/sdd-archive-replay.test.ts +82 -0
  97. package/tests/sdd-classical-continuation.test.ts +74 -0
  98. package/tests/sdd-execution-routing-contract.test.ts +18 -2
  99. package/tests/sdd-managed-runtime-settlement.test.ts +42 -330
  100. package/tests/sdd-native-managed-uptake.test.ts +11 -21
  101. package/tests/sdd-no-attempts-contract.test.ts +15 -0
  102. package/tests/sdd-odd-integration.test.ts +33 -0
  103. package/tests/sdd-optional-research.test.ts +124 -0
  104. package/tests/sdd-planning-routing-contract.test.ts +1 -1
  105. package/tests/sdd-preflight-rpc-input.test.ts +125 -0
  106. package/tests/sdd-preflight.test.ts +1 -1
  107. package/tests/sdd-research-capabilities.test.ts +20 -162
  108. package/tests/sdd-selection-transport.test.ts +180 -88
  109. package/tests/sdd-status.test.ts +5 -778
  110. package/tests/sdd-task-truth.test.ts +43 -0
  111. package/tests/session-change-capture.test.ts +86 -0
  112. package/tests/session-changes-shell.test.ts +38 -0
  113. package/tests/session-changes.test.ts +114 -0
  114. package/tests/shell-bar.test.ts +35 -0
  115. package/tests/shell-card.test.ts +8 -6
  116. package/tests/shell-changes.test.ts +8 -0
  117. package/tests/shell-prompt.test.ts +41 -7
  118. package/tests/shell-sidebar-banner.test.ts +4 -4
  119. package/tests/shell-sidebar-layout.test.ts +97 -13
  120. package/tests/startup-banner.test.ts +55 -2
  121. package/tests/windows-hidden-processes.test.ts +303 -0
  122. package/tests/windows-session-bootstrap.test.ts +1772 -0
  123. package/tests/windows-session-compile.test.ts +170 -0
  124. package/tests/windows-session-transport.test.ts +754 -0
  125. package/assets/agents/sdd-sync.md +0 -146
  126. package/lib/openspec-guardrails.ts +0 -99
  127. package/tests/native-sdd-attempt-authority.test.ts +0 -240
  128. package/tests/openspec-guardrails.test.ts +0 -71
@@ -2,8 +2,8 @@ import assert from "node:assert/strict";
2
2
  import test from "node:test";
3
3
  import { PassThrough } from "node:stream";
4
4
  import { AGENT_MODE, parseAgentsConfig, resolveAgentProfile, type AgentDefinition } from "../lib/agents-config.ts";
5
- import { TASK_STATUS, TaskStore, type RemediationTaskState, type TaskRecord } from "../lib/agents-protocol.ts";
6
- import { AgentRunner, childArguments, JsonLines, piCommand, abortReasonText, type RemediationPlan, type RemediationTerminalFacts, type RunnerDeps, type RunnerHooks, type TaskRequest } from "../lib/agents-runner.ts";
5
+ import { TASK_STATUS, TaskStore, type TaskRecord } from "../lib/agents-protocol.ts";
6
+ import { AgentRunner, childArguments, JsonLines, piCommand, abortReasonText, type ChildLike, type RunnerDeps, type RunnerHooks, type TaskRequest } from "../lib/agents-runner.ts";
7
7
  import { fakeChild, type FakeChild } from "./agents-fake-child.ts";
8
8
 
9
9
  // Gentle Agents runner: every subagent is a child `pi --mode rpc` process.
@@ -26,7 +26,7 @@ interface Harness {
26
26
  spawnOptions: Array<{ env: NodeJS.ProcessEnv; stdio?: string[] }>;
27
27
  }
28
28
 
29
- function harness(options: { pid?: number; maxConcurrency?: number; stallTimeoutMs?: number; answer?: Record<string, unknown>; exitOnKill?: boolean; state?: Record<string, unknown>; stateSuccess?: boolean; onNotification?: RunnerHooks["onNotification"]; onSuccessfulMutation?: RunnerHooks["onSuccessfulMutation"]; onFinish?: RunnerHooks["onFinish"] } = {}): Harness {
29
+ function harness(options: { failStart?: boolean; process?: RunnerDeps["process"]; pid?: number; maxConcurrency?: number; stallTimeoutMs?: number; toolStallTimeoutMs?: number; answer?: Record<string, unknown>; exitOnKill?: boolean; state?: Record<string, unknown>; stateSuccess?: boolean; onNotification?: RunnerHooks["onNotification"]; onSuccessfulMutation?: RunnerHooks["onSuccessfulMutation"]; onFinish?: RunnerHooks["onFinish"] } = {}): Harness {
30
30
  const children: FakeChild[] = [];
31
31
  const timers: Harness["timers"] = [];
32
32
  const asks: Harness["asks"] = [];
@@ -34,7 +34,9 @@ function harness(options: { pid?: number; maxConcurrency?: number; stallTimeoutM
34
34
  const spawnOptions: Harness["spawnOptions"] = [];
35
35
  let clock = 1000;
36
36
  const deps: RunnerDeps = {
37
+ process: options.process,
37
38
  spawn: (_command, _args, launchOptions) => {
39
+ if (options.failStart) throw new Error("fixture spawn failed");
38
40
  spawnOptions.push({ env: launchOptions.env, stdio: launchOptions.stdio });
39
41
  const fake = fakeChild({ exitOnKill: options.exitOnKill, pid: options.pid });
40
42
  if (options.state !== undefined) {
@@ -60,7 +62,7 @@ function harness(options: { pid?: number; maxConcurrency?: number; stallTimeoutM
60
62
  pi: { command: "pi", args: [] },
61
63
  };
62
64
  const store = new TaskStore();
63
- const runner = new AgentRunner(store, { maxConcurrency: options.maxConcurrency ?? 2, stallTimeoutMs: options.stallTimeoutMs ?? 10_000 }, deps, {
65
+ const runner = new AgentRunner(store, { maxConcurrency: options.maxConcurrency ?? 2, stallTimeoutMs: options.stallTimeoutMs ?? 10_000, toolStallTimeoutMs: options.toolStallTimeoutMs }, deps, {
64
66
  askUser: async (taskId, ask) => {
65
67
  asks.push({ taskId, method: ask.method });
66
68
  return options.answer ?? { value: "yes" };
@@ -127,7 +129,38 @@ test("stall after get_state and prompt responses records the prompt accepted sta
127
129
  assert.ok(stall);
128
130
  stall!.fn();
129
131
  await tick();
130
- assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted");
132
+ assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted; no first run event received for model: openai-codex/gpt-5.6-terra");
133
+ });
134
+
135
+ test("text progress after prompt acceptance retains the generic idle timeout diagnostic", async () => {
136
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS });
137
+ const task = h.runner.run(request());
138
+ await tick();
139
+ assert.equal(h.store.get(task.id)?.lastStep, "prompt accepted");
140
+ h.children[0].emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "progress" } });
141
+ await tick();
142
+ assert.equal(h.store.get(task.id)?.lastStep, "prompt accepted");
143
+ const stall = h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).at(-1);
144
+ assert.ok(stall);
145
+ stall.fn();
146
+ await tick();
147
+ const error = h.store.get(task.id)?.error;
148
+ assert.doesNotMatch(error!, /no first run event received/);
149
+ assert.equal(error, "stalled for 4 min after: prompt accepted");
150
+ });
151
+
152
+ test("prompt-accepted stall names the resolved model and preserves the stderr suffix", async () => {
153
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS, state: { model: { provider: "resolved-provider", id: "resolved-model" } } });
154
+ const task = h.runner.run(request());
155
+ await tick();
156
+ assert.equal(h.store.get(task.id)?.model, "resolved-provider/resolved-model");
157
+ (h.children[0].child.stderr as unknown as PassThrough).write("child diagnostic\n");
158
+ await tick();
159
+ const stall = h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).at(-1);
160
+ assert.ok(stall);
161
+ stall.fn();
162
+ await tick();
163
+ assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted; no first run event received for model: resolved-provider/resolved-model; stderr: child diagnostic");
131
164
  });
132
165
 
133
166
  test("stderr tail bounds the child's raw output to 512 characters before stripping ANSI escapes", async () => {
@@ -321,6 +354,20 @@ test("child observation guard is never consulted when collection is default-off"
321
354
  assert.equal(calls, 0);
322
355
  });
323
356
 
357
+ test("child session diff evidence travels only with a paired successful tool outcome", async () => {
358
+ const observed: any[] = [];
359
+ const h = harness({ onSuccessfulMutation: (_task, tool) => { observed.push(tool); } });
360
+ const task = h.runner.run(request()); await tick();
361
+ const child = h.children[0];
362
+ const evidence = {id:"w",root:"/repo",path:"src/file.ts",before:{kind:"absent"},after:{kind:"text",text:"agent\n"}};
363
+ child.emit({type:"tool_execution_start",toolCallId:"w",toolName:"write",args:{path:"src/file.ts"}});
364
+ child.emit({type:"tool_execution_end",toolCallId:"w",isError:false,result:{content:[],details:{gentleSessionChange:evidence}}});
365
+ assert.deepEqual(observed[0].evidence,evidence);
366
+ child.emit({type:"tool_execution_end",toolCallId:"w",isError:false,result:{content:[],details:{gentleSessionChange:evidence}}});
367
+ assert.equal(observed.length,1);
368
+ h.runner.cancel(task.id); await tick();
369
+ });
370
+
324
371
  for (const ending of ["cancel", "failure", "hook-error", "hook-async-error"] as const) {
325
372
  test(`successful child mutations require paired RPC events and survive ${ending}`, async () => {
326
373
  const mutations: unknown[] = [];
@@ -759,20 +806,108 @@ for (const [platform, detached] of [["win32", false], ["linux", true]] as const)
759
806
  assert.equal((await runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED);
760
807
  });
761
808
 
809
+ function ipcCleanupHarness(connected: boolean | undefined) {
810
+ const child = fakeChild({ exitOnKill: false });
811
+ const disconnectListeners: Array<(...args: unknown[]) => void> = [];
812
+ const originalOn = child.child.on as unknown as (event: string, listener: (...args: unknown[]) => void) => unknown;
813
+ child.child.on = ((event: string, listener: (...args: unknown[]) => void) => {
814
+ if (event === "disconnect") disconnectListeners.push(listener);
815
+ return originalOn(event, listener);
816
+ }) as ChildLike["on"];
817
+ child.child.connected = connected;
818
+ let resolveAnswer!: (answer: { cancelled: true }) => void;
819
+ const answer = new Promise<{ cancelled: true }>((resolve) => { resolveAnswer = resolve; });
820
+ const runner = new AgentRunner(new TaskStore(), { maxConcurrency: 1, stallTimeoutMs: 1_000 }, {
821
+ spawn: () => child.child,
822
+ now: () => 1,
823
+ schedule: () => () => {},
824
+ pi: { command: "pi", args: [] },
825
+ }, { askUser: async () => answer });
826
+ return {
827
+ child,
828
+ runner,
829
+ resolveAnswer,
830
+ emitNativeDisconnect: () => {
831
+ assert.equal(disconnectListeners.length, 1, "the runner listens for the native disconnect event");
832
+ disconnectListeners[0]();
833
+ },
834
+ };
835
+ }
836
+
837
+ test("AgentRunner primary IPC cleanup respects native connection state", async () => {
838
+ for (const scenario of [
839
+ { name: "connected=false finalize", connected: false, ending: "finalize", expectedDisconnects: 0, reentrant: false },
840
+ { name: "connected=false cancel", connected: false, ending: "cancel", expectedDisconnects: 0, reentrant: false },
841
+ { name: "connected=true reentrant cleanup", connected: true, ending: "cancel", expectedDisconnects: 1, reentrant: true },
842
+ { name: "partial fake without connected", connected: undefined, ending: "cancel", expectedDisconnects: 1, reentrant: false },
843
+ ] as const) {
844
+ const h = ipcCleanupHarness(scenario.connected);
845
+ const task = h.runner.run(request());
846
+ await tick();
847
+ h.child.emit({ type: "extension_ui_request", id: "pending", method: "confirm", title: "Pending?" });
848
+ await tick();
849
+ h.emitNativeDisconnect();
850
+ if (scenario.reentrant) h.emitNativeDisconnect();
851
+ assert.equal(h.child.disconnects, scenario.expectedDisconnects, `${scenario.name}: native disconnect does not duplicate the physical close`);
852
+ h.resolveAnswer({ cancelled: true });
853
+ await tick();
854
+ assert.equal(h.child.written.filter((command) => command.type === "extension_ui_response").length, 1, `${scenario.name}: IPC closure does not suppress the independent live RPC UI response`);
855
+ if (scenario.ending === "finalize") {
856
+ h.child.emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "done" }], stopReason: "stop" }] });
857
+ h.child.emit({ type: "agent_settled" });
858
+ h.child.exit(0);
859
+ assert.equal((await h.runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED, `${scenario.name}: later finalization remains intact`);
860
+ } else {
861
+ h.runner.cancel(task.id);
862
+ h.child.exit(0);
863
+ assert.equal((await h.runner.waitFor(task.id)).status, TASK_STATUS.CANCELLED, `${scenario.name}: later cancellation remains intact`);
864
+ }
865
+ assert.equal(h.child.disconnects, scenario.expectedDisconnects, `${scenario.name}: later cleanup remains idempotent`);
866
+ }
867
+ });
868
+
762
869
  test("AgentRunner retains permission broker fd3 and assigns messaging IPC to fd4", async () => {
763
870
  const { runner, children, spawnOptions } = harness();
764
871
  const task = runner.run(request({ authorizeParentStandingReviewPermission: () => true }));
765
872
  await tick();
766
873
  const launch = spawnOptions[0];
874
+ const permissionChannelStdio = process.platform === "win32" ? "overlapped" : "pipe";
767
875
  assert.match(launch?.env.GENTLE_PI_AGENTS_OWNED_IPC ?? "", /^\d+-[a-z0-9]+$/, "the owned-IPC marker has the runner's opaque shape");
768
876
  assert.deepEqual(launch?.env, { GENTLE_PI_AGENTS_CHILD: "1", GENTLE_PI_AGENTS_OWNED_IPC: launch?.env.GENTLE_PI_AGENTS_OWNED_IPC, GENTLE_PI_AGENTS_PARENT_PERMISSION_FD: "3" });
769
- assert.deepEqual(launch?.stdio, ["pipe", "pipe", "pipe", "pipe", "ipc"]);
877
+ assert.deepEqual(launch?.stdio, ["pipe", "pipe", "pipe", permissionChannelStdio, "ipc"]);
770
878
  assert.equal(launch?.stdio?.length, 5);
771
879
  children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "channel checked" }], stopReason: "stop" }] });
772
880
  children[0].emit({ type: "agent_settled" });
773
881
  assert.equal((await runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED);
774
882
  });
775
883
 
884
+ test("AgentRunner platform matrix scopes permission fd3 transport", async () => {
885
+ for (const platform of ["win32", "linux", "darwin"] as const) {
886
+ for (const eligible of [false, true]) {
887
+ const launches: Array<Parameters<RunnerDeps["spawn"]>[2]> = [];
888
+ const child = fakeChild();
889
+ const runner = new AgentRunner(new TaskStore(), { maxConcurrency: 1, stallTimeoutMs: 1_000 }, {
890
+ spawn: (_command, _args, options) => {
891
+ launches.push(options);
892
+ return child.child;
893
+ },
894
+ now: () => 1,
895
+ schedule: () => () => {},
896
+ pi: { command: "pi-fixture", args: [] },
897
+ process: { platform, kill: () => {} },
898
+ }, { askUser: async () => ({ cancelled: true }) });
899
+ const task = runner.run(request({ authorizeParentStandingReviewPermission: eligible ? () => true : undefined }));
900
+ await tick();
901
+ const launch = launches[0];
902
+ assert.ok(launch, `${platform} ${eligible ? "eligible" : "ineligible"} child launches`);
903
+ assert.equal(launch.env.GENTLE_PI_AGENTS_PARENT_PERMISSION_FD, eligible ? "3" : undefined, "only eligible children receive the fd3 marker");
904
+ assert.deepEqual(launch.stdio, eligible ? ["pipe", "pipe", "pipe", platform === "win32" ? "overlapped" : "pipe", "ipc"] : ["pipe", "pipe", "pipe", "ipc"]);
905
+ assert.equal(launch.stdio?.indexOf("ipc"), eligible ? 4 : 3, "messaging IPC follows fd3 only for eligible children");
906
+ runner.cancel(task.id);
907
+ }
908
+ }
909
+ });
910
+
776
911
  test("AgentRunner answers dialogs through askUser in task mode and cancels them in background mode", async () => {
777
912
  const { store, runner, children, asks } = harness({ answer: { confirmed: true } });
778
913
  const task = runner.run(request());
@@ -815,8 +950,11 @@ test("AgentRunner has no total-duration watchdog but keeps active work alive and
815
950
  children[0].emit({ type: "response", id: "r1", success: true });
816
951
  await tick();
817
952
  assert.equal(initialStall.cancelled, true, "every child RPC event, including a response, re-arms the inactivity watchdog");
953
+ const afterResponse = timers.filter((timer) => timer.ms === 10_000 && !timer.cancelled).at(-1);
954
+ assert.ok(afterResponse);
818
955
  children[0].emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "still working" } });
819
956
  await tick();
957
+ assert.equal(afterResponse!.cancelled, true, "normalized task progress re-arms the inactivity watchdog");
820
958
  assert.equal(store.get(task.id)?.status, TASK_STATUS.RUNNING, "ongoing RPC activity keeps a long-running task active");
821
959
  const stall = timers.filter((timer) => timer.ms === 10_000 && !timer.cancelled).at(-1);
822
960
  assert.ok(stall);
@@ -826,6 +964,104 @@ test("AgentRunner has no total-duration watchdog but keeps active work alive and
826
964
  assert.match(store.get(task.id)?.error ?? "", /stalled/);
827
965
  });
828
966
 
967
+ test("an announced tool call in flight arms the tool ceiling and names the tool when it fires", async () => {
968
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
969
+ const task = h.runner.run(request());
970
+ await tick();
971
+ h.children[0].emit({ type: "tool_execution_start", toolCallId: "t1", toolName: "bash", args: { command: "pnpm test" } });
972
+ await tick();
973
+ assert.equal(h.store.get(task.id)?.lastStep, "bash");
974
+ assert.equal(h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).length, 0, "the idle budget no longer bounds a task with a tool in flight");
975
+ const toolTimer = h.timers.filter((timer) => timer.ms === 30 * 60_000 && !timer.cancelled).at(-1);
976
+ assert.ok(toolTimer, "an in-flight tool call arms the tool ceiling");
977
+ toolTimer!.fn();
978
+ await tick();
979
+ assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
980
+ assert.equal(h.store.get(task.id)?.error, 'stalled for 30 min with tool "bash" still running after: bash');
981
+ });
982
+
983
+ test("a finished tool call returns the task to the idle silence budget", async () => {
984
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
985
+ const task = h.runner.run(request());
986
+ await tick();
987
+ h.children[0].emit({ type: "tool_execution_start", toolCallId: "t1", toolName: "bash", args: { command: "pnpm test" } });
988
+ await tick();
989
+ h.children[0].emit({ type: "tool_execution_end", toolCallId: "t1", isError: false });
990
+ await tick();
991
+ assert.equal(h.timers.filter((timer) => timer.ms === 30 * 60_000 && !timer.cancelled).length, 0, "a finished tool is back on the idle budget");
992
+ const idle = h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).at(-1);
993
+ assert.ok(idle, "tool_end re-arms the idle budget");
994
+ idle!.fn();
995
+ await tick();
996
+ assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
997
+ assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: bash");
998
+ });
999
+
1000
+ test("ignored non-dialog UI traffic does not renew the idle silence budget", async () => {
1001
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
1002
+ const task = h.runner.run(request());
1003
+ await tick();
1004
+ const armed = h.timers.at(-1);
1005
+ assert.ok(armed, "launch arms the idle silence budget");
1006
+ assert.equal(armed!.ms, FOUR_MIN_MS);
1007
+ // Fire-and-forget UI notifications normalize to zero task events and prove
1008
+ // only that the transport is alive; they must not postpone the silence bound.
1009
+ h.children[0].emit({ type: "extension_ui_request", id: "u1", method: "setStatus", statusKey: "fixture", statusText: "idle" });
1010
+ h.children[0].emit({ type: "extension_ui_request", id: "u2", method: "notify", message: "still here" });
1011
+ await tick();
1012
+ assert.equal(armed!.cancelled, false, "ignored UI traffic must not cancel the armed silence budget");
1013
+ assert.equal(h.timers.filter((timer) => !timer.cancelled && timer.ms === FOUR_MIN_MS).length, 1, "no replacement timer is scheduled for ignored UI traffic");
1014
+ assert.equal(h.timers.filter((timer) => timer.ms === 30 * 60_000).length, 0, "ignored UI traffic never earns the tool ceiling");
1015
+ armed!.fn();
1016
+ await tick();
1017
+ assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
1018
+ assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted; no first run event received for model: openai-codex/gpt-5.6-terra");
1019
+ });
1020
+
1021
+ test("an unrecognized RPC object does not renew the idle silence budget", async () => {
1022
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS });
1023
+ const task = h.runner.run(request());
1024
+ await tick();
1025
+ const armed = h.timers.at(-1);
1026
+ assert.ok(armed);
1027
+ h.children[0].emit({ type: "some_future_event", payload: { nested: true } });
1028
+ await tick();
1029
+ assert.equal(armed!.cancelled, false, "an unknown object is not progress");
1030
+ assert.equal(h.timers.filter((timer) => !timer.cancelled).length, 1);
1031
+ armed!.fn();
1032
+ await tick();
1033
+ assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
1034
+ });
1035
+
1036
+ test("a blocking child dialog still re-arms the idle silence budget", async () => {
1037
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS, answer: { confirmed: true } });
1038
+ h.runner.run(request());
1039
+ await tick();
1040
+ const armed = h.timers.at(-1);
1041
+ assert.ok(armed);
1042
+ h.children[0].emit({ type: "extension_ui_request", id: "u1", method: "confirm", title: "Continue?" });
1043
+ await tick();
1044
+ assert.equal(armed!.cancelled, true, "a dialog the parent must answer is meaningful activity");
1045
+ assert.equal(h.asks.length, 1);
1046
+ });
1047
+
1048
+ test("the tool ceiling holds while any announced tool call is still in flight", async () => {
1049
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
1050
+ h.runner.run(request());
1051
+ await tick();
1052
+ h.children[0].emit({ type: "tool_execution_start", toolCallId: "t1", toolName: "bash", args: { command: "pnpm test" } });
1053
+ await tick();
1054
+ h.children[0].emit({ type: "tool_execution_start", toolCallId: "t2", toolName: "bash", args: { command: "pnpm run typecheck" } });
1055
+ await tick();
1056
+ h.children[0].emit({ type: "tool_execution_end", toolCallId: "t1", isError: false });
1057
+ await tick();
1058
+ assert.ok(h.timers.filter((timer) => timer.ms === 30 * 60_000 && !timer.cancelled).length > 0, "a second tool still in flight keeps the tool ceiling");
1059
+ assert.equal(h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).length, 0);
1060
+ h.children[0].emit({ type: "tool_execution_end", toolCallId: "t2", isError: false });
1061
+ await tick();
1062
+ assert.ok(h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).length > 0, "ending the last tool returns to the idle budget");
1063
+ });
1064
+
829
1065
  test("AgentRunner.cancelAll stops every queued and running task", async () => {
830
1066
  const { store, runner, children } = harness({ maxConcurrency: 1 });
831
1067
  const running = runner.run(request());
@@ -998,8 +1234,9 @@ test("an unprobeable process group quarantines at its deadline and still records
998
1234
  const finishes: string[] = [];
999
1235
  let now = 1_000;
1000
1236
  let launches = 0;
1237
+ let child: FakeChild;
1001
1238
  const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 10_000 }, {
1002
- spawn: () => fakeChild({ exitOnKill: false, pid: 90 + (launches += 1) }).child, now: () => now,
1239
+ spawn: () => (child = fakeChild({ exitOnKill: false, pid: 90 + (launches += 1) })).child, now: () => now,
1003
1240
  schedule: (fn, ms) => {
1004
1241
  const timer = { fn, ms, cancelled: false };
1005
1242
  timers.push(timer);
@@ -1008,7 +1245,7 @@ test("an unprobeable process group quarantines at its deadline and still records
1008
1245
  pi: { command: "pi", args: [] },
1009
1246
  process: { platform: "win32", kill: () => {} },
1010
1247
  }, { askUser: async () => ({ value: "yes" }), onFinish: (task) => { finishes.push(task.id); } });
1011
- const first = runner.run(request());
1248
+ const first = runner.run(managedRequest());
1012
1249
  const second = runner.run(request({ prompt: "queued" }));
1013
1250
  await tick();
1014
1251
  runner.cancel(first.id);
@@ -1025,6 +1262,12 @@ test("an unprobeable process group quarantines at its deadline and still records
1025
1262
  assert.equal(finishes.length, 1, "the run is recorded exactly once");
1026
1263
  assert.equal(store.get(second.id)?.status, TASK_STATUS.QUEUED, "an unconfirmed exit retains its capacity");
1027
1264
  assert.equal(launches, 1, "no further launch happens while the slot is quarantined");
1265
+ assert.throws(() => runner.run(managedRequest()), /Remediation already queued or running/, "a failed record does not release its quarantined child");
1266
+ child!.exit(0);
1267
+ await tick();
1268
+ assert.doesNotThrow(() => runner.run(managedRequest()), "confirmed cleanup releases the managed workspace");
1269
+ runner.cancelAll();
1270
+ child!.exit(0);
1028
1271
  });
1029
1272
 
1030
1273
  test("abortReasonText renders an Error, a string, and nothing for unknown reasons", () => {
@@ -1039,17 +1282,12 @@ test("research narrowing transport keeps exact argv paths and replaces inherited
1039
1282
  const h = harness();
1040
1283
  const selection = { documentation: { tools: ["fetch_content"], extensions: { fetch_content: "/installed/docs tools.ts" } } };
1041
1284
  for (const researchSelection of [selection, undefined]) {
1042
- const artifact = { store: "none" as const, worktree: "/work", changeName: "demo", retainedIntent: "denied questions", locators: [] };
1043
- const expected = structuredClone(artifact);
1044
- const launch = request({ researchSelection, researchArtifact: artifact, extensionPaths: researchSelection ? ["/installed/docs tools.ts"] : [],
1045
- env: { PATH: "/bin", GENTLE_PI_RESEARCH_SELECTION: "stale broad selection", GENTLE_PI_RESEARCH_ARTIFACT: "stale broader scope" } });
1285
+ const launch = request({ researchSelection, extensionPaths: researchSelection ? ["/installed/docs tools.ts"] : [],
1286
+ env: { PATH: "/bin", GENTLE_PI_RESEARCH_SELECTION: "stale broad selection" } });
1046
1287
  const argv = childArguments(launch);
1047
1288
  assert.deepEqual(argv.filter((_, i) => argv[i - 1] === "--extension"), launch.extensionPaths);
1048
1289
  const task = h.runner.run(launch);
1049
- artifact.worktree = "/wrong";
1050
1290
  await tick();
1051
- assert.deepEqual(JSON.parse(h.spawnOptions.at(-1)!.env.GENTLE_PI_RESEARCH_ARTIFACT!), expected);
1052
- assert.deepEqual("researchArtifact" in task ? task.researchArtifact : undefined, expected);
1053
1291
  assert.deepEqual(JSON.parse(h.spawnOptions.at(-1)!.env.GENTLE_PI_RESEARCH_SELECTION!), researchSelection ?? null);
1054
1292
  assert.equal(h.spawnOptions.at(-1)!.env.PATH, "/bin");
1055
1293
  h.runner.cancel(task.id);
@@ -1057,55 +1295,80 @@ test("research narrowing transport keeps exact argv paths and replaces inherited
1057
1295
  }
1058
1296
  });
1059
1297
 
1298
+ function managedRequest(cwd = "/repo"): TaskRequest {
1299
+ return request({ agent: { ...explorer, name: "sdd-remediate" }, cwd, sddRemediation: {
1300
+ failedEvidenceRevision: "failed-revision",
1301
+ plan: { cwd, commands: ["pnpm test"], runtimeHarness: { naReason: "Not applicable because this tests runner admission." }, rollback: { boundary: "fixture", command: "git diff --check" } },
1302
+ scope: { cwd, editPaths: [], commands: ["pnpm test", "git diff --check"], allowedEditRoots: [cwd] },
1303
+ } });
1304
+ }
1060
1305
 
1061
- test("runner retains remediation observations and awaits terminal settlement", async () => {
1062
- const h = harness({ pid: 123 });
1063
- let finalized: { record: TaskRecord; facts: RemediationTerminalFacts } | undefined;
1064
- let release: () => void;
1065
- const gate = new Promise<void>(resolve => { release = resolve; });
1066
- const remediationPlan: RemediationPlan = { cwd: "/repo", commands: ["pnpm test"], runtimeHarness: { naReason: "Not applicable because the fixture has no runtime boundary." }, rollback: { boundary: "Revert fixture", command: "git diff --check" } };
1067
- const remediation: RemediationTaskState = { failedEvidenceRevision: `sha256:${"a".repeat(64)}`, plan: remediationPlan, pending: {}, observations: [], invalid: false, token: "opaque", acquire: { workspaceRoot: "/repo", changeName: "demo", requestId: "fixture", workUnit: "correct", evidenceGoal: "Observed correction" } };
1068
- const task = h.runner.run(request({ sddRemediation: remediation, finalizeRemediation: async (record, facts) => { finalized = { record, facts }; await gate; } }));
1069
- await tick();
1070
- for (const command of ["pnpm test", "git diff --check"]) {
1071
- h.children[0].emit({ type: "tool_execution_start", toolName: "bash", toolCallId: command, args: { command } });
1072
- h.children[0].emit({ type: "tool_execution_end", toolName: "bash", toolCallId: command, isError: false, result: { content: [{ type: "text", text: "command output" }], details: { remediationCommand: { command, toolCallId: command, cwd: "/repo", exitCode: 0 } } } });
1073
- }
1074
- h.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "done" }], stopReason: "stop" }] });
1075
- h.children[0].emit({ type: "agent_settled" });
1076
- await tick();
1077
- assert.ok(finalized);
1078
- assert.equal(finalized.record.sddRemediation?.observations.length, 2);
1079
- const prompt = h.children[0].written.find(value => value.type === "prompt").message;
1080
- assert.ok(typeof prompt === "string");
1081
- assert.match(prompt, /pnpm test/);
1082
- assert.doesNotMatch(prompt, /opaque/);
1083
- assert.equal(finalized.facts.cleanupConfirmed, true);
1084
- assert.equal(h.finishes.length, 0);
1085
- release(); await h.runner.waitFor(task.id);
1086
- assert.equal(h.finishes.length, 1);
1087
- assert.equal(remediation.observations.length, 0);
1306
+ for (const queued of [true, false]) test(`managed exclusion covers ${queued ? "queued" : "running"} same-workspace actors`, async () => {
1307
+ const h = harness({ pid: 123, process: { platform: "win32", kill() {} } });
1308
+ const first = h.runner.run(managedRequest());
1309
+ if (!queued) await tick();
1310
+ try {
1311
+ assert.throws(() => h.runner.run(managedRequest()), /Remediation already queued or running/);
1312
+ assert.equal(h.store.list().length, 1, "rejection creates no task or queue entry");
1313
+ await tick();
1314
+ assert.equal(h.children.length, 1);
1315
+ assert.equal(h.store.get(first.id)?.status, TASK_STATUS.RUNNING);
1316
+ } finally { h.runner.cancelAll(); await tick(); }
1088
1317
  });
1089
1318
 
1319
+ test("managed exclusion does not serialize other workspaces or ordinary tasks", async () => {
1320
+ const h = harness({ maxConcurrency: 3, pid: 123, process: { platform: "win32", kill() {} } });
1321
+ h.runner.run(managedRequest());
1322
+ h.runner.run(managedRequest("/other"));
1323
+ h.runner.run(request());
1324
+ await tick();
1325
+ assert.equal(h.children.length, 3);
1326
+ h.runner.cancelAll();
1327
+ await tick();
1328
+ });
1090
1329
 
1091
- test("admitted no-PID failure settles interrupted before sending a prompt", async () => {
1092
- const h = harness(); let facts;
1093
- const task = h.runner.run(request({ sddRemediation: { plan: {} } as unknown as RemediationTaskState, finalizeRemediation: async (_task, observed) => { facts = observed; } }));
1330
+ for (const ending of ["complete", "failure", "cancel", "queued-cancel"] as const) test(`managed exclusion releases after ${ending}`, async () => {
1331
+ const h = harness({ pid: 123, process: { platform: "win32", kill() {} } });
1332
+ const first = h.runner.run(managedRequest());
1333
+ if (ending === "queued-cancel") h.runner.cancel(first.id);
1334
+ else {
1335
+ await tick();
1336
+ if (ending === "complete") {
1337
+ h.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "Finished" }], stopReason: "stop" }] });
1338
+ h.children[0].emit({ type: "agent_settled" });
1339
+ } else if (ending === "failure") h.children[0].exit(1);
1340
+ else h.runner.cancel(first.id);
1341
+ }
1342
+ await h.runner.waitFor(first.id);
1343
+ const next = h.runner.run(managedRequest());
1344
+ await tick();
1345
+ assert.equal(h.store.get(next.id)?.status, TASK_STATUS.RUNNING);
1346
+ h.runner.cancelAll();
1094
1347
  await tick();
1095
- assert.equal(h.children[0].written.some(value => value.type === "prompt"), false);
1096
- await h.runner.waitFor(task.id);
1097
- assert.equal(facts.spawned, false);
1098
- assert.equal(facts.exited, false);
1099
- assert.equal(facts.cleanupConfirmed, false);
1100
1348
  });
1101
1349
 
1350
+ test("managed exclusion lasts until child cleanup is confirmed", async () => {
1351
+ const h = harness({ pid: 123, exitOnKill: false, process: { platform: "win32", kill() {} } });
1352
+ const first = h.runner.run(managedRequest());
1353
+ await tick();
1354
+ h.runner.cancel(first.id);
1355
+ try { assert.throws(() => h.runner.run(managedRequest()), /Remediation already queued or running/); }
1356
+ finally { h.children[0].exit(0); }
1357
+ await h.runner.waitFor(first.id);
1358
+ const next = h.runner.run(managedRequest());
1359
+ await tick();
1360
+ assert.equal(h.store.get(next.id)?.status, TASK_STATUS.RUNNING);
1361
+ h.runner.cancelAll();
1362
+ for (const child of h.children) child.exit(0);
1363
+ await tick();
1364
+ });
1102
1365
 
1103
- test("admitted synchronous spawn failure finalizes without launching another actor", async () => {
1104
- let spawns = 0, facts;
1105
- const store = new TaskStore();
1106
- const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 100 }, { spawn: () => { spawns++; throw new Error("spawn refused"); }, pi: { command: "pi", args: [] }, now: () => 1, schedule: () => () => {} }, { askUser: async () => ({}) });
1107
- const task = runner.run(request({ sddRemediation: { plan: {} } as unknown as RemediationTaskState, finalizeRemediation: async (_task, value) => { facts = value; } }));
1108
- await runner.waitFor(task.id);
1109
- assert.deepEqual(facts, { spawned: false, exited: false, cleanupConfirmed: true });
1110
- assert.equal(spawns, 1);
1366
+ test("managed exclusion releases failed startup and ignores historical-only tasks", async () => {
1367
+ const h = harness({ failStart: true });
1368
+ const first = h.runner.run(managedRequest());
1369
+ assert.equal((await h.runner.waitFor(first.id)).status, TASK_STATUS.FAILED);
1370
+ h.store.add({ ...h.store.get(first.id)!, id: "historical-only", status: TASK_STATUS.RUNNING });
1371
+ const next = h.runner.run(managedRequest());
1372
+ assert.equal((await h.runner.waitFor(next.id)).status, TASK_STATUS.FAILED, "the next actor reaches spawn, not a historical admission lock");
1373
+ assert.match(h.store.get(next.id)?.error ?? "", /fixture spawn failed/);
1111
1374
  });