gentle-pi 2.7.0 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (122) hide show
  1. package/README.md +24 -6
  2. package/assets/agents/gentle-ai-worker.md +5 -1
  3. package/assets/agents/sdd-apply.md +9 -7
  4. package/assets/agents/sdd-archive.md +42 -23
  5. package/assets/agents/sdd-proposal.md +2 -2
  6. package/assets/agents/sdd-remediate.md +4 -4
  7. package/assets/agents/sdd-research.md +20 -48
  8. package/assets/agents/sdd-tasks.md +5 -5
  9. package/assets/agents/sdd-verify.md +6 -28
  10. package/assets/chains/sdd-full.chain.md +4 -22
  11. package/assets/chains/sdd-verify.chain.md +3 -12
  12. package/assets/orchestrator-delegation.md +33 -3
  13. package/assets/orchestrator-memory.md +20 -7
  14. package/assets/orchestrator.md +5 -3
  15. package/assets/sdd-orchestrator-workflow.md +25 -58
  16. package/assets/support/sdd-status-contract.md +9 -12
  17. package/docs/gentle-shell.md +14 -4
  18. package/docs/readme-reference.md +171 -32
  19. package/extensions/codegraph-tools.ts +2 -0
  20. package/extensions/gentle-agents.ts +281 -361
  21. package/extensions/gentle-ai.ts +588 -117
  22. package/extensions/gentle-shell.ts +74 -30
  23. package/extensions/pi-pretty.ts +63 -14
  24. package/extensions/quiet-tools.ts +1 -2
  25. package/extensions/startup-banner.ts +10 -9
  26. package/lib/agent-home.ts +8 -0
  27. package/lib/agent-profile-pin.ts +336 -0
  28. package/lib/agent-profiles.ts +28 -8
  29. package/lib/agents-config.ts +24 -2
  30. package/lib/agents-history.ts +3 -97
  31. package/lib/agents-keys.ts +27 -0
  32. package/lib/agents-protocol.ts +2 -15
  33. package/lib/agents-runner.ts +67 -111
  34. package/lib/agents-session-transport.ts +691 -0
  35. package/lib/command-palette-catalog.ts +87 -0
  36. package/lib/command-palette.ts +346 -0
  37. package/lib/native-choice-list.ts +5 -0
  38. package/lib/native-review-cli.ts +10 -97
  39. package/lib/review-publication-gate.ts +11 -1
  40. package/lib/review-repository.ts +1 -1
  41. package/lib/review-snapshot.ts +1 -0
  42. package/lib/review-transaction.ts +4 -2
  43. package/lib/sdd-preflight.ts +2 -1
  44. package/lib/sdd-research-capabilities.ts +18 -152
  45. package/lib/sdd-status.ts +7 -779
  46. package/lib/session-change-capture.ts +2 -1
  47. package/lib/session-changes.ts +8 -1
  48. package/lib/shell-bar.ts +21 -12
  49. package/lib/shell-card.ts +8 -12
  50. package/lib/shell-changes.ts +3 -2
  51. package/lib/shell-prompt.ts +25 -8
  52. package/lib/shell-sidebar-banner.ts +2 -2
  53. package/lib/shell-sidebar-layout.ts +5 -2
  54. package/lib/windows-session-transport.ts +877 -0
  55. package/package.json +3 -3
  56. package/runtime/native-review-cli.mjs +9 -96
  57. package/runtime/windows-session-transport.ps1 +791 -0
  58. package/scripts/test-packed-runner.mjs +1668 -20
  59. package/scripts/verify-package-files.mjs +0 -1
  60. package/tests/agent-home.test.ts +52 -0
  61. package/tests/agent-profiles.test.ts +30 -1
  62. package/tests/agents-config.test.ts +44 -0
  63. package/tests/agents-history.test.ts +12 -24
  64. package/tests/agents-runner.test.ts +307 -58
  65. package/tests/agents-session-transport-process.test.ts +249 -0
  66. package/tests/agents-session-transport.test.ts +823 -0
  67. package/tests/artifact-language.test.ts +10 -7
  68. package/tests/command-palette.test.ts +378 -0
  69. package/tests/delegated-key-learnings-contract.test.ts +2 -2
  70. package/tests/fixtures/agents-session-transport-process.mjs +108 -0
  71. package/tests/fixtures/legacy/sdd-research-v2.5.0.md +54 -0
  72. package/tests/fixtures/windows-session-bootstrap.ps1 +129 -0
  73. package/tests/fixtures/windows-session-compile.ps1 +110 -0
  74. package/tests/gentle-agents.test.ts +849 -356
  75. package/tests/gentle-ai.test.ts +472 -4
  76. package/tests/gentle-shell.test.ts +201 -8
  77. package/tests/native-choice-list.test.ts +13 -0
  78. package/tests/native-review-cli.test.ts +0 -33
  79. package/tests/odd-routing-contract.test.ts +208 -0
  80. package/tests/orchestrator-budget.test.ts +17 -2
  81. package/tests/package-manifest.test.ts +115 -27
  82. package/tests/persona-single-channel.test.ts +3 -3
  83. package/tests/pi-pretty.test.ts +45 -0
  84. package/tests/profile-pin.test.ts +370 -0
  85. package/tests/quiet-tool-rendering.test.ts +32 -5
  86. package/tests/review-contract-prompt.test.ts +9 -0
  87. package/tests/review-controller.test.ts +0 -44
  88. package/tests/review-session-standing-permission-ipc.test.ts +427 -13
  89. package/tests/runtime-harness.mjs +4 -4
  90. package/tests/sdd-agent-tools.test.ts +15 -36
  91. package/tests/sdd-archive-replay.test.ts +82 -0
  92. package/tests/sdd-classical-continuation.test.ts +74 -0
  93. package/tests/sdd-execution-routing-contract.test.ts +18 -2
  94. package/tests/sdd-managed-runtime-settlement.test.ts +42 -330
  95. package/tests/sdd-native-managed-uptake.test.ts +11 -21
  96. package/tests/sdd-no-attempts-contract.test.ts +15 -0
  97. package/tests/sdd-odd-integration.test.ts +33 -0
  98. package/tests/sdd-optional-research.test.ts +124 -0
  99. package/tests/sdd-planning-routing-contract.test.ts +1 -1
  100. package/tests/sdd-preflight-rpc-input.test.ts +125 -0
  101. package/tests/sdd-preflight.test.ts +1 -1
  102. package/tests/sdd-research-capabilities.test.ts +20 -162
  103. package/tests/sdd-selection-transport.test.ts +180 -88
  104. package/tests/sdd-status.test.ts +5 -778
  105. package/tests/sdd-task-truth.test.ts +43 -0
  106. package/tests/session-change-capture.test.ts +20 -2
  107. package/tests/session-changes.test.ts +11 -0
  108. package/tests/shell-bar.test.ts +21 -0
  109. package/tests/shell-card.test.ts +8 -6
  110. package/tests/shell-changes.test.ts +8 -0
  111. package/tests/shell-prompt.test.ts +41 -7
  112. package/tests/shell-sidebar-banner.test.ts +4 -4
  113. package/tests/shell-sidebar-layout.test.ts +97 -13
  114. package/tests/startup-banner.test.ts +55 -2
  115. package/tests/windows-hidden-processes.test.ts +303 -0
  116. package/tests/windows-session-bootstrap.test.ts +1772 -0
  117. package/tests/windows-session-compile.test.ts +170 -0
  118. package/tests/windows-session-transport.test.ts +754 -0
  119. package/assets/agents/sdd-sync.md +0 -146
  120. package/lib/openspec-guardrails.ts +0 -99
  121. package/tests/native-sdd-attempt-authority.test.ts +0 -240
  122. package/tests/openspec-guardrails.test.ts +0 -71
@@ -2,8 +2,8 @@ import assert from "node:assert/strict";
2
2
  import test from "node:test";
3
3
  import { PassThrough } from "node:stream";
4
4
  import { AGENT_MODE, parseAgentsConfig, resolveAgentProfile, type AgentDefinition } from "../lib/agents-config.ts";
5
- import { TASK_STATUS, TaskStore, type RemediationTaskState, type TaskRecord } from "../lib/agents-protocol.ts";
6
- import { AgentRunner, childArguments, JsonLines, piCommand, abortReasonText, type RemediationPlan, type RemediationTerminalFacts, type RunnerDeps, type RunnerHooks, type TaskRequest } from "../lib/agents-runner.ts";
5
+ import { TASK_STATUS, TaskStore, type TaskRecord } from "../lib/agents-protocol.ts";
6
+ import { AgentRunner, childArguments, JsonLines, piCommand, abortReasonText, type ChildLike, type RunnerDeps, type RunnerHooks, type TaskRequest } from "../lib/agents-runner.ts";
7
7
  import { fakeChild, type FakeChild } from "./agents-fake-child.ts";
8
8
 
9
9
  // Gentle Agents runner: every subagent is a child `pi --mode rpc` process.
@@ -26,7 +26,7 @@ interface Harness {
26
26
  spawnOptions: Array<{ env: NodeJS.ProcessEnv; stdio?: string[] }>;
27
27
  }
28
28
 
29
- function harness(options: { pid?: number; maxConcurrency?: number; stallTimeoutMs?: number; answer?: Record<string, unknown>; exitOnKill?: boolean; state?: Record<string, unknown>; stateSuccess?: boolean; onNotification?: RunnerHooks["onNotification"]; onSuccessfulMutation?: RunnerHooks["onSuccessfulMutation"]; onFinish?: RunnerHooks["onFinish"] } = {}): Harness {
29
+ function harness(options: { failStart?: boolean; process?: RunnerDeps["process"]; pid?: number; maxConcurrency?: number; stallTimeoutMs?: number; toolStallTimeoutMs?: number; answer?: Record<string, unknown>; exitOnKill?: boolean; state?: Record<string, unknown>; stateSuccess?: boolean; onNotification?: RunnerHooks["onNotification"]; onSuccessfulMutation?: RunnerHooks["onSuccessfulMutation"]; onFinish?: RunnerHooks["onFinish"] } = {}): Harness {
30
30
  const children: FakeChild[] = [];
31
31
  const timers: Harness["timers"] = [];
32
32
  const asks: Harness["asks"] = [];
@@ -34,7 +34,9 @@ function harness(options: { pid?: number; maxConcurrency?: number; stallTimeoutM
34
34
  const spawnOptions: Harness["spawnOptions"] = [];
35
35
  let clock = 1000;
36
36
  const deps: RunnerDeps = {
37
+ process: options.process,
37
38
  spawn: (_command, _args, launchOptions) => {
39
+ if (options.failStart) throw new Error("fixture spawn failed");
38
40
  spawnOptions.push({ env: launchOptions.env, stdio: launchOptions.stdio });
39
41
  const fake = fakeChild({ exitOnKill: options.exitOnKill, pid: options.pid });
40
42
  if (options.state !== undefined) {
@@ -60,7 +62,7 @@ function harness(options: { pid?: number; maxConcurrency?: number; stallTimeoutM
60
62
  pi: { command: "pi", args: [] },
61
63
  };
62
64
  const store = new TaskStore();
63
- const runner = new AgentRunner(store, { maxConcurrency: options.maxConcurrency ?? 2, stallTimeoutMs: options.stallTimeoutMs ?? 10_000 }, deps, {
65
+ const runner = new AgentRunner(store, { maxConcurrency: options.maxConcurrency ?? 2, stallTimeoutMs: options.stallTimeoutMs ?? 10_000, toolStallTimeoutMs: options.toolStallTimeoutMs }, deps, {
64
66
  askUser: async (taskId, ask) => {
65
67
  asks.push({ taskId, method: ask.method });
66
68
  return options.answer ?? { value: "yes" };
@@ -127,7 +129,38 @@ test("stall after get_state and prompt responses records the prompt accepted sta
127
129
  assert.ok(stall);
128
130
  stall!.fn();
129
131
  await tick();
130
- assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted");
132
+ assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted; no first run event received for model: openai-codex/gpt-5.6-terra");
133
+ });
134
+
135
+ test("text progress after prompt acceptance retains the generic idle timeout diagnostic", async () => {
136
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS });
137
+ const task = h.runner.run(request());
138
+ await tick();
139
+ assert.equal(h.store.get(task.id)?.lastStep, "prompt accepted");
140
+ h.children[0].emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "progress" } });
141
+ await tick();
142
+ assert.equal(h.store.get(task.id)?.lastStep, "prompt accepted");
143
+ const stall = h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).at(-1);
144
+ assert.ok(stall);
145
+ stall.fn();
146
+ await tick();
147
+ const error = h.store.get(task.id)?.error;
148
+ assert.doesNotMatch(error!, /no first run event received/);
149
+ assert.equal(error, "stalled for 4 min after: prompt accepted");
150
+ });
151
+
152
+ test("prompt-accepted stall names the resolved model and preserves the stderr suffix", async () => {
153
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS, state: { model: { provider: "resolved-provider", id: "resolved-model" } } });
154
+ const task = h.runner.run(request());
155
+ await tick();
156
+ assert.equal(h.store.get(task.id)?.model, "resolved-provider/resolved-model");
157
+ (h.children[0].child.stderr as unknown as PassThrough).write("child diagnostic\n");
158
+ await tick();
159
+ const stall = h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).at(-1);
160
+ assert.ok(stall);
161
+ stall.fn();
162
+ await tick();
163
+ assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted; no first run event received for model: resolved-provider/resolved-model; stderr: child diagnostic");
131
164
  });
132
165
 
133
166
  test("stderr tail bounds the child's raw output to 512 characters before stripping ANSI escapes", async () => {
@@ -773,20 +806,108 @@ for (const [platform, detached] of [["win32", false], ["linux", true]] as const)
773
806
  assert.equal((await runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED);
774
807
  });
775
808
 
809
+ function ipcCleanupHarness(connected: boolean | undefined) {
810
+ const child = fakeChild({ exitOnKill: false });
811
+ const disconnectListeners: Array<(...args: unknown[]) => void> = [];
812
+ const originalOn = child.child.on as unknown as (event: string, listener: (...args: unknown[]) => void) => unknown;
813
+ child.child.on = ((event: string, listener: (...args: unknown[]) => void) => {
814
+ if (event === "disconnect") disconnectListeners.push(listener);
815
+ return originalOn(event, listener);
816
+ }) as ChildLike["on"];
817
+ child.child.connected = connected;
818
+ let resolveAnswer!: (answer: { cancelled: true }) => void;
819
+ const answer = new Promise<{ cancelled: true }>((resolve) => { resolveAnswer = resolve; });
820
+ const runner = new AgentRunner(new TaskStore(), { maxConcurrency: 1, stallTimeoutMs: 1_000 }, {
821
+ spawn: () => child.child,
822
+ now: () => 1,
823
+ schedule: () => () => {},
824
+ pi: { command: "pi", args: [] },
825
+ }, { askUser: async () => answer });
826
+ return {
827
+ child,
828
+ runner,
829
+ resolveAnswer,
830
+ emitNativeDisconnect: () => {
831
+ assert.equal(disconnectListeners.length, 1, "the runner listens for the native disconnect event");
832
+ disconnectListeners[0]();
833
+ },
834
+ };
835
+ }
836
+
837
+ test("AgentRunner primary IPC cleanup respects native connection state", async () => {
838
+ for (const scenario of [
839
+ { name: "connected=false finalize", connected: false, ending: "finalize", expectedDisconnects: 0, reentrant: false },
840
+ { name: "connected=false cancel", connected: false, ending: "cancel", expectedDisconnects: 0, reentrant: false },
841
+ { name: "connected=true reentrant cleanup", connected: true, ending: "cancel", expectedDisconnects: 1, reentrant: true },
842
+ { name: "partial fake without connected", connected: undefined, ending: "cancel", expectedDisconnects: 1, reentrant: false },
843
+ ] as const) {
844
+ const h = ipcCleanupHarness(scenario.connected);
845
+ const task = h.runner.run(request());
846
+ await tick();
847
+ h.child.emit({ type: "extension_ui_request", id: "pending", method: "confirm", title: "Pending?" });
848
+ await tick();
849
+ h.emitNativeDisconnect();
850
+ if (scenario.reentrant) h.emitNativeDisconnect();
851
+ assert.equal(h.child.disconnects, scenario.expectedDisconnects, `${scenario.name}: native disconnect does not duplicate the physical close`);
852
+ h.resolveAnswer({ cancelled: true });
853
+ await tick();
854
+ assert.equal(h.child.written.filter((command) => command.type === "extension_ui_response").length, 1, `${scenario.name}: IPC closure does not suppress the independent live RPC UI response`);
855
+ if (scenario.ending === "finalize") {
856
+ h.child.emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "done" }], stopReason: "stop" }] });
857
+ h.child.emit({ type: "agent_settled" });
858
+ h.child.exit(0);
859
+ assert.equal((await h.runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED, `${scenario.name}: later finalization remains intact`);
860
+ } else {
861
+ h.runner.cancel(task.id);
862
+ h.child.exit(0);
863
+ assert.equal((await h.runner.waitFor(task.id)).status, TASK_STATUS.CANCELLED, `${scenario.name}: later cancellation remains intact`);
864
+ }
865
+ assert.equal(h.child.disconnects, scenario.expectedDisconnects, `${scenario.name}: later cleanup remains idempotent`);
866
+ }
867
+ });
868
+
776
869
  test("AgentRunner retains permission broker fd3 and assigns messaging IPC to fd4", async () => {
777
870
  const { runner, children, spawnOptions } = harness();
778
871
  const task = runner.run(request({ authorizeParentStandingReviewPermission: () => true }));
779
872
  await tick();
780
873
  const launch = spawnOptions[0];
874
+ const permissionChannelStdio = process.platform === "win32" ? "overlapped" : "pipe";
781
875
  assert.match(launch?.env.GENTLE_PI_AGENTS_OWNED_IPC ?? "", /^\d+-[a-z0-9]+$/, "the owned-IPC marker has the runner's opaque shape");
782
876
  assert.deepEqual(launch?.env, { GENTLE_PI_AGENTS_CHILD: "1", GENTLE_PI_AGENTS_OWNED_IPC: launch?.env.GENTLE_PI_AGENTS_OWNED_IPC, GENTLE_PI_AGENTS_PARENT_PERMISSION_FD: "3" });
783
- assert.deepEqual(launch?.stdio, ["pipe", "pipe", "pipe", "pipe", "ipc"]);
877
+ assert.deepEqual(launch?.stdio, ["pipe", "pipe", "pipe", permissionChannelStdio, "ipc"]);
784
878
  assert.equal(launch?.stdio?.length, 5);
785
879
  children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "channel checked" }], stopReason: "stop" }] });
786
880
  children[0].emit({ type: "agent_settled" });
787
881
  assert.equal((await runner.waitFor(task.id)).status, TASK_STATUS.COMPLETED);
788
882
  });
789
883
 
884
+ test("AgentRunner platform matrix scopes permission fd3 transport", async () => {
885
+ for (const platform of ["win32", "linux", "darwin"] as const) {
886
+ for (const eligible of [false, true]) {
887
+ const launches: Array<Parameters<RunnerDeps["spawn"]>[2]> = [];
888
+ const child = fakeChild();
889
+ const runner = new AgentRunner(new TaskStore(), { maxConcurrency: 1, stallTimeoutMs: 1_000 }, {
890
+ spawn: (_command, _args, options) => {
891
+ launches.push(options);
892
+ return child.child;
893
+ },
894
+ now: () => 1,
895
+ schedule: () => () => {},
896
+ pi: { command: "pi-fixture", args: [] },
897
+ process: { platform, kill: () => {} },
898
+ }, { askUser: async () => ({ cancelled: true }) });
899
+ const task = runner.run(request({ authorizeParentStandingReviewPermission: eligible ? () => true : undefined }));
900
+ await tick();
901
+ const launch = launches[0];
902
+ assert.ok(launch, `${platform} ${eligible ? "eligible" : "ineligible"} child launches`);
903
+ assert.equal(launch.env.GENTLE_PI_AGENTS_PARENT_PERMISSION_FD, eligible ? "3" : undefined, "only eligible children receive the fd3 marker");
904
+ assert.deepEqual(launch.stdio, eligible ? ["pipe", "pipe", "pipe", platform === "win32" ? "overlapped" : "pipe", "ipc"] : ["pipe", "pipe", "pipe", "ipc"]);
905
+ assert.equal(launch.stdio?.indexOf("ipc"), eligible ? 4 : 3, "messaging IPC follows fd3 only for eligible children");
906
+ runner.cancel(task.id);
907
+ }
908
+ }
909
+ });
910
+
790
911
  test("AgentRunner answers dialogs through askUser in task mode and cancels them in background mode", async () => {
791
912
  const { store, runner, children, asks } = harness({ answer: { confirmed: true } });
792
913
  const task = runner.run(request());
@@ -829,8 +950,11 @@ test("AgentRunner has no total-duration watchdog but keeps active work alive and
829
950
  children[0].emit({ type: "response", id: "r1", success: true });
830
951
  await tick();
831
952
  assert.equal(initialStall.cancelled, true, "every child RPC event, including a response, re-arms the inactivity watchdog");
953
+ const afterResponse = timers.filter((timer) => timer.ms === 10_000 && !timer.cancelled).at(-1);
954
+ assert.ok(afterResponse);
832
955
  children[0].emit({ type: "message_update", assistantMessageEvent: { type: "text_delta", delta: "still working" } });
833
956
  await tick();
957
+ assert.equal(afterResponse!.cancelled, true, "normalized task progress re-arms the inactivity watchdog");
834
958
  assert.equal(store.get(task.id)?.status, TASK_STATUS.RUNNING, "ongoing RPC activity keeps a long-running task active");
835
959
  const stall = timers.filter((timer) => timer.ms === 10_000 && !timer.cancelled).at(-1);
836
960
  assert.ok(stall);
@@ -840,6 +964,104 @@ test("AgentRunner has no total-duration watchdog but keeps active work alive and
840
964
  assert.match(store.get(task.id)?.error ?? "", /stalled/);
841
965
  });
842
966
 
967
+ test("an announced tool call in flight arms the tool ceiling and names the tool when it fires", async () => {
968
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
969
+ const task = h.runner.run(request());
970
+ await tick();
971
+ h.children[0].emit({ type: "tool_execution_start", toolCallId: "t1", toolName: "bash", args: { command: "pnpm test" } });
972
+ await tick();
973
+ assert.equal(h.store.get(task.id)?.lastStep, "bash");
974
+ assert.equal(h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).length, 0, "the idle budget no longer bounds a task with a tool in flight");
975
+ const toolTimer = h.timers.filter((timer) => timer.ms === 30 * 60_000 && !timer.cancelled).at(-1);
976
+ assert.ok(toolTimer, "an in-flight tool call arms the tool ceiling");
977
+ toolTimer!.fn();
978
+ await tick();
979
+ assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
980
+ assert.equal(h.store.get(task.id)?.error, 'stalled for 30 min with tool "bash" still running after: bash');
981
+ });
982
+
983
+ test("a finished tool call returns the task to the idle silence budget", async () => {
984
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
985
+ const task = h.runner.run(request());
986
+ await tick();
987
+ h.children[0].emit({ type: "tool_execution_start", toolCallId: "t1", toolName: "bash", args: { command: "pnpm test" } });
988
+ await tick();
989
+ h.children[0].emit({ type: "tool_execution_end", toolCallId: "t1", isError: false });
990
+ await tick();
991
+ assert.equal(h.timers.filter((timer) => timer.ms === 30 * 60_000 && !timer.cancelled).length, 0, "a finished tool is back on the idle budget");
992
+ const idle = h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).at(-1);
993
+ assert.ok(idle, "tool_end re-arms the idle budget");
994
+ idle!.fn();
995
+ await tick();
996
+ assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
997
+ assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: bash");
998
+ });
999
+
1000
+ test("ignored non-dialog UI traffic does not renew the idle silence budget", async () => {
1001
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
1002
+ const task = h.runner.run(request());
1003
+ await tick();
1004
+ const armed = h.timers.at(-1);
1005
+ assert.ok(armed, "launch arms the idle silence budget");
1006
+ assert.equal(armed!.ms, FOUR_MIN_MS);
1007
+ // Fire-and-forget UI notifications normalize to zero task events and prove
1008
+ // only that the transport is alive; they must not postpone the silence bound.
1009
+ h.children[0].emit({ type: "extension_ui_request", id: "u1", method: "setStatus", statusKey: "fixture", statusText: "idle" });
1010
+ h.children[0].emit({ type: "extension_ui_request", id: "u2", method: "notify", message: "still here" });
1011
+ await tick();
1012
+ assert.equal(armed!.cancelled, false, "ignored UI traffic must not cancel the armed silence budget");
1013
+ assert.equal(h.timers.filter((timer) => !timer.cancelled && timer.ms === FOUR_MIN_MS).length, 1, "no replacement timer is scheduled for ignored UI traffic");
1014
+ assert.equal(h.timers.filter((timer) => timer.ms === 30 * 60_000).length, 0, "ignored UI traffic never earns the tool ceiling");
1015
+ armed!.fn();
1016
+ await tick();
1017
+ assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
1018
+ assert.equal(h.store.get(task.id)?.error, "stalled for 4 min after: prompt accepted; no first run event received for model: openai-codex/gpt-5.6-terra");
1019
+ });
1020
+
1021
+ test("an unrecognized RPC object does not renew the idle silence budget", async () => {
1022
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS });
1023
+ const task = h.runner.run(request());
1024
+ await tick();
1025
+ const armed = h.timers.at(-1);
1026
+ assert.ok(armed);
1027
+ h.children[0].emit({ type: "some_future_event", payload: { nested: true } });
1028
+ await tick();
1029
+ assert.equal(armed!.cancelled, false, "an unknown object is not progress");
1030
+ assert.equal(h.timers.filter((timer) => !timer.cancelled).length, 1);
1031
+ armed!.fn();
1032
+ await tick();
1033
+ assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
1034
+ });
1035
+
1036
+ test("a blocking child dialog still re-arms the idle silence budget", async () => {
1037
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS, answer: { confirmed: true } });
1038
+ h.runner.run(request());
1039
+ await tick();
1040
+ const armed = h.timers.at(-1);
1041
+ assert.ok(armed);
1042
+ h.children[0].emit({ type: "extension_ui_request", id: "u1", method: "confirm", title: "Continue?" });
1043
+ await tick();
1044
+ assert.equal(armed!.cancelled, true, "a dialog the parent must answer is meaningful activity");
1045
+ assert.equal(h.asks.length, 1);
1046
+ });
1047
+
1048
+ test("the tool ceiling holds while any announced tool call is still in flight", async () => {
1049
+ const h = harness({ stallTimeoutMs: FOUR_MIN_MS, toolStallTimeoutMs: 30 * 60_000 });
1050
+ h.runner.run(request());
1051
+ await tick();
1052
+ h.children[0].emit({ type: "tool_execution_start", toolCallId: "t1", toolName: "bash", args: { command: "pnpm test" } });
1053
+ await tick();
1054
+ h.children[0].emit({ type: "tool_execution_start", toolCallId: "t2", toolName: "bash", args: { command: "pnpm run typecheck" } });
1055
+ await tick();
1056
+ h.children[0].emit({ type: "tool_execution_end", toolCallId: "t1", isError: false });
1057
+ await tick();
1058
+ assert.ok(h.timers.filter((timer) => timer.ms === 30 * 60_000 && !timer.cancelled).length > 0, "a second tool still in flight keeps the tool ceiling");
1059
+ assert.equal(h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).length, 0);
1060
+ h.children[0].emit({ type: "tool_execution_end", toolCallId: "t2", isError: false });
1061
+ await tick();
1062
+ assert.ok(h.timers.filter((timer) => timer.ms === FOUR_MIN_MS && !timer.cancelled).length > 0, "ending the last tool returns to the idle budget");
1063
+ });
1064
+
843
1065
  test("AgentRunner.cancelAll stops every queued and running task", async () => {
844
1066
  const { store, runner, children } = harness({ maxConcurrency: 1 });
845
1067
  const running = runner.run(request());
@@ -1012,8 +1234,9 @@ test("an unprobeable process group quarantines at its deadline and still records
1012
1234
  const finishes: string[] = [];
1013
1235
  let now = 1_000;
1014
1236
  let launches = 0;
1237
+ let child: FakeChild;
1015
1238
  const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 10_000 }, {
1016
- spawn: () => fakeChild({ exitOnKill: false, pid: 90 + (launches += 1) }).child, now: () => now,
1239
+ spawn: () => (child = fakeChild({ exitOnKill: false, pid: 90 + (launches += 1) })).child, now: () => now,
1017
1240
  schedule: (fn, ms) => {
1018
1241
  const timer = { fn, ms, cancelled: false };
1019
1242
  timers.push(timer);
@@ -1022,7 +1245,7 @@ test("an unprobeable process group quarantines at its deadline and still records
1022
1245
  pi: { command: "pi", args: [] },
1023
1246
  process: { platform: "win32", kill: () => {} },
1024
1247
  }, { askUser: async () => ({ value: "yes" }), onFinish: (task) => { finishes.push(task.id); } });
1025
- const first = runner.run(request());
1248
+ const first = runner.run(managedRequest());
1026
1249
  const second = runner.run(request({ prompt: "queued" }));
1027
1250
  await tick();
1028
1251
  runner.cancel(first.id);
@@ -1039,6 +1262,12 @@ test("an unprobeable process group quarantines at its deadline and still records
1039
1262
  assert.equal(finishes.length, 1, "the run is recorded exactly once");
1040
1263
  assert.equal(store.get(second.id)?.status, TASK_STATUS.QUEUED, "an unconfirmed exit retains its capacity");
1041
1264
  assert.equal(launches, 1, "no further launch happens while the slot is quarantined");
1265
+ assert.throws(() => runner.run(managedRequest()), /Remediation already queued or running/, "a failed record does not release its quarantined child");
1266
+ child!.exit(0);
1267
+ await tick();
1268
+ assert.doesNotThrow(() => runner.run(managedRequest()), "confirmed cleanup releases the managed workspace");
1269
+ runner.cancelAll();
1270
+ child!.exit(0);
1042
1271
  });
1043
1272
 
1044
1273
  test("abortReasonText renders an Error, a string, and nothing for unknown reasons", () => {
@@ -1053,17 +1282,12 @@ test("research narrowing transport keeps exact argv paths and replaces inherited
1053
1282
  const h = harness();
1054
1283
  const selection = { documentation: { tools: ["fetch_content"], extensions: { fetch_content: "/installed/docs tools.ts" } } };
1055
1284
  for (const researchSelection of [selection, undefined]) {
1056
- const artifact = { store: "none" as const, worktree: "/work", changeName: "demo", retainedIntent: "denied questions", locators: [] };
1057
- const expected = structuredClone(artifact);
1058
- const launch = request({ researchSelection, researchArtifact: artifact, extensionPaths: researchSelection ? ["/installed/docs tools.ts"] : [],
1059
- env: { PATH: "/bin", GENTLE_PI_RESEARCH_SELECTION: "stale broad selection", GENTLE_PI_RESEARCH_ARTIFACT: "stale broader scope" } });
1285
+ const launch = request({ researchSelection, extensionPaths: researchSelection ? ["/installed/docs tools.ts"] : [],
1286
+ env: { PATH: "/bin", GENTLE_PI_RESEARCH_SELECTION: "stale broad selection" } });
1060
1287
  const argv = childArguments(launch);
1061
1288
  assert.deepEqual(argv.filter((_, i) => argv[i - 1] === "--extension"), launch.extensionPaths);
1062
1289
  const task = h.runner.run(launch);
1063
- artifact.worktree = "/wrong";
1064
1290
  await tick();
1065
- assert.deepEqual(JSON.parse(h.spawnOptions.at(-1)!.env.GENTLE_PI_RESEARCH_ARTIFACT!), expected);
1066
- assert.deepEqual("researchArtifact" in task ? task.researchArtifact : undefined, expected);
1067
1291
  assert.deepEqual(JSON.parse(h.spawnOptions.at(-1)!.env.GENTLE_PI_RESEARCH_SELECTION!), researchSelection ?? null);
1068
1292
  assert.equal(h.spawnOptions.at(-1)!.env.PATH, "/bin");
1069
1293
  h.runner.cancel(task.id);
@@ -1071,55 +1295,80 @@ test("research narrowing transport keeps exact argv paths and replaces inherited
1071
1295
  }
1072
1296
  });
1073
1297
 
1298
+ function managedRequest(cwd = "/repo"): TaskRequest {
1299
+ return request({ agent: { ...explorer, name: "sdd-remediate" }, cwd, sddRemediation: {
1300
+ failedEvidenceRevision: "failed-revision",
1301
+ plan: { cwd, commands: ["pnpm test"], runtimeHarness: { naReason: "Not applicable because this tests runner admission." }, rollback: { boundary: "fixture", command: "git diff --check" } },
1302
+ scope: { cwd, editPaths: [], commands: ["pnpm test", "git diff --check"], allowedEditRoots: [cwd] },
1303
+ } });
1304
+ }
1074
1305
 
1075
- test("runner retains remediation observations and awaits terminal settlement", async () => {
1076
- const h = harness({ pid: 123 });
1077
- let finalized: { record: TaskRecord; facts: RemediationTerminalFacts } | undefined;
1078
- let release: () => void;
1079
- const gate = new Promise<void>(resolve => { release = resolve; });
1080
- const remediationPlan: RemediationPlan = { cwd: "/repo", commands: ["pnpm test"], runtimeHarness: { naReason: "Not applicable because the fixture has no runtime boundary." }, rollback: { boundary: "Revert fixture", command: "git diff --check" } };
1081
- const remediation: RemediationTaskState = { failedEvidenceRevision: `sha256:${"a".repeat(64)}`, plan: remediationPlan, pending: {}, observations: [], invalid: false, token: "opaque", acquire: { workspaceRoot: "/repo", changeName: "demo", requestId: "fixture", workUnit: "correct", evidenceGoal: "Observed correction" } };
1082
- const task = h.runner.run(request({ sddRemediation: remediation, finalizeRemediation: async (record, facts) => { finalized = { record, facts }; await gate; } }));
1083
- await tick();
1084
- for (const command of ["pnpm test", "git diff --check"]) {
1085
- h.children[0].emit({ type: "tool_execution_start", toolName: "bash", toolCallId: command, args: { command } });
1086
- h.children[0].emit({ type: "tool_execution_end", toolName: "bash", toolCallId: command, isError: false, result: { content: [{ type: "text", text: "command output" }], details: { remediationCommand: { command, toolCallId: command, cwd: "/repo", exitCode: 0 } } } });
1087
- }
1088
- h.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "done" }], stopReason: "stop" }] });
1089
- h.children[0].emit({ type: "agent_settled" });
1090
- await tick();
1091
- assert.ok(finalized);
1092
- assert.equal(finalized.record.sddRemediation?.observations.length, 2);
1093
- const prompt = h.children[0].written.find(value => value.type === "prompt").message;
1094
- assert.ok(typeof prompt === "string");
1095
- assert.match(prompt, /pnpm test/);
1096
- assert.doesNotMatch(prompt, /opaque/);
1097
- assert.equal(finalized.facts.cleanupConfirmed, true);
1098
- assert.equal(h.finishes.length, 0);
1099
- release(); await h.runner.waitFor(task.id);
1100
- assert.equal(h.finishes.length, 1);
1101
- assert.equal(remediation.observations.length, 0);
1306
+ for (const queued of [true, false]) test(`managed exclusion covers ${queued ? "queued" : "running"} same-workspace actors`, async () => {
1307
+ const h = harness({ pid: 123, process: { platform: "win32", kill() {} } });
1308
+ const first = h.runner.run(managedRequest());
1309
+ if (!queued) await tick();
1310
+ try {
1311
+ assert.throws(() => h.runner.run(managedRequest()), /Remediation already queued or running/);
1312
+ assert.equal(h.store.list().length, 1, "rejection creates no task or queue entry");
1313
+ await tick();
1314
+ assert.equal(h.children.length, 1);
1315
+ assert.equal(h.store.get(first.id)?.status, TASK_STATUS.RUNNING);
1316
+ } finally { h.runner.cancelAll(); await tick(); }
1102
1317
  });
1103
1318
 
1319
+ test("managed exclusion does not serialize other workspaces or ordinary tasks", async () => {
1320
+ const h = harness({ maxConcurrency: 3, pid: 123, process: { platform: "win32", kill() {} } });
1321
+ h.runner.run(managedRequest());
1322
+ h.runner.run(managedRequest("/other"));
1323
+ h.runner.run(request());
1324
+ await tick();
1325
+ assert.equal(h.children.length, 3);
1326
+ h.runner.cancelAll();
1327
+ await tick();
1328
+ });
1104
1329
 
1105
- test("admitted no-PID failure settles interrupted before sending a prompt", async () => {
1106
- const h = harness(); let facts;
1107
- const task = h.runner.run(request({ sddRemediation: { plan: {} } as unknown as RemediationTaskState, finalizeRemediation: async (_task, observed) => { facts = observed; } }));
1330
+ for (const ending of ["complete", "failure", "cancel", "queued-cancel"] as const) test(`managed exclusion releases after ${ending}`, async () => {
1331
+ const h = harness({ pid: 123, process: { platform: "win32", kill() {} } });
1332
+ const first = h.runner.run(managedRequest());
1333
+ if (ending === "queued-cancel") h.runner.cancel(first.id);
1334
+ else {
1335
+ await tick();
1336
+ if (ending === "complete") {
1337
+ h.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "Finished" }], stopReason: "stop" }] });
1338
+ h.children[0].emit({ type: "agent_settled" });
1339
+ } else if (ending === "failure") h.children[0].exit(1);
1340
+ else h.runner.cancel(first.id);
1341
+ }
1342
+ await h.runner.waitFor(first.id);
1343
+ const next = h.runner.run(managedRequest());
1344
+ await tick();
1345
+ assert.equal(h.store.get(next.id)?.status, TASK_STATUS.RUNNING);
1346
+ h.runner.cancelAll();
1108
1347
  await tick();
1109
- assert.equal(h.children[0].written.some(value => value.type === "prompt"), false);
1110
- await h.runner.waitFor(task.id);
1111
- assert.equal(facts.spawned, false);
1112
- assert.equal(facts.exited, false);
1113
- assert.equal(facts.cleanupConfirmed, false);
1114
1348
  });
1115
1349
 
1350
+ test("managed exclusion lasts until child cleanup is confirmed", async () => {
1351
+ const h = harness({ pid: 123, exitOnKill: false, process: { platform: "win32", kill() {} } });
1352
+ const first = h.runner.run(managedRequest());
1353
+ await tick();
1354
+ h.runner.cancel(first.id);
1355
+ try { assert.throws(() => h.runner.run(managedRequest()), /Remediation already queued or running/); }
1356
+ finally { h.children[0].exit(0); }
1357
+ await h.runner.waitFor(first.id);
1358
+ const next = h.runner.run(managedRequest());
1359
+ await tick();
1360
+ assert.equal(h.store.get(next.id)?.status, TASK_STATUS.RUNNING);
1361
+ h.runner.cancelAll();
1362
+ for (const child of h.children) child.exit(0);
1363
+ await tick();
1364
+ });
1116
1365
 
1117
- test("admitted synchronous spawn failure finalizes without launching another actor", async () => {
1118
- let spawns = 0, facts;
1119
- const store = new TaskStore();
1120
- const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 100 }, { spawn: () => { spawns++; throw new Error("spawn refused"); }, pi: { command: "pi", args: [] }, now: () => 1, schedule: () => () => {} }, { askUser: async () => ({}) });
1121
- const task = runner.run(request({ sddRemediation: { plan: {} } as unknown as RemediationTaskState, finalizeRemediation: async (_task, value) => { facts = value; } }));
1122
- await runner.waitFor(task.id);
1123
- assert.deepEqual(facts, { spawned: false, exited: false, cleanupConfirmed: true });
1124
- assert.equal(spawns, 1);
1366
+ test("managed exclusion releases failed startup and ignores historical-only tasks", async () => {
1367
+ const h = harness({ failStart: true });
1368
+ const first = h.runner.run(managedRequest());
1369
+ assert.equal((await h.runner.waitFor(first.id)).status, TASK_STATUS.FAILED);
1370
+ h.store.add({ ...h.store.get(first.id)!, id: "historical-only", status: TASK_STATUS.RUNNING });
1371
+ const next = h.runner.run(managedRequest());
1372
+ assert.equal((await h.runner.waitFor(next.id)).status, TASK_STATUS.FAILED, "the next actor reaches spawn, not a historical admission lock");
1373
+ assert.match(h.store.get(next.id)?.error ?? "", /fixture spawn failed/);
1125
1374
  });