gentle-pi 3.7.0 → 4.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (316) hide show
  1. package/README.md +37 -6
  2. package/assets/agents/gentle-ai-explore.md +4 -4
  3. package/assets/agents/gentle-ai-verify.md +6 -4
  4. package/assets/agents/gentle-ai-worker.md +7 -9
  5. package/assets/orchestrator-delegation.md +40 -41
  6. package/assets/orchestrator-memory.md +1 -22
  7. package/assets/orchestrator-skills.md +1 -1
  8. package/assets/orchestrator.md +12 -24
  9. package/assets/support/strict-tdd-verify.md +4 -266
  10. package/assets/support/strict-tdd.md +8 -360
  11. package/bin/gentle-shell.mjs +254 -29
  12. package/docs/delegated-verification.md +26 -1
  13. package/docs/gentle-agents-activity.md +24 -0
  14. package/docs/gentle-shell.md +92 -28
  15. package/docs/native-authority-architecture.md +2 -2
  16. package/docs/prompt-history.md +280 -0
  17. package/docs/readme-reference.md +112 -206
  18. package/docs/telemetry.md +1 -1
  19. package/docs/yolo-mode.md +86 -0
  20. package/extensions/child-context.ts +26 -0
  21. package/extensions/child-safety.ts +23 -0
  22. package/extensions/gentle-agents.ts +429 -262
  23. package/extensions/gentle-ai.ts +781 -683
  24. package/extensions/gentle-shell.ts +1375 -88
  25. package/extensions/gentle-stats.ts +101 -0
  26. package/extensions/gentle-todo.ts +18 -7
  27. package/extensions/history/atomic-write.ts +38 -0
  28. package/extensions/history/hide-prompts.ts +183 -0
  29. package/extensions/history/index.ts +1419 -0
  30. package/extensions/history/load-shared-history.ts +39 -0
  31. package/extensions/history/selector-helpers.ts +538 -0
  32. package/extensions/history/session-scan.ts +233 -0
  33. package/extensions/history/store.ts +1119 -0
  34. package/extensions/nan-provider.ts +6 -0
  35. package/extensions/quiet-tools.ts +179 -87
  36. package/extensions/resume-hint.ts +60 -0
  37. package/extensions/skill-registry.ts +16 -12
  38. package/extensions/startup-banner.ts +60 -31
  39. package/lib/agent-assets.ts +604 -0
  40. package/lib/agent-profile-pin.ts +12 -0
  41. package/lib/agents-message-delivery.ts +181 -0
  42. package/lib/agents-protocol.ts +60 -0
  43. package/lib/agents-runner.ts +96 -98
  44. package/lib/agents-view.ts +26 -3
  45. package/lib/agents-widget.ts +16 -9
  46. package/lib/append-system-prompt.ts +21 -0
  47. package/lib/bounded-writer-admission.ts +147 -0
  48. package/lib/card-style-policy.ts +60 -0
  49. package/lib/child-context-files.ts +166 -0
  50. package/lib/codemode-renderer.ts +185 -0
  51. package/lib/command-palette-catalog.ts +3 -9
  52. package/lib/command-palette.ts +25 -14
  53. package/lib/destructive-command-guard.ts +144 -0
  54. package/lib/gentle-ai-elapsed-store.ts +87 -0
  55. package/lib/gentle-ai-renderer.ts +196 -39
  56. package/lib/gentle-shell-launcher.ts +24 -13
  57. package/lib/gentle-shell-resume-hint.ts +176 -0
  58. package/lib/history-capture-policy.ts +95 -0
  59. package/lib/model-routing-authority.ts +5 -1
  60. package/lib/nan-provider.ts +227 -0
  61. package/lib/native-review-cli.ts +49 -108
  62. package/lib/odd-phase-inference.ts +231 -0
  63. package/lib/odd-phase.ts +141 -0
  64. package/lib/overlay-repaint.ts +26 -0
  65. package/lib/pi-tui-keys.ts +53 -0
  66. package/lib/review-candidate-view-owner.ts +67 -17
  67. package/lib/review-candidate-view.ts +112 -25
  68. package/lib/review-reminder-receipt.ts +48 -8
  69. package/lib/review-risk-assessment.ts +156 -11
  70. package/lib/review-sidebar-state.ts +223 -0
  71. package/lib/selection-engine.ts +515 -0
  72. package/lib/session-messaging-grants.ts +135 -0
  73. package/lib/session-worktree-registry.ts +14 -2
  74. package/lib/shell-bar.ts +179 -24
  75. package/lib/shell-card.ts +287 -18
  76. package/lib/shell-changes-view.ts +2 -1
  77. package/lib/shell-prompt.ts +102 -6
  78. package/lib/shell-sidebar-layout.ts +78 -14
  79. package/lib/shell-sidebar.ts +15 -1
  80. package/lib/shell-todo.ts +23 -13
  81. package/lib/shell-usage-view.ts +9 -4
  82. package/lib/shell-usage.ts +66 -10
  83. package/lib/stats-collector.ts +381 -0
  84. package/lib/stats-view.ts +431 -0
  85. package/lib/theme-customization.ts +52 -0
  86. package/lib/vim-editor-adapter.ts +379 -0
  87. package/lib/vim-normal-engine.ts +154 -0
  88. package/lib/vim-operator-engine.ts +416 -0
  89. package/lib/vim-policy.ts +49 -0
  90. package/lib/vim-visual-engine.ts +107 -0
  91. package/lib/visual-customization-policy.ts +108 -0
  92. package/lib/visual-customize-view.ts +330 -0
  93. package/lib/visual-profiles.ts +228 -0
  94. package/lib/yolo-session-policy.ts +240 -0
  95. package/package.json +20 -8
  96. package/runtime/gentle-shell-launcher.mjs +23 -12
  97. package/runtime/gentle-shell-resume-hint.mjs +177 -0
  98. package/runtime/native-review-cli.mjs +49 -108
  99. package/runtime/review-risk-assessment.mjs +154 -9
  100. package/scripts/build-runtime-modules.mjs +1 -0
  101. package/scripts/gentle-ai-installer.mjs +14 -13
  102. package/scripts/mirror-odd-routing.mjs +2 -2
  103. package/scripts/run-test-suite.mjs +76 -0
  104. package/scripts/test-packed-runner.mjs +31 -14
  105. package/scripts/verify-package-files.mjs +11 -22
  106. package/skills/branch-pr/SKILL.md +24 -52
  107. package/skills/chained-pr/SKILL.md +31 -15
  108. package/skills/chained-pr/references/chaining-details.md +31 -20
  109. package/skills/gentle-ai/SKILL.md +9 -15
  110. package/skills/issue-creation/SKILL.md +8 -2
  111. package/skills/issue-creation/references/delegated-workflow-actions.md +19 -0
  112. package/skills/work-unit-commits/SKILL.md +4 -3
  113. package/tests/agent-profiles.test.ts +18 -0
  114. package/tests/agents-fake-child.ts +2 -2
  115. package/tests/agents-message-delivery.test.ts +106 -0
  116. package/tests/agents-protocol.test.ts +40 -0
  117. package/tests/agents-runner.test.ts +378 -89
  118. package/tests/agents-view-thread-identity.test.ts +169 -0
  119. package/tests/agents-view.test.ts +8 -2
  120. package/tests/agents-widget.test.ts +154 -15
  121. package/tests/append-system-prompt-route.test.ts +160 -0
  122. package/tests/append-system-prompt.test.ts +46 -0
  123. package/tests/artifact-language.test.ts +19 -213
  124. package/tests/ask-user-question.test.ts +44 -1
  125. package/tests/asset-installation-runtime.test.ts +5 -16
  126. package/tests/autonomous-guard.test.ts +69 -1
  127. package/tests/bounded-writer-admission.test.ts +95 -0
  128. package/tests/branch-pr-skill.test.ts +43 -0
  129. package/tests/card-style-policy.test.ts +55 -0
  130. package/tests/chained-pr-skill.test.ts +124 -0
  131. package/tests/child-context-files.test.ts +255 -0
  132. package/tests/child-safety.test.ts +82 -0
  133. package/tests/codemode-rendering.test.ts +491 -0
  134. package/tests/command-palette.test.ts +39 -3
  135. package/tests/delegated-key-learnings-contract.test.ts +0 -76
  136. package/tests/destructive-command-guard.test.ts +84 -0
  137. package/tests/devbinary/native-review-parity.devtest.ts +170 -2
  138. package/tests/devbinary/non-git-subagent-bootstrap.devtest.ts +193 -0
  139. package/tests/fixtures/stats/sessions/--work-alpha--/2026-09-28T10-00-00-000Z_aaa.jsonl +7 -0
  140. package/tests/fixtures/stats/sessions/--work-alpha--/2026-09-29T23-00-00-000Z_bbb.jsonl +3 -0
  141. package/tests/fixtures/stats/sessions/--work-alpha--/2026-09-30T08-00-00-000Z_ddd.jsonl +2 -0
  142. package/tests/fixtures/stats/sessions/--work-alpha--/2026-09-30T09-00-00-000Z_eee.jsonl +3 -0
  143. package/tests/fixtures/stats/sessions/--work-alpha--/run-1/session.jsonl +2 -0
  144. package/tests/fixtures/stats/sessions/--work-beta--/2026-09-01T12-00-00-000Z_ccc.jsonl +2 -0
  145. package/tests/fixtures/stats/user-pi/sessions/--work-alpha--/2026-09-28T10-00-00-000Z_aaa.jsonl +2 -0
  146. package/tests/fixtures/stats/user-pi/sessions/--work-alpha--/2026-09-29T23-00-00-000Z_bbb.jsonl +4 -0
  147. package/tests/fixtures/stats/user-pi/sessions/--work-gamma--/2026-09-20T09-00-00-000Z_fff.jsonl +3 -0
  148. package/tests/generic-agent-tools.test.ts +54 -0
  149. package/tests/gentle-agents.test.ts +1047 -252
  150. package/tests/gentle-ai-binary.test.ts +3 -3
  151. package/tests/gentle-ai-elapsed-store.test.ts +68 -0
  152. package/tests/gentle-ai-installer.test.ts +68 -54
  153. package/tests/gentle-ai-renderer.test.ts +487 -8
  154. package/tests/gentle-ai.test.ts +288 -65
  155. package/tests/gentle-card-text.ts +2 -1
  156. package/tests/gentle-shell-bin.test.ts +651 -114
  157. package/tests/gentle-shell-launcher.test.ts +99 -52
  158. package/tests/gentle-shell-resume-hint.test.ts +270 -0
  159. package/tests/gentle-shell.test.ts +3839 -187
  160. package/tests/gentle-stats.test.ts +152 -0
  161. package/tests/gentle-theme.test.ts +4 -1
  162. package/tests/gentle-todo.test.ts +80 -8
  163. package/tests/history-atomic-write.test.ts +57 -0
  164. package/tests/history-capture-policy.test.ts +102 -0
  165. package/tests/history-command-registration.test.ts +164 -0
  166. package/tests/history-dedupe-entries.test.ts +123 -0
  167. package/tests/history-delete-backfill.test.ts +190 -0
  168. package/tests/history-delete-confirm.test.ts +460 -0
  169. package/tests/history-dispatch.test.ts +180 -0
  170. package/tests/history-drain-hidden.test.ts +110 -0
  171. package/tests/history-drain-order.test.ts +98 -0
  172. package/tests/history-expanded-globals.test.ts +62 -0
  173. package/tests/history-gc.test.ts +832 -0
  174. package/tests/history-header-layout.test.ts +265 -0
  175. package/tests/history-hide-prompts.test.ts +275 -0
  176. package/tests/history-lazy-windowing.test.ts +508 -0
  177. package/tests/history-legacy-migrate-v2.test.ts +297 -0
  178. package/tests/history-load-shared-history.test.ts +53 -0
  179. package/tests/history-max-results-cap.test.ts +76 -0
  180. package/tests/history-multi-reader.test.ts +203 -0
  181. package/tests/history-off-path.test.ts +170 -0
  182. package/tests/history-openflow-integration.test.ts +173 -0
  183. package/tests/history-overlay-margin.test.ts +326 -0
  184. package/tests/history-preview-layout.test.ts +93 -0
  185. package/tests/history-registry.test.ts +143 -0
  186. package/tests/history-scope-delete.test.ts +411 -0
  187. package/tests/history-search-caret-keys.test.ts +142 -0
  188. package/tests/history-seed-bootstrap.test.ts +170 -0
  189. package/tests/history-seed-regen.test.ts +129 -0
  190. package/tests/history-selector-windowing.test.ts +94 -0
  191. package/tests/history-session-scan-directory.test.ts +87 -0
  192. package/tests/history-session-scan-extract.test.ts +583 -0
  193. package/tests/history-session-writer.test.ts +351 -0
  194. package/tests/history-store-paths.test.ts +79 -0
  195. package/tests/history-tombstone-exact.test.ts +139 -0
  196. package/tests/history-wheel-mouse.test.ts +242 -0
  197. package/tests/inprocess-reviewer.test.ts +29 -19
  198. package/tests/issue-creation-skill.test.ts +61 -0
  199. package/tests/model-routing-authority.test.ts +16 -0
  200. package/tests/nan-provider.test.ts +471 -0
  201. package/tests/native-review-capability-contract.test.ts +7 -1
  202. package/tests/native-review-cli.test.ts +6 -120
  203. package/tests/native-review-parity-runtime.test.ts +100 -3
  204. package/tests/odd-integration.test.ts +67 -0
  205. package/tests/odd-phase-inference.test.ts +213 -0
  206. package/tests/odd-phase-loader.test.ts +253 -0
  207. package/tests/odd-phase.test.ts +307 -0
  208. package/tests/odd-routing-canonical-ratchet.test.ts +11 -6
  209. package/tests/odd-routing-contract.test.ts +86 -35
  210. package/tests/orchestrator-budget.test.ts +14 -39
  211. package/tests/orchestrator-rdd-ownership.test.ts +3 -3
  212. package/tests/overlay-repaint.test.ts +74 -0
  213. package/tests/package-manifest.test.ts +252 -115
  214. package/tests/packed-runner-owned-path.test.ts +46 -0
  215. package/tests/persona-single-channel.test.ts +6 -6
  216. package/tests/provider-defect-handoff.test.ts +3 -11
  217. package/tests/quiet-bash-runtime.test.ts +76 -0
  218. package/tests/quiet-tool-rendering.test.ts +409 -184
  219. package/tests/rdd-aware-verification-contract.test.ts +76 -1
  220. package/tests/rdd-status-line.test.ts +9 -4
  221. package/tests/resume-hint-extension.test.ts +122 -0
  222. package/tests/review-agent-end-preflight.test.ts +176 -12
  223. package/tests/review-candidate-owner-retry.test.ts +22 -1
  224. package/tests/review-candidate-view.test.ts +298 -0
  225. package/tests/review-contract-prompt.test.ts +108 -43
  226. package/tests/review-controller-lock-status.test.ts +0 -1
  227. package/tests/review-controller-native-routing.test.ts +611 -5
  228. package/tests/review-controller-workspace-root.test.ts +163 -4
  229. package/tests/review-host-relay-routing.test.ts +338 -2
  230. package/tests/review-integration-v2-forward.test.ts +200 -0
  231. package/tests/review-ledger-contract.test.ts +10 -34
  232. package/tests/review-reminder-receipt.test.ts +47 -1
  233. package/tests/review-risk-assessment.test.ts +498 -6
  234. package/tests/review-sidebar-state.test.ts +402 -0
  235. package/tests/run-test-suite.test.ts +124 -0
  236. package/tests/runtime-harness.mjs +145 -786
  237. package/tests/runtime-metrics-children.test.ts +16 -23
  238. package/tests/selection-engine.test.ts +421 -0
  239. package/tests/session-messaging-grants.test.ts +255 -0
  240. package/tests/session-worktree-registry.test.ts +77 -0
  241. package/tests/shell-bar.test.ts +382 -1
  242. package/tests/shell-card.test.ts +353 -1
  243. package/tests/shell-changes-view.test.ts +52 -0
  244. package/tests/shell-prompt.test.ts +94 -2
  245. package/tests/shell-sidebar-layout.test.ts +325 -21
  246. package/tests/shell-sidebar-scroll-benchmark.test.ts +255 -0
  247. package/tests/shell-todo.test.ts +87 -1
  248. package/tests/shell-usage-view.test.ts +27 -0
  249. package/tests/shell-usage.test.ts +73 -0
  250. package/tests/skill-registry.test.ts +50 -1
  251. package/tests/startup-banner.test.ts +130 -2
  252. package/tests/stats-collector.test.ts +195 -0
  253. package/tests/stats-view.test.ts +202 -0
  254. package/tests/telemetry-trigger.test.ts +81 -20
  255. package/tests/theme-customization.test.ts +72 -0
  256. package/tests/vim-editor-adapter-host-resolution.test.ts +37 -0
  257. package/tests/vim-editor-adapter.test.ts +804 -0
  258. package/tests/vim-normal-engine.test.ts +101 -0
  259. package/tests/vim-operator-engine.test.ts +215 -0
  260. package/tests/vim-policy.test.ts +19 -0
  261. package/tests/vim-visual-engine.test.ts +52 -0
  262. package/tests/visual-customization-policy.test.ts +110 -0
  263. package/tests/visual-customize-view.test.ts +418 -0
  264. package/tests/visual-profiles.test.ts +87 -0
  265. package/tests/yolo-customize.test.ts +256 -0
  266. package/tests/yolo-mode-runtime.test.ts +161 -0
  267. package/tests/yolo-mode.test.ts +261 -0
  268. package/tests/yolo-session-policy.test.ts +59 -0
  269. package/themes/Gentle.json +2 -1
  270. package/themes/Gentleman-Cute.json +2 -1
  271. package/themes/Gentleman-Sexy.json +2 -1
  272. package/assets/agents/sdd-apply.md +0 -159
  273. package/assets/agents/sdd-archive.md +0 -228
  274. package/assets/agents/sdd-design.md +0 -49
  275. package/assets/agents/sdd-explore.md +0 -48
  276. package/assets/agents/sdd-init.md +0 -56
  277. package/assets/agents/sdd-onboard.md +0 -52
  278. package/assets/agents/sdd-proposal.md +0 -64
  279. package/assets/agents/sdd-remediate.md +0 -37
  280. package/assets/agents/sdd-research.md +0 -49
  281. package/assets/agents/sdd-spec.md +0 -192
  282. package/assets/agents/sdd-status.md +0 -54
  283. package/assets/agents/sdd-tasks.md +0 -108
  284. package/assets/agents/sdd-verify.md +0 -124
  285. package/assets/chains/sdd-full.chain.md +0 -83
  286. package/assets/chains/sdd-plan.chain.md +0 -56
  287. package/assets/chains/sdd-verify.chain.md +0 -43
  288. package/assets/sdd-orchestrator-workflow.md +0 -319
  289. package/assets/support/sdd-status-contract.md +0 -77
  290. package/docs/assets/diagrams/sdd-cycle.svg +0 -14
  291. package/extensions/sdd-init.ts +0 -816
  292. package/lib/openspec-deltas.ts +0 -156
  293. package/lib/sdd-preflight.ts +0 -1066
  294. package/lib/sdd-research-capabilities.ts +0 -94
  295. package/lib/sdd-status.ts +0 -26
  296. package/tests/fixtures/legacy/sdd-research-v2.5.0.md +0 -54
  297. package/tests/fixtures/native-review-cli/v2.1.3/bind-sdd.json +0 -25
  298. package/tests/fixtures/v0.10.7/assets/agents/sdd-apply.md +0 -132
  299. package/tests/openspec-deltas.test.ts +0 -209
  300. package/tests/sdd-agent-tools.test.ts +0 -156
  301. package/tests/sdd-archive-replay.test.ts +0 -82
  302. package/tests/sdd-classical-continuation.test.ts +0 -74
  303. package/tests/sdd-execution-routing-contract.test.ts +0 -44
  304. package/tests/sdd-managed-runtime-settlement.test.ts +0 -155
  305. package/tests/sdd-native-managed-uptake.test.ts +0 -243
  306. package/tests/sdd-no-attempts-contract.test.ts +0 -15
  307. package/tests/sdd-odd-integration.test.ts +0 -33
  308. package/tests/sdd-optional-research.test.ts +0 -124
  309. package/tests/sdd-planning-routing-contract.test.ts +0 -45
  310. package/tests/sdd-preflight-rpc-input.test.ts +0 -125
  311. package/tests/sdd-preflight.test.ts +0 -541
  312. package/tests/sdd-research-capabilities.test.ts +0 -114
  313. package/tests/sdd-research-live.test.ts +0 -241
  314. package/tests/sdd-selection-transport.test.ts +0 -653
  315. package/tests/sdd-status.test.ts +0 -9
  316. package/tests/sdd-task-truth.test.ts +0 -43
@@ -1,5 +1,8 @@
1
1
  import assert from "node:assert/strict";
2
2
  import test from "node:test";
3
+ import fs, { existsSync, readFileSync, statSync } from "node:fs";
4
+ import { syncBuiltinESMExports } from "node:module";
5
+ import { basename, dirname } from "node:path";
3
6
  import { PassThrough } from "node:stream";
4
7
  import { AGENT_MODE, parseAgentsConfig, resolveAgentProfile, type AgentDefinition } from "../lib/agents-config.ts";
5
8
  import { TASK_STATUS, TaskStore, type TaskRecord } from "../lib/agents-protocol.ts";
@@ -24,21 +27,24 @@ interface Harness {
24
27
  timers: Array<{ fn: () => void; ms: number; cancelled: boolean }>;
25
28
  asks: Array<{ taskId: string; method: string }>;
26
29
  finishes: string[];
27
- spawnOptions: Array<{ env: NodeJS.ProcessEnv; stdio?: string[] }>;
30
+ spawnOptions: Array<{ command: string; args: string[]; env: NodeJS.ProcessEnv; stdio?: string[] }>;
31
+ advance(ms: number): void;
28
32
  }
29
33
 
30
- function harness(options: { failStart?: boolean; process?: RunnerDeps["process"]; pid?: number; maxConcurrency?: number; stallTimeoutMs?: number; toolStallTimeoutMs?: number; answer?: Record<string, unknown>; exitOnKill?: boolean; state?: Record<string, unknown>; stateSuccess?: boolean; onNotification?: RunnerHooks["onNotification"]; onSuccessfulMutation?: RunnerHooks["onSuccessfulMutation"]; onFinish?: RunnerHooks["onFinish"] } = {}): Harness {
34
+ function harness(options: { resolvePi?: RunnerDeps["resolvePi"]; failStart?: boolean; process?: RunnerDeps["process"]; pid?: number; maxConcurrency?: number; stallTimeoutMs?: number; toolStallTimeoutMs?: number; answer?: Record<string, unknown>; exitOnKill?: boolean; state?: Record<string, unknown>; stateSuccess?: boolean; onNotification?: RunnerHooks["onNotification"]; onSuccessfulMutation?: RunnerHooks["onSuccessfulMutation"]; onFinish?: RunnerHooks["onFinish"] } = {}): Harness {
31
35
  const children: FakeChild[] = [];
32
36
  const timers: Harness["timers"] = [];
33
37
  const asks: Harness["asks"] = [];
34
38
  const finishes: string[] = [];
35
39
  const spawnOptions: Harness["spawnOptions"] = [];
36
40
  let clock = 1000;
41
+ const deadlines = new Map<Harness["timers"][number], number>();
37
42
  const deps: RunnerDeps = {
38
43
  process: options.process,
39
- spawn: (_command, _args, launchOptions) => {
44
+ resolvePi: options.resolvePi,
45
+ spawn: (command, args, launchOptions) => {
40
46
  if (options.failStart) throw new Error("fixture spawn failed");
41
- spawnOptions.push({ env: launchOptions.env, stdio: launchOptions.stdio });
47
+ spawnOptions.push({ command, args, env: launchOptions.env, stdio: launchOptions.stdio });
42
48
  const fake = fakeChild({ exitOnKill: options.exitOnKill, pid: options.pid });
43
49
  if (options.state !== undefined) {
44
50
  fake.child.stdin.removeAllListeners("data");
@@ -56,6 +62,7 @@ function harness(options: { failStart?: boolean; process?: RunnerDeps["process"]
56
62
  schedule: (fn, ms) => {
57
63
  const timer = { fn, ms, cancelled: false };
58
64
  timers.push(timer);
65
+ deadlines.set(timer, clock + ms);
59
66
  return () => {
60
67
  timer.cancelled = true;
61
68
  };
@@ -72,13 +79,111 @@ function harness(options: { failStart?: boolean; process?: RunnerDeps["process"]
72
79
  onNotification: options.onNotification,
73
80
  onSuccessfulMutation: options.onSuccessfulMutation,
74
81
  });
75
- return { store, runner, children, timers, asks, finishes, spawnOptions };
82
+ return { store, runner, children, timers, asks, finishes, spawnOptions, advance(ms) {
83
+ clock += ms;
84
+ for (const timer of timers) {
85
+ if (!timer.cancelled && deadlines.get(timer)! <= clock) {
86
+ timer.cancelled = true;
87
+ timer.fn();
88
+ }
89
+ }
90
+ } };
76
91
  }
77
92
 
78
93
  const tick = () => new Promise((resolve) => setImmediate(resolve));
79
94
 
80
95
  const FOUR_MIN_MS = 4 * 60_000;
81
96
 
97
+ function argumentUpdate(type: string, fields: Record<string, unknown> = {}): Record<string, unknown> {
98
+ return { type: "message_update", usage: { totalTokens: 999, cost: { total: 99 } }, assistantMessageEvent: { type, contentIndex: 0, ...fields } };
99
+ }
100
+
101
+ function beginArguments(child: FakeChild, timestamp = 1000): void {
102
+ child.emit({ type: "message_start", message: { role: "assistant", timestamp, content: [] } });
103
+ child.emit(argumentUpdate("toolcall_start", { id: "call-1", toolName: "write" }));
104
+ }
105
+
106
+ test("fresh argument streaming renews idle liveness without execution or provisional usage", async () => {
107
+ const h = harness({ stallTimeoutMs: 100, toolStallTimeoutMs: 1000 });
108
+ const task = h.runner.run(request());
109
+ await tick();
110
+ const child = h.children[0];
111
+ beginArguments(child);
112
+ // Each chunk arrives before the current idle deadline. Four renewals allow
113
+ // generation to outlast the original budget; only the latest timer can fire.
114
+ for (const delta of ['{"path":', '"private-path",', '"content":', '"private-arguments"}']) {
115
+ h.advance(80);
116
+ assert.equal(h.store.get(task.id)?.status, TASK_STATUS.RUNNING);
117
+ const before = h.timers.filter(timer => !timer.cancelled).at(-1)!;
118
+ child.emit(argumentUpdate("toolcall_delta", { delta }));
119
+ assert.equal(before.cancelled, true, "fresh argument data cancels the prior idle deadline");
120
+ assert.equal(h.timers.filter(timer => !timer.cancelled).at(-1)?.ms, 100);
121
+ }
122
+ const current = h.store.get(task.id)!;
123
+ assert.equal(current.toolCalls, 0);
124
+ assert.equal(current.tokens, 0);
125
+ assert.equal(current.cost, 0);
126
+ assert.equal(current.lastStep, "generating tool arguments");
127
+ assert.doesNotMatch(JSON.stringify(h.store.thread(task.id)), /private/);
128
+ h.advance(101);
129
+ await tick();
130
+ assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT, "later silence still times out");
131
+ assert.doesNotMatch(h.store.get(task.id)?.error ?? "", /private/);
132
+ });
133
+
134
+ test("empty, replayed, malformed and unrelated argument traffic cannot renew idle liveness", async () => {
135
+ const h = harness({ stallTimeoutMs: 100 });
136
+ const task = h.runner.run(request());
137
+ await tick();
138
+ const child = h.children[0];
139
+ beginArguments(child);
140
+ const fresh = argumentUpdate("toolcall_delta", { delta: "private-chunk" });
141
+ child.emit(fresh);
142
+ const timer = h.timers.filter(timer => !timer.cancelled).at(-1)!;
143
+ for (const event of [fresh, argumentUpdate("toolcall_delta", { delta: "" }),
144
+ argumentUpdate("toolcall_delta", { delta: 123 }), argumentUpdate("toolcall_delta", { delta: "new", contentIndex: -1 }),
145
+ argumentUpdate("toolcall_delta", { delta: "new", contentIndex: 1 }),
146
+ argumentUpdate("toolcall_start", { id: "call-1", toolName: "write" }), fresh,
147
+ { type: "message_start", message: { role: "assistant", timestamp: 1000 } }, fresh,
148
+ { type: "queue_update" }, { type: "extension_ui_request", method: "setWidget", widgetLines: ["noise"] },
149
+ { type: "bash_execution_update", delta: "noise" }]) child.emit(event);
150
+ assert.equal(timer.cancelled, false);
151
+ timer.fn();
152
+ await tick();
153
+ assert.equal(h.store.get(task.id)?.status, TASK_STATUS.TIMED_OUT);
154
+ });
155
+
156
+ test("argument generation closes at message end, preserves final usage and execution budgets", async () => {
157
+ const h = harness({ stallTimeoutMs: 100, toolStallTimeoutMs: 1000 });
158
+ const task = h.runner.run(request());
159
+ await tick();
160
+ const child = h.children[0];
161
+ beginArguments(child);
162
+ child.emit(argumentUpdate("toolcall_delta", { delta: "private-chunk" }));
163
+ child.emit({ type: "message_end", message: { role: "assistant", usage: { totalTokens: 12, cost: { total: 0.1 } } } });
164
+ const idle = h.timers.filter(timer => !timer.cancelled).at(-1)!;
165
+ child.emit(argumentUpdate("toolcall_delta", { delta: "late" }));
166
+ assert.equal(idle.cancelled, false);
167
+ assert.equal(h.store.get(task.id)?.tokens, 12);
168
+ assert.equal(h.store.get(task.id)?.cost, 0.1);
169
+ child.emit({ type: "tool_execution_start", toolCallId: "call-1", toolName: "write", args: {} });
170
+ assert.equal(h.store.get(task.id)?.toolCalls, 1);
171
+ assert.equal(h.timers.filter(timer => !timer.cancelled).at(-1)?.ms, 1000);
172
+ child.emit({ type: "tool_execution_end", toolCallId: "call-1", result: { content: [] }, isError: false });
173
+ assert.equal(h.timers.filter(timer => !timer.cancelled).at(-1)?.ms, 100);
174
+ beginArguments(child, 1001);
175
+ child.emit(argumentUpdate("toolcall_delta", { delta: "private-chunk" }));
176
+ assert.equal(h.store.get(task.id)?.lastStep, "generating tool arguments", "a new generation admits the same chunk");
177
+ h.runner.cancel(task.id, "cancelled during arguments");
178
+ const timerCount = h.timers.length;
179
+ child.emit(argumentUpdate("toolcall_delta", { delta: "after cancellation" }));
180
+ await tick();
181
+ assert.equal(h.timers.length, timerCount);
182
+ assert.equal(h.store.get(task.id)?.status, TASK_STATUS.CANCELLED);
183
+ assert.equal(h.store.get(task.id)?.toolCalls, 1);
184
+ assert.equal(h.store.get(task.id)?.tokens, 12);
185
+ });
186
+
82
187
  // A child that never answers the launch RPC commands (get_state, prompt), so
83
188
  // the task's lastStep never leaves its initial "starting" stage. Used to
84
189
  // exercise the stall watchdog before any child response arrives.
@@ -657,10 +762,32 @@ test("AgentRunner retains only a 64-notification duplicate window", async () =>
657
762
  assert.equal(notifications.length, 66, "an ID evicted from the recent 64-ack window can be admitted again");
658
763
  });
659
764
 
660
- test("piCommand reuses the running pi entry point and honors the override", () => {
661
- assert.deepEqual(piCommand({ execPath: "/bin/node", argv: ["/bin/node", "/x/dist/cli.js"], env: {} }), { command: "/bin/node", args: ["/x/dist/cli.js"] });
662
- assert.deepEqual(piCommand({ execPath: "/bin/node", argv: ["/bin/node", "/x/other.js"], env: {} }), { command: "pi", args: [] });
663
- assert.deepEqual(piCommand({ execPath: "/bin/node", argv: [], env: { GENTLE_PI_AGENTS_PI: "/opt/pi --flag" } }), { command: "/opt/pi", args: ["--flag"] });
765
+ test("piCommand reuses an existing pi entry point and falls back when it disappears", () => {
766
+ const proc = { execPath: "/bin/node", argv: ["/bin/node", "/x/dist/cli.js"], env: {} };
767
+ assert.deepEqual(piCommand(proc, (entry) => entry === "/x/dist/cli.js"), { command: "/bin/node", args: ["/x/dist/cli.js"] });
768
+ assert.deepEqual(piCommand(proc, () => false), { command: "pi", args: [] });
769
+ assert.deepEqual(piCommand({ ...proc, argv: ["/bin/node", "/x/other.js"] }, () => true), { command: "pi", args: [] });
770
+ assert.deepEqual(piCommand({ ...proc, argv: [] }, () => true), { command: "pi", args: [] });
771
+ });
772
+
773
+ test("piCommand honors the override without checking its entry", () => {
774
+ const proc = { execPath: "/bin/node", argv: ["/bin/node", "/x/dist/cli.js"], env: { GENTLE_PI_AGENTS_PI: " /bin/node /override/cli.js " } };
775
+ assert.deepEqual(piCommand(proc, () => { assert.fail("override must bypass the existence check"); }), { command: "/bin/node", args: ["/override/cli.js"] });
776
+ });
777
+
778
+ test("runner resolves the pi command at each spawn after the entry disappears", async () => {
779
+ let exists = true;
780
+ const proc = { execPath: "/bin/node", argv: ["/bin/node", "/x/dist/cli.js"], env: {} };
781
+ const h = harness({ resolvePi: () => piCommand(proc, () => exists) });
782
+ h.runner.run(request());
783
+ await tick();
784
+ assert.equal(h.spawnOptions[0].command, "/bin/node");
785
+ assert.equal(h.spawnOptions[0].args[0], "/x/dist/cli.js");
786
+ exists = false;
787
+ h.runner.run(request());
788
+ await tick();
789
+ assert.equal(h.spawnOptions[1].command, "pi");
790
+ assert.deepEqual(h.spawnOptions[1].args, childArguments(request()));
664
791
  });
665
792
 
666
793
  test("JsonLines splits on LF only, tolerates CRLF, and skips lines that are not JSON", () => {
@@ -1277,7 +1404,7 @@ test("an unprobeable process group quarantines at its deadline and still records
1277
1404
  pi: { command: "pi", args: [] },
1278
1405
  process: { platform: "win32", kill: () => {} },
1279
1406
  }, { askUser: async () => ({ value: "yes" }), onFinish: (task) => { finishes.push(task.id); } });
1280
- const first = runner.run(managedRequest());
1407
+ const first = runner.run(request());
1281
1408
  const second = runner.run(request({ prompt: "queued" }));
1282
1409
  await tick();
1283
1410
  runner.cancel(first.id);
@@ -1294,10 +1421,11 @@ test("an unprobeable process group quarantines at its deadline and still records
1294
1421
  assert.equal(finishes.length, 1, "the run is recorded exactly once");
1295
1422
  assert.equal(store.get(second.id)?.status, TASK_STATUS.QUEUED, "an unconfirmed exit retains its capacity");
1296
1423
  assert.equal(launches, 1, "no further launch happens while the slot is quarantined");
1297
- assert.throws(() => runner.run(managedRequest()), /Remediation already queued or running/, "a failed record does not release its quarantined child");
1424
+ const third = runner.run(request({ prompt: "another ordinary task" }));
1425
+ assert.equal(store.get(third.id)?.status, TASK_STATUS.QUEUED);
1298
1426
  child!.exit(0);
1299
1427
  await tick();
1300
- assert.doesNotThrow(() => runner.run(managedRequest()), "confirmed cleanup releases the managed workspace");
1428
+ assert.equal(store.get(second.id)?.status, TASK_STATUS.RUNNING, "confirmed cleanup frees capacity for ordinary work");
1301
1429
  runner.cancelAll();
1302
1430
  child!.exit(0);
1303
1431
  });
@@ -1310,97 +1438,258 @@ test("abortReasonText renders an Error, a string, and nothing for unknown reason
1310
1438
  assert.equal(abortReasonText(42), "");
1311
1439
  });
1312
1440
 
1313
- test("research narrowing transport keeps exact argv paths and replaces inherited selection", async () => {
1441
+ test("generic child extension paths do not forward legacy research selection", async () => {
1314
1442
  const h = harness();
1315
- const selection = { documentation: { tools: ["fetch_content"], extensions: { fetch_content: "/installed/docs tools.ts" } } };
1316
- for (const researchSelection of [selection, undefined]) {
1317
- const launch = request({ researchSelection, extensionPaths: researchSelection ? ["/installed/docs tools.ts"] : [],
1318
- env: { PATH: "/bin", GENTLE_PI_RESEARCH_SELECTION: "stale broad selection" } });
1319
- const argv = childArguments(launch);
1320
- assert.deepEqual(argv.filter((_, i) => argv[i - 1] === "--extension"), launch.extensionPaths);
1321
- const task = h.runner.run(launch);
1322
- await tick();
1323
- assert.deepEqual(JSON.parse(h.spawnOptions.at(-1)!.env.GENTLE_PI_RESEARCH_SELECTION!), researchSelection ?? null);
1324
- assert.equal(h.spawnOptions.at(-1)!.env.PATH, "/bin");
1325
- h.runner.cancel(task.id);
1326
- assert.equal((await h.runner.waitFor(task.id)).status, TASK_STATUS.CANCELLED);
1327
- }
1328
- });
1329
-
1330
- function managedRequest(cwd = "/repo"): TaskRequest {
1331
- return request({ agent: { ...explorer, name: "sdd-remediate" }, cwd, sddRemediation: {
1332
- failedEvidenceRevision: "failed-revision",
1333
- plan: { cwd, commands: ["pnpm test"], runtimeHarness: { naReason: "Not applicable because this tests runner admission." }, rollback: { boundary: "fixture", command: "git diff --check" } },
1334
- scope: { cwd, editPaths: [], commands: ["pnpm test", "git diff --check"], allowedEditRoots: [cwd] },
1335
- } });
1336
- }
1337
-
1338
- for (const queued of [true, false]) test(`managed exclusion covers ${queued ? "queued" : "running"} same-workspace actors`, async () => {
1339
- const h = harness({ pid: 123, process: { platform: "win32", kill() {} } });
1340
- const first = h.runner.run(managedRequest());
1341
- if (!queued) await tick();
1342
- try {
1343
- assert.throws(() => h.runner.run(managedRequest()), /Remediation already queued or running/);
1344
- assert.equal(h.store.list().length, 1, "rejection creates no task or queue entry");
1345
- await tick();
1346
- assert.equal(h.children.length, 1);
1347
- assert.equal(h.store.get(first.id)?.status, TASK_STATUS.RUNNING);
1348
- } finally { h.runner.cancelAll(); await tick(); }
1443
+ const launch = request({ extensionPaths: ["/installed/docs tools.ts"], env: { PATH: "/bin", GENTLE_PI_RESEARCH_SELECTION: "stale" } });
1444
+ const argv = childArguments(launch);
1445
+ assert.deepEqual(argv.filter((_, i) => argv[i - 1] === "--extension"), launch.extensionPaths);
1446
+ const task = h.runner.run(launch);
1447
+ await tick();
1448
+ assert.equal(h.spawnOptions.at(-1)!.env.GENTLE_PI_RESEARCH_SELECTION, undefined);
1449
+ assert.equal(h.spawnOptions.at(-1)!.env.PATH, "/bin");
1450
+ h.runner.cancel(task.id);
1451
+ assert.equal((await h.runner.waitFor(task.id)).status, TASK_STATUS.CANCELLED);
1349
1452
  });
1350
1453
 
1351
- test("managed exclusion does not serialize other workspaces or ordinary tasks", async () => {
1352
- const h = harness({ maxConcurrency: 3, pid: 123, process: { platform: "win32", kill() {} } });
1353
- h.runner.run(managedRequest());
1354
- h.runner.run(managedRequest("/other"));
1355
- h.runner.run(request());
1454
+ test("ordinary tasks never inherit orphaned SDD launch metadata", async () => {
1455
+ const h = harness();
1456
+ const launch = request({ prompt: "Ordinary task", context: "Relevant context", env: { PATH: "/bin", GENTLE_PI_SDD_REMEDIATION_PLAN: "stale" },
1457
+ // Deliberately pass a legacy-shaped payload to prove that no runner path consumes it.
1458
+ ...({ sddChange: { changeName: "old", workspaceRoot: "/repo", phase: "apply" }, sddPreflightContext: "stale", sddRemediation: { failedEvidenceRevision: "old", plan: { commands: ["unsafe"] } } } as object),
1459
+ });
1460
+ assert.doesNotMatch(childArguments(launch).join(" "), /gentle-sdd-change/);
1461
+ const task = h.runner.run(launch);
1356
1462
  await tick();
1357
- assert.equal(h.children.length, 3);
1358
- h.runner.cancelAll();
1463
+ assert.equal(h.store.get(task.id)?.sddPreflightContext, undefined);
1464
+ assert.equal(h.spawnOptions[0].env.GENTLE_PI_SDD_REMEDIATION_PLAN, undefined);
1465
+ assert.equal(h.spawnOptions[0].env.PATH, "/bin");
1466
+ assert.equal(h.children[0].written.find(command => command.type === "prompt")?.message, "Ordinary task\n\n## Context\nRelevant context");
1467
+ h.runner.cancel(task.id);
1468
+ assert.equal((await h.runner.waitFor(task.id)).status, TASK_STATUS.CANCELLED);
1469
+ });
1470
+
1471
+ test("AgentRunner preserves the terminating signal when child exits with null code before settlement", async () => {
1472
+ const { runner, children, store } = harness();
1473
+ const task = runner.run(request());
1359
1474
  await tick();
1475
+ assert.equal(children.length, 1);
1476
+ children[0].exit(null, "SIGKILL");
1477
+ const finished = await runner.waitFor(task.id);
1478
+ assert.equal(finished.status, TASK_STATUS.FAILED);
1479
+ assert.equal(finished.error, "pi exited with signal SIGKILL before agent_settled");
1480
+ assert.equal(store.get(task.id)?.error, "pi exited with signal SIGKILL before agent_settled");
1360
1481
  });
1361
1482
 
1362
- for (const ending of ["complete", "failure", "cancel", "queued-cancel"] as const) test(`managed exclusion releases after ${ending}`, async () => {
1363
- const h = harness({ pid: 123, process: { platform: "win32", kill() {} } });
1364
- const first = h.runner.run(managedRequest());
1365
- if (ending === "queued-cancel") h.runner.cancel(first.id);
1366
- else {
1367
- await tick();
1368
- if (ending === "complete") {
1369
- h.children[0].emit({ type: "agent_end", messages: [{ role: "assistant", content: [{ type: "text", text: "Finished" }], stopReason: "stop" }] });
1370
- h.children[0].emit({ type: "agent_settled" });
1371
- } else if (ending === "failure") h.children[0].exit(1);
1372
- else h.runner.cancel(first.id);
1483
+ test("large agent instructions are transported via owner-only temporary file rather than inline argv", async () => {
1484
+ const largeInstructions = "Instructions header:\n" + "x".repeat(2500);
1485
+ const largeAgent: AgentDefinition = { ...explorer, instructions: largeInstructions };
1486
+ const launches: Array<{ command: string; args: string[]; options: Parameters<RunnerDeps["spawn"]>[2] }> = [];
1487
+ const fake = fakeChild();
1488
+ let clock = 1000;
1489
+ const deps: RunnerDeps = {
1490
+ spawn: (command, args, options) => {
1491
+ launches.push({ command, args, options });
1492
+ return fake.child;
1493
+ },
1494
+ now: () => (clock += 1),
1495
+ schedule: (_fn, _ms) => () => {},
1496
+ pi: { command: "pi", args: [] },
1497
+ };
1498
+ const store = new TaskStore();
1499
+ const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 10_000 }, deps, {
1500
+ askUser: async () => ({ value: "yes" }),
1501
+ });
1502
+ const task = runner.run(request({ agent: largeAgent }));
1503
+ await tick();
1504
+
1505
+ assert.equal(launches.length, 1);
1506
+ const promptArgIndex = launches[0].args.indexOf("--append-system-prompt");
1507
+ assert.ok(promptArgIndex !== -1, "--append-system-prompt must be present");
1508
+ const promptValue = launches[0].args[promptArgIndex + 1];
1509
+ assert.notEqual(promptValue, largeInstructions, "large instructions must not be passed inline in argv");
1510
+ assert.ok(existsSync(promptValue), "temporary instructions transport file must exist on disk");
1511
+ assert.equal(readFileSync(promptValue, "utf8"), largeInstructions, "transport file must contain the exact instructions");
1512
+
1513
+ if (process.platform !== "win32") {
1514
+ const fileStat = statSync(promptValue);
1515
+ assert.equal(fileStat.mode & 0o777, 0o600, "transport file must be owner-only (0o600)");
1516
+ const dirStat = statSync(dirname(promptValue));
1517
+ assert.equal(dirStat.mode & 0o777, 0o700, "transport directory must be owner-only (0o700)");
1373
1518
  }
1374
- await h.runner.waitFor(first.id);
1375
- const next = h.runner.run(managedRequest());
1376
- await tick();
1377
- assert.equal(h.store.get(next.id)?.status, TASK_STATUS.RUNNING);
1378
- h.runner.cancelAll();
1519
+
1520
+ fake.exit(0);
1521
+ await runner.waitFor(task.id);
1522
+ assert.ok(!existsSync(promptValue), "temporary transport file must be cleaned up on child exit");
1523
+ assert.ok(!existsSync(dirname(promptValue)), "temporary transport directory must be cleaned up on child exit");
1524
+ });
1525
+
1526
+ test("temporary instructions transport file is cleaned up if spawn throws synchronously", async () => {
1527
+ const largeInstructions = "Instructions header:\n" + "x".repeat(2500);
1528
+ const largeAgent: AgentDefinition = { ...explorer, instructions: largeInstructions };
1529
+ let capturedPromptPath: string | undefined;
1530
+ let clock = 1000;
1531
+ const deps: RunnerDeps = {
1532
+ spawn: (_command, args) => {
1533
+ const idx = args.indexOf("--append-system-prompt");
1534
+ if (idx !== -1) capturedPromptPath = args[idx + 1];
1535
+ throw new Error("spawn failed intentionally");
1536
+ },
1537
+ now: () => (clock += 1),
1538
+ schedule: (_fn, _ms) => () => {},
1539
+ pi: { command: "pi", args: [] },
1540
+ };
1541
+ const store = new TaskStore();
1542
+ const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 10_000 }, deps, {
1543
+ askUser: async () => ({ value: "yes" }),
1544
+ });
1545
+ const task = runner.run(request({ agent: largeAgent }));
1379
1546
  await tick();
1547
+
1548
+ const finished = await runner.waitFor(task.id);
1549
+ assert.equal(finished.status, TASK_STATUS.FAILED);
1550
+ assert.ok(capturedPromptPath, "should have captured a transport file path");
1551
+ assert.ok(!existsSync(capturedPromptPath), "temporary transport file must be cleaned up even when spawn throws");
1552
+ assert.ok(!existsSync(dirname(capturedPromptPath)), "temporary transport directory must be cleaned up even when spawn throws");
1380
1553
  });
1381
1554
 
1382
- test("managed exclusion lasts until child cleanup is confirmed", async () => {
1383
- const h = harness({ pid: 123, exitOnKill: false, process: { platform: "win32", kill() {} } });
1384
- const first = h.runner.run(managedRequest());
1555
+ test("agent instructions over the byte threshold are transported via file even when under the character threshold", async () => {
1556
+ const multibyteInstructions = "界".repeat(400);
1557
+ assert.ok(multibyteInstructions.length < 1000 && Buffer.byteLength(multibyteInstructions, "utf8") > 1000);
1558
+ const launches: string[][] = [];
1559
+ const fake = fakeChild();
1560
+ let clock = 1000;
1561
+ const deps: RunnerDeps = {
1562
+ spawn: (_command, args) => {
1563
+ launches.push(args);
1564
+ return fake.child;
1565
+ },
1566
+ now: () => (clock += 1),
1567
+ schedule: (_fn, _ms) => () => {},
1568
+ pi: { command: "pi", args: [] },
1569
+ };
1570
+ const runner = new AgentRunner(new TaskStore(), { maxConcurrency: 1, stallTimeoutMs: 10_000 }, deps, {
1571
+ askUser: async () => ({ value: "yes" }),
1572
+ });
1573
+ const task = runner.run(request({ agent: { ...explorer, instructions: multibyteInstructions } }));
1385
1574
  await tick();
1386
- h.runner.cancel(first.id);
1387
- try { assert.throws(() => h.runner.run(managedRequest()), /Remediation already queued or running/); }
1388
- finally { h.children[0].exit(0); }
1389
- await h.runner.waitFor(first.id);
1390
- const next = h.runner.run(managedRequest());
1575
+
1576
+ assert.equal(launches.length, 1);
1577
+ const promptValue = launches[0][launches[0].indexOf("--append-system-prompt") + 1];
1578
+ assert.notEqual(promptValue, multibyteInstructions, "multibyte instructions over the byte threshold must not be passed inline");
1579
+ assert.ok(existsSync(promptValue), "temporary instructions transport file must exist on disk");
1580
+ assert.equal(readFileSync(promptValue, "utf8"), multibyteInstructions);
1581
+
1582
+ fake.exit(0);
1583
+ await runner.waitFor(task.id);
1584
+ assert.ok(!existsSync(dirname(promptValue)), "temporary transport directory must be cleaned up on child exit");
1585
+ });
1586
+
1587
+ test("long agent names are truncated in the instructions transport directory name", async () => {
1588
+ const largeInstructions = "Instructions header:\n" + "x".repeat(2500);
1589
+ const launches: string[][] = [];
1590
+ const fake = fakeChild();
1591
+ let clock = 1000;
1592
+ const deps: RunnerDeps = {
1593
+ spawn: (_command, args) => {
1594
+ launches.push(args);
1595
+ return fake.child;
1596
+ },
1597
+ now: () => (clock += 1),
1598
+ schedule: (_fn, _ms) => () => {},
1599
+ pi: { command: "pi", args: [] },
1600
+ };
1601
+ const runner = new AgentRunner(new TaskStore(), { maxConcurrency: 1, stallTimeoutMs: 10_000 }, deps, {
1602
+ askUser: async () => ({ value: "yes" }),
1603
+ });
1604
+ const task = runner.run(request({ agent: { ...explorer, name: "a".repeat(300), instructions: largeInstructions } }));
1391
1605
  await tick();
1392
- assert.equal(h.store.get(next.id)?.status, TASK_STATUS.RUNNING);
1393
- h.runner.cancelAll();
1394
- for (const child of h.children) child.exit(0);
1606
+
1607
+ assert.equal(launches.length, 1, "launch must succeed despite a long agent name");
1608
+ const promptValue = launches[0][launches[0].indexOf("--append-system-prompt") + 1];
1609
+ assert.equal(readFileSync(promptValue, "utf8"), largeInstructions);
1610
+ const dirName = basename(dirname(promptValue));
1611
+ assert.ok(dirName.startsWith(`gentle-pi-subagent-${"a".repeat(64)}-`), dirName);
1612
+ assert.ok(dirName.length <= "gentle-pi-subagent-".length + 64 + 1 + 6, `directory name too long: ${dirName.length}`);
1613
+
1614
+ fake.exit(0);
1615
+ await runner.waitFor(task.id);
1616
+ assert.ok(!existsSync(dirname(promptValue)));
1617
+ });
1618
+
1619
+ test("temporary instructions transport directory is cleaned up if writing instructions fails", async (t) => {
1620
+ // Fail only the transport write; the ESM named import is refreshed via syncBuiltinESMExports.
1621
+ const originalWriteFileSync = fs.writeFileSync;
1622
+ let transportDir: string | undefined;
1623
+ t.mock.method(fs, "writeFileSync", (...args: Parameters<typeof fs.writeFileSync>) => {
1624
+ const [target] = args;
1625
+ if (typeof target === "string" && basename(target) === "instructions.md" && basename(dirname(target)).startsWith("gentle-pi-subagent-")) {
1626
+ transportDir = dirname(target);
1627
+ throw new Error("EACCES: simulated write failure");
1628
+ }
1629
+ return originalWriteFileSync(...args);
1630
+ });
1631
+ syncBuiltinESMExports();
1632
+ t.after(() => {
1633
+ t.mock.restoreAll();
1634
+ syncBuiltinESMExports();
1635
+ });
1636
+ const failingAgent: AgentDefinition = {
1637
+ ...explorer,
1638
+ instructions: "Instructions header:\n" + "x".repeat(2500),
1639
+ };
1640
+ let clock = 1000;
1641
+ const deps: RunnerDeps = {
1642
+ spawn: () => {
1643
+ throw new Error("spawn should not be called when writing instructions fails");
1644
+ },
1645
+ now: () => (clock += 1),
1646
+ schedule: (_fn, _ms) => () => {},
1647
+ pi: { command: "pi", args: [] },
1648
+ };
1649
+ const store = new TaskStore();
1650
+ const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 10_000 }, deps, {
1651
+ askUser: async () => ({ value: "yes" }),
1652
+ });
1653
+ const task = runner.run(request({ agent: failingAgent }));
1395
1654
  await tick();
1655
+
1656
+ const finished = await runner.waitFor(task.id);
1657
+ assert.equal(finished.status, TASK_STATUS.FAILED);
1658
+ assert.match(finished.error ?? "", /could not write agent instructions: EACCES: simulated write failure/);
1659
+ assert.ok(transportDir, "transport directory must have been created before the write failed");
1660
+ assert.ok(!existsSync(transportDir), "transport directory must be cleaned up on write failure");
1396
1661
  });
1397
1662
 
1398
- test("managed exclusion releases failed startup and ignores historical-only tasks", async () => {
1399
- const h = harness({ failStart: true });
1400
- const first = h.runner.run(managedRequest());
1401
- assert.equal((await h.runner.waitFor(first.id)).status, TASK_STATUS.FAILED);
1402
- h.store.add({ ...h.store.get(first.id)!, id: "historical-only", status: TASK_STATUS.RUNNING });
1403
- const next = h.runner.run(managedRequest());
1404
- assert.equal((await h.runner.waitFor(next.id)).status, TASK_STATUS.FAILED, "the next actor reaches spawn, not a historical admission lock");
1405
- assert.match(h.store.get(next.id)?.error ?? "", /fixture spawn failed/);
1663
+ test("temporary instructions transport file is cleaned up if child emits an early error before PID", async () => {
1664
+ const largeInstructions = "Instructions header:\n" + "x".repeat(2500);
1665
+ const largeAgent: AgentDefinition = { ...explorer, instructions: largeInstructions };
1666
+ let capturedPromptPath: string | undefined;
1667
+ const fake = fakeChild({ pid: undefined });
1668
+ let clock = 1000;
1669
+ const deps: RunnerDeps = {
1670
+ spawn: (_command, args) => {
1671
+ const idx = args.indexOf("--append-system-prompt");
1672
+ if (idx !== -1) capturedPromptPath = args[idx + 1];
1673
+ queueMicrotask(() => {
1674
+ fake.fail("spawn ENOENT");
1675
+ });
1676
+ return fake.child;
1677
+ },
1678
+ now: () => (clock += 1),
1679
+ schedule: (_fn, _ms) => () => {},
1680
+ pi: { command: "pi", args: [] },
1681
+ };
1682
+ const store = new TaskStore();
1683
+ const runner = new AgentRunner(store, { maxConcurrency: 1, stallTimeoutMs: 10_000 }, deps, {
1684
+ askUser: async () => ({ value: "yes" }),
1685
+ });
1686
+ const task = runner.run(request({ agent: largeAgent }));
1687
+ await tick();
1688
+
1689
+ const finished = await runner.waitFor(task.id);
1690
+ assert.equal(finished.status, TASK_STATUS.FAILED);
1691
+ assert.match(finished.error ?? "", /could not start pi: spawn ENOENT/);
1692
+ assert.ok(capturedPromptPath, "should have captured a transport file path");
1693
+ assert.ok(!existsSync(capturedPromptPath), "temporary transport file must be cleaned up on early child error");
1694
+ assert.ok(!existsSync(dirname(capturedPromptPath)), "temporary transport directory must be cleaned up on early child error");
1406
1695
  });