@mjasnikovs/pi-task 0.38.29 → 0.38.30

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (370) hide show
  1. package/dist/config/config.d.ts +70 -70
  2. package/dist/config/config.js +26 -35
  3. package/dist/config/extension-list.d.ts +6 -5
  4. package/dist/config/extension-list.js +3 -2
  5. package/dist/config/reasoning-args.d.ts +9 -7
  6. package/dist/config/reasoning-args.js +12 -10
  7. package/dist/config/reasoning.d.ts +44 -105
  8. package/dist/config/reasoning.js +27 -704
  9. package/dist/config/register.d.ts +34 -48
  10. package/dist/config/register.js +41 -51
  11. package/dist/config/tool-list.d.ts +16 -16
  12. package/dist/config/tool-list.js +1 -1
  13. package/dist/remote/bridge.d.ts +19 -10
  14. package/dist/remote/bridge.js +3 -2
  15. package/dist/remote/broadcast.js +3 -1
  16. package/dist/remote/events.js +12 -11
  17. package/dist/remote/history.d.ts +1 -1
  18. package/dist/remote/protocol.d.ts +6 -3
  19. package/dist/remote/protocol.js +2 -1
  20. package/dist/remote/push.d.ts +16 -16
  21. package/dist/remote/push.js +27 -27
  22. package/dist/remote/register.d.ts +3 -3
  23. package/dist/remote/register.js +17 -19
  24. package/dist/remote/server.d.ts +9 -8
  25. package/dist/remote/server.js +15 -14
  26. package/dist/remote/session-state.d.ts +5 -4
  27. package/dist/remote/session-state.js +8 -5
  28. package/dist/remote/sw.d.ts +7 -6
  29. package/dist/remote/sw.js +7 -6
  30. package/dist/remote/tailscale.d.ts +4 -2
  31. package/dist/remote/tailscale.js +4 -2
  32. package/dist/remote/ui-highlight.js +6 -5
  33. package/dist/remote/ui-render.js +4 -4
  34. package/dist/remote/ui-script.js +24 -24
  35. package/dist/remote/ui-styles.d.ts +1 -1
  36. package/dist/remote/ui-styles.js +10 -13
  37. package/dist/remote/ui-tools.js +9 -6
  38. package/dist/shared/child-extensions.d.ts +29 -17
  39. package/dist/shared/child-extensions.js +29 -17
  40. package/dist/shared/child-output.d.ts +30 -24
  41. package/dist/shared/child-output.js +25 -17
  42. package/dist/shared/child-process.d.ts +47 -40
  43. package/dist/shared/child-process.js +50 -59
  44. package/dist/shared/command-watchdog.d.ts +22 -16
  45. package/dist/shared/command-watchdog.js +28 -21
  46. package/dist/shared/fs-text.d.ts +16 -10
  47. package/dist/shared/fs-text.js +16 -10
  48. package/dist/shared/git-runner.d.ts +25 -25
  49. package/dist/shared/git-runner.js +25 -25
  50. package/dist/shared/leaked-tool-call.d.ts +17 -11
  51. package/dist/shared/leaked-tool-call.js +23 -15
  52. package/dist/shared/model-endpoint.d.ts +29 -16
  53. package/dist/shared/model-endpoint.js +33 -21
  54. package/dist/shared/pi-invocation.d.ts +7 -4
  55. package/dist/shared/pi-invocation.js +12 -7
  56. package/dist/shared/pkg-version.d.ts +13 -5
  57. package/dist/shared/pkg-version.js +13 -5
  58. package/dist/shared/reasoning-capability.d.ts +35 -24
  59. package/dist/shared/reasoning-capability.js +35 -24
  60. package/dist/shared/stream-watchdog.d.ts +60 -44
  61. package/dist/shared/stream-watchdog.js +62 -45
  62. package/dist/task/accept-debt.d.ts +41 -43
  63. package/dist/task/accept-debt.js +73 -65
  64. package/dist/task/api-synthesis.d.ts +24 -21
  65. package/dist/task/api-synthesis.js +32 -26
  66. package/dist/task/apis-contract.d.ts +32 -64
  67. package/dist/task/apis-contract.js +32 -64
  68. package/dist/task/artifact-closure.d.ts +27 -13
  69. package/dist/task/artifact-closure.js +95 -67
  70. package/dist/task/auto-commit.d.ts +46 -35
  71. package/dist/task/auto-commit.js +51 -38
  72. package/dist/task/auto-io.d.ts +45 -25
  73. package/dist/task/auto-io.js +57 -29
  74. package/dist/task/auto-orchestrator.d.ts +26 -24
  75. package/dist/task/auto-orchestrator.js +178 -162
  76. package/dist/task/auto-prompts.d.ts +36 -24
  77. package/dist/task/auto-prompts.js +40 -26
  78. package/dist/task/autofix-ledger.d.ts +27 -25
  79. package/dist/task/autofix-ledger.js +29 -26
  80. package/dist/task/batch-test-task.d.ts +20 -12
  81. package/dist/task/batch-test-task.js +67 -60
  82. package/dist/task/boot-probe.d.ts +60 -44
  83. package/dist/task/boot-probe.js +91 -72
  84. package/dist/task/cancel-input.d.ts +30 -16
  85. package/dist/task/cancel-input.js +20 -11
  86. package/dist/task/cancel-points.d.ts +27 -20
  87. package/dist/task/cancel-points.js +30 -22
  88. package/dist/task/child-runner.d.ts +46 -51
  89. package/dist/task/child-runner.js +48 -49
  90. package/dist/task/child-status.d.ts +23 -16
  91. package/dist/task/child-status.js +23 -16
  92. package/dist/task/clamp-output.js +12 -5
  93. package/dist/task/command-run.d.ts +31 -28
  94. package/dist/task/command-run.js +44 -35
  95. package/dist/task/command-shrink.d.ts +25 -18
  96. package/dist/task/command-shrink.js +37 -31
  97. package/dist/task/command-watchdog.d.ts +9 -6
  98. package/dist/task/command-watchdog.js +21 -15
  99. package/dist/task/context-attribution.d.ts +34 -26
  100. package/dist/task/context-attribution.js +34 -26
  101. package/dist/task/context-silence.d.ts +39 -29
  102. package/dist/task/context-silence.js +35 -25
  103. package/dist/task/context-usage.d.ts +16 -9
  104. package/dist/task/context-usage.js +16 -9
  105. package/dist/task/contracts.d.ts +8 -4
  106. package/dist/task/contracts.js +25 -17
  107. package/dist/task/coverage-loop.d.ts +22 -18
  108. package/dist/task/coverage-loop.js +35 -30
  109. package/dist/task/critique-probes.d.ts +13 -14
  110. package/dist/task/critique-probes.js +50 -39
  111. package/dist/task/debug-log.d.ts +13 -5
  112. package/dist/task/debug-log.js +32 -20
  113. package/dist/task/decompose-fidelity.d.ts +11 -9
  114. package/dist/task/decompose-fidelity.js +38 -33
  115. package/dist/task/decompose-granularity.d.ts +41 -38
  116. package/dist/task/decompose-granularity.js +41 -38
  117. package/dist/task/deep-render-check.d.ts +22 -14
  118. package/dist/task/deep-render-check.js +40 -31
  119. package/dist/task/dropped-input.d.ts +12 -7
  120. package/dist/task/dropped-input.js +5 -2
  121. package/dist/task/enforce-attribution.d.ts +38 -47
  122. package/dist/task/enforce-attribution.js +46 -52
  123. package/dist/task/enforce-guidelines.d.ts +31 -20
  124. package/dist/task/enforce-guidelines.js +32 -21
  125. package/dist/task/enrichment.d.ts +7 -2
  126. package/dist/task/enrichment.js +26 -14
  127. package/dist/task/env-notes.d.ts +16 -7
  128. package/dist/task/env-notes.js +48 -31
  129. package/dist/task/env-template-closure.d.ts +4 -4
  130. package/dist/task/env-template-closure.js +42 -34
  131. package/dist/task/external-context.d.ts +28 -21
  132. package/dist/task/external-context.js +17 -12
  133. package/dist/task/failure-classifier.d.ts +4 -5
  134. package/dist/task/failure-classifier.js +6 -7
  135. package/dist/task/file-inventory.d.ts +15 -11
  136. package/dist/task/file-inventory.js +25 -22
  137. package/dist/task/final-gate-fix.d.ts +74 -86
  138. package/dist/task/final-gate-fix.js +97 -116
  139. package/dist/task/final-gate-progress.d.ts +29 -46
  140. package/dist/task/final-gate-progress.js +40 -51
  141. package/dist/task/final-gate.d.ts +64 -97
  142. package/dist/task/final-gate.js +192 -199
  143. package/dist/task/fix-child.d.ts +21 -27
  144. package/dist/task/fix-child.js +21 -27
  145. package/dist/task/foreign-path.d.ts +6 -5
  146. package/dist/task/foreign-path.js +0 -0
  147. package/dist/task/frozen-conflict.d.ts +9 -10
  148. package/dist/task/frozen-conflict.js +61 -64
  149. package/dist/task/frozen-path-guard.d.ts +35 -14
  150. package/dist/task/frozen-path-guard.js +56 -39
  151. package/dist/task/gate-child.d.ts +27 -28
  152. package/dist/task/gate-child.js +36 -35
  153. package/dist/task/gate-deps.d.ts +34 -27
  154. package/dist/task/gate-deps.js +169 -159
  155. package/dist/task/gate-tally.d.ts +77 -80
  156. package/dist/task/gate-tally.js +65 -68
  157. package/dist/task/git-state-guard.d.ts +15 -11
  158. package/dist/task/git-state-guard.js +76 -66
  159. package/dist/task/impl-widget.d.ts +25 -16
  160. package/dist/task/impl-widget.js +27 -17
  161. package/dist/task/implementation-thinking.d.ts +33 -31
  162. package/dist/task/implementation-thinking.js +5 -6
  163. package/dist/task/implementation-turn.d.ts +34 -31
  164. package/dist/task/implementation-turn.js +29 -27
  165. package/dist/task/inline-markdown.d.ts +20 -7
  166. package/dist/task/inline-markdown.js +15 -6
  167. package/dist/task/launch-config-gap.js +25 -39
  168. package/dist/task/launch-contract.d.ts +18 -21
  169. package/dist/task/launch-contract.js +28 -30
  170. package/dist/task/launch-manifest.d.ts +6 -2
  171. package/dist/task/launch-manifest.js +35 -34
  172. package/dist/task/ledger.js +16 -14
  173. package/dist/task/lint-fix.d.ts +6 -8
  174. package/dist/task/lint-fix.js +67 -69
  175. package/dist/task/loop-detector.d.ts +9 -8
  176. package/dist/task/loop-detector.js +16 -12
  177. package/dist/task/mid-run-input.d.ts +17 -15
  178. package/dist/task/mid-run-input.js +17 -15
  179. package/dist/task/orchestrator.d.ts +24 -28
  180. package/dist/task/orchestrator.js +62 -64
  181. package/dist/task/orientation.d.ts +18 -23
  182. package/dist/task/orientation.js +24 -31
  183. package/dist/task/owned-freeze-conflict.d.ts +21 -20
  184. package/dist/task/owned-freeze-conflict.js +52 -85
  185. package/dist/task/owned-freeze-reassign.d.ts +40 -60
  186. package/dist/task/owned-freeze-reassign.js +41 -61
  187. package/dist/task/parsers.d.ts +4 -2
  188. package/dist/task/parsers.js +4 -4
  189. package/dist/task/phases.d.ts +41 -48
  190. package/dist/task/phases.js +179 -248
  191. package/dist/task/plan-io.d.ts +6 -7
  192. package/dist/task/plan-io.js +6 -7
  193. package/dist/task/plan-orchestrator.d.ts +10 -8
  194. package/dist/task/plan-orchestrator.js +14 -10
  195. package/dist/task/plan-prompts.d.ts +6 -5
  196. package/dist/task/plan-prompts.js +6 -5
  197. package/dist/task/plan-readonly.d.ts +4 -5
  198. package/dist/task/plan-readonly.js +4 -5
  199. package/dist/task/plan-rounds.d.ts +17 -29
  200. package/dist/task/plan-rounds.js +21 -34
  201. package/dist/task/plan-session.d.ts +58 -72
  202. package/dist/task/plan-session.js +61 -83
  203. package/dist/task/probe-gaming.d.ts +28 -27
  204. package/dist/task/probe-gaming.js +0 -0
  205. package/dist/task/prohibition-probe.d.ts +14 -16
  206. package/dist/task/prompts.d.ts +3 -4
  207. package/dist/task/prompts.js +17 -26
  208. package/dist/task/qa-transcript.d.ts +15 -22
  209. package/dist/task/qa-transcript.js +15 -21
  210. package/dist/task/question-box.d.ts +17 -13
  211. package/dist/task/question-box.js +19 -15
  212. package/dist/task/question-dedup.d.ts +6 -7
  213. package/dist/task/question-dedup.js +13 -14
  214. package/dist/task/question-dialog.d.ts +22 -32
  215. package/dist/task/question-dialog.js +22 -32
  216. package/dist/task/question-source.d.ts +18 -44
  217. package/dist/task/question-source.js +22 -51
  218. package/dist/task/refuted-constraint.d.ts +11 -31
  219. package/dist/task/refuted-constraint.js +27 -51
  220. package/dist/task/regenerable-artifacts.d.ts +12 -31
  221. package/dist/task/regenerable-artifacts.js +12 -31
  222. package/dist/task/render-check.d.ts +11 -22
  223. package/dist/task/render-check.js +33 -46
  224. package/dist/task/repo-health-check.d.ts +10 -14
  225. package/dist/task/repo-health-check.js +17 -23
  226. package/dist/task/requirements.d.ts +38 -71
  227. package/dist/task/requirements.js +78 -126
  228. package/dist/task/research-fanout-budget.d.ts +51 -88
  229. package/dist/task/research-fanout-budget.js +51 -88
  230. package/dist/task/research-worker.d.ts +29 -39
  231. package/dist/task/research-worker.js +37 -61
  232. package/dist/task/resume-gap.d.ts +14 -15
  233. package/dist/task/root-cause-repair.d.ts +9 -9
  234. package/dist/task/root-cause-repair.js +28 -40
  235. package/dist/task/run-bracket.d.ts +10 -13
  236. package/dist/task/run-end.d.ts +12 -22
  237. package/dist/task/run-end.js +8 -16
  238. package/dist/task/run-final-gate.d.ts +19 -21
  239. package/dist/task/run-final-gate.js +62 -80
  240. package/dist/task/runner-globs.d.ts +12 -13
  241. package/dist/task/runner-globs.js +12 -13
  242. package/dist/task/runner-resolve.d.ts +9 -9
  243. package/dist/task/runner-resolve.js +22 -23
  244. package/dist/task/script-escape.d.ts +10 -12
  245. package/dist/task/script-escape.js +13 -14
  246. package/dist/task/serve-entry.d.ts +1 -1
  247. package/dist/task/serve-entry.js +22 -25
  248. package/dist/task/service-blocks.js +4 -2
  249. package/dist/task/shipped-source.d.ts +11 -29
  250. package/dist/task/shipped-source.js +11 -29
  251. package/dist/task/skip-escape.js +10 -14
  252. package/dist/task/spec-urls.d.ts +26 -65
  253. package/dist/task/spec-urls.js +26 -65
  254. package/dist/task/spec-validation.d.ts +17 -20
  255. package/dist/task/spec-validation.js +17 -20
  256. package/dist/task/stall-detector.d.ts +23 -30
  257. package/dist/task/stall-detector.js +23 -30
  258. package/dist/task/stream-watchdog.d.ts +14 -12
  259. package/dist/task/stream-watchdog.js +14 -12
  260. package/dist/task/substitution-probe.d.ts +17 -20
  261. package/dist/task/substitution-probe.js +17 -20
  262. package/dist/task/task-gates.d.ts +36 -41
  263. package/dist/task/task-gates.js +95 -106
  264. package/dist/task/task-io.d.ts +4 -4
  265. package/dist/task/task-io.js +4 -4
  266. package/dist/task/task-parsers.js +4 -3
  267. package/dist/task/task-provenance.d.ts +2 -2
  268. package/dist/task/task-provenance.js +11 -13
  269. package/dist/task/task-types.d.ts +4 -3
  270. package/dist/task/terminal-outcome.d.ts +14 -16
  271. package/dist/task/terminal-outcome.js +12 -14
  272. package/dist/task/test-assembly.d.ts +13 -20
  273. package/dist/task/test-assembly.js +13 -20
  274. package/dist/task/timings.d.ts +5 -3
  275. package/dist/task/timings.js +5 -3
  276. package/dist/task/title-label.d.ts +9 -4
  277. package/dist/task/title-label.js +9 -4
  278. package/dist/task/type-only-answer.d.ts +44 -52
  279. package/dist/task/type-only-answer.js +44 -52
  280. package/dist/task/unfailable-command.d.ts +18 -24
  281. package/dist/task/unfailable-command.js +21 -27
  282. package/dist/task/unknown-routing.d.ts +10 -4
  283. package/dist/task/unknown-routing.js +10 -4
  284. package/dist/task/user-directives.d.ts +5 -8
  285. package/dist/task/user-directives.js +5 -8
  286. package/dist/task/verify-quality.d.ts +18 -22
  287. package/dist/task/verify-quality.js +45 -46
  288. package/dist/task/verify-reconcile.d.ts +15 -10
  289. package/dist/task/verify-reconcile.js +45 -43
  290. package/dist/task/verify-resolution.d.ts +24 -20
  291. package/dist/task/verify-resolution.js +51 -50
  292. package/dist/task/verify-work.d.ts +59 -66
  293. package/dist/task/verify-work.js +101 -138
  294. package/dist/task/widget.d.ts +15 -14
  295. package/dist/task/widget.js +22 -17
  296. package/dist/task/wiring-claims.d.ts +25 -32
  297. package/dist/task/wiring-claims.js +30 -35
  298. package/dist/task/write-guard.d.ts +39 -39
  299. package/dist/task/write-guard.js +48 -51
  300. package/dist/task/yolo.d.ts +34 -30
  301. package/dist/task/yolo.js +42 -37
  302. package/dist/workers/abstention.d.ts +21 -41
  303. package/dist/workers/abstention.js +27 -48
  304. package/dist/workers/brave-search.d.ts +4 -3
  305. package/dist/workers/brave-search.js +5 -2
  306. package/dist/workers/brave-warning.d.ts +7 -4
  307. package/dist/workers/brave-warning.js +19 -7
  308. package/dist/workers/ddg-search.d.ts +6 -6
  309. package/dist/workers/ddg-search.js +18 -12
  310. package/dist/workers/docs-cache.js +5 -2
  311. package/dist/workers/docs-chunk.d.ts +30 -37
  312. package/dist/workers/docs-chunk.js +37 -41
  313. package/dist/workers/docs-core.d.ts +28 -44
  314. package/dist/workers/docs-core.js +25 -44
  315. package/dist/workers/docs-index.js +4 -3
  316. package/dist/workers/docs-lookup.d.ts +15 -22
  317. package/dist/workers/docs-lookup.js +12 -21
  318. package/dist/workers/docs-project.d.ts +15 -9
  319. package/dist/workers/docs-project.js +17 -10
  320. package/dist/workers/docs-resolve.d.ts +19 -20
  321. package/dist/workers/docs-resolve.js +35 -32
  322. package/dist/workers/docs-retrieve.d.ts +5 -6
  323. package/dist/workers/docs-retrieve.js +18 -15
  324. package/dist/workers/exa-search.d.ts +9 -6
  325. package/dist/workers/exa-search.js +23 -12
  326. package/dist/workers/fetch-core.d.ts +13 -16
  327. package/dist/workers/fetch-core.js +23 -23
  328. package/dist/workers/focused-extractor.d.ts +12 -12
  329. package/dist/workers/focused-extractor.js +16 -19
  330. package/dist/workers/html-clean.js +24 -14
  331. package/dist/workers/http-request.d.ts +28 -20
  332. package/dist/workers/http-request.js +22 -17
  333. package/dist/workers/npm-version.d.ts +28 -11
  334. package/dist/workers/npm-version.js +24 -15
  335. package/dist/workers/phantom-imports.d.ts +15 -12
  336. package/dist/workers/phantom-imports.js +30 -24
  337. package/dist/workers/pi-worker-core.d.ts +69 -71
  338. package/dist/workers/pi-worker-core.js +100 -109
  339. package/dist/workers/pi-worker-docs.d.ts +24 -19
  340. package/dist/workers/pi-worker-docs.js +67 -76
  341. package/dist/workers/pi-worker-fetch.d.ts +7 -3
  342. package/dist/workers/pi-worker-fetch.js +27 -19
  343. package/dist/workers/pi-worker-search.js +12 -8
  344. package/dist/workers/pi-worker.d.ts +9 -4
  345. package/dist/workers/pi-worker.js +21 -14
  346. package/dist/workers/reasoning-warning.d.ts +18 -17
  347. package/dist/workers/reasoning-warning.js +22 -20
  348. package/dist/workers/research-cache.js +50 -78
  349. package/dist/workers/search-core.js +7 -5
  350. package/dist/workers/search-types.d.ts +10 -9
  351. package/dist/workers/search-types.js +9 -8
  352. package/dist/workers/session-hint.d.ts +13 -14
  353. package/dist/workers/session-hint.js +8 -9
  354. package/dist/workers/shared.d.ts +21 -25
  355. package/dist/workers/shared.js +0 -0
  356. package/dist/workers/single-read-extension.d.ts +14 -7
  357. package/dist/workers/single-read-extension.js +14 -7
  358. package/dist/workers/single-read-guard.d.ts +25 -28
  359. package/dist/workers/single-read-guard.js +32 -32
  360. package/dist/workers/typeonly-log.d.ts +12 -9
  361. package/dist/workers/typeonly-log.js +29 -33
  362. package/dist/workers/worker-channels.d.ts +15 -23
  363. package/dist/workers/worker-channels.js +15 -23
  364. package/dist/workers/worker-failure.d.ts +38 -46
  365. package/dist/workers/worker-failure.js +31 -39
  366. package/dist/workers/worker-kill.d.ts +25 -26
  367. package/dist/workers/worker-kill.js +16 -19
  368. package/dist/workers/worker-profiles.d.ts +43 -53
  369. package/dist/workers/worker-profiles.js +30 -38
  370. package/package.json +10 -8
@@ -6,44 +6,48 @@ import { type WorkerGuardOverride, type WorkerGuardPolicy, type WorkerPolicyInpu
6
6
  * command could be cited from. `pi-worker-docs` (the primary), `read` and `grep`
7
7
  * (project source), and the web escalations `pi-worker-search`/`pi-worker-fetch`.
8
8
  *
9
- * `ls` and `find` are deliberately EXCLUDED: they return file/directory NAMES,
10
- * and APIS owns symbols by name only, never paths (RESEARCH_APIS_PROMPT). Bare
11
- * enumeration cannot verify a signature, so a worker that fabricates its section
12
- * from memory does not launder itself grounded by calling `ls` once. That
13
- * exclusion is the anti-gaming property of any gate built on this count: "one
14
- * trivial `ls` then fabricate the rest" leaves groundingRetrievalCount at 0.
9
+ * `ls` and `find` are deliberately EXCLUDED: they return file and directory
10
+ * NAMES, and APIS owns symbols by name only, never paths (see
11
+ * RESEARCH_APIS_PROMPT in prompts.ts). Bare enumeration cannot verify a
12
+ * signature, so a worker that fabricates its section from memory cannot launder
13
+ * itself grounded by calling `ls` once "one trivial `ls` then fabricate the
14
+ * rest" still leaves groundingRetrievalCount at 0.
15
+ *
16
+ * The set is DERIVED from WORKER_CHANNELS (worker-channels.ts), not hand-kept.
17
+ * Re-exported here only so worker-channels.test.ts can assert this module hands
18
+ * back the same predicate.
15
19
  */
16
20
  export { isGroundingRetrieval } from './worker-channels.js';
17
21
  /**
18
22
  * Does this partial output carry ANSWER CONTENT, or is it the model clearing its
19
23
  * throat?
20
24
  *
21
- * Salvage originally kept the LONGEST partial, which is not the same question. On
22
- * the live carry arm, TASK_0020 and TASK_0021 both timed out on all three
23
- * attempts and salvage shipped this as the section:
25
+ * Keeping the LONGEST partial is not the same question, and it has an obvious
26
+ * failure: a preamble sentence like
24
27
  *
25
28
  * "Now let me get more details on the specific APIs and components I need:"
26
29
  *
27
- * — a preamble sentence, which beats an empty string on length and carries
28
- * nothing. Both trials scored 2 entries and DEGRADED, against 22 and 5 for the
29
- * same fixtures in baseline.
30
+ * beats an empty string on length and carries nothing.
30
31
  *
31
32
  * A research worker's answer is a list of lines that each name something and
32
- * describe it. The test is therefore structural, not lexical: at least two lines
33
- * that look like entries — a name, then a gap, then a description. Prose wraps
34
- * at no particular column and does not repeat that shape.
33
+ * describe it. The test is therefore structural, not lexical: at least TWO lines
34
+ * that look like entries — a name, then a gap, then a description. Prose wraps at
35
+ * no particular column and does not repeat that shape; the sentence above scores
36
+ * zero entry lines.
35
37
  */
36
38
  export declare function hasAnswerContent(text: string): boolean;
37
39
  /**
38
40
  * Is ONE line an entry — a name, a gap, then a description — rather than prose?
39
41
  *
40
- * Split out of `hasAnswerContent` so the same rule can decide what a line IS,
41
- * not just how many of them there are. A FILES section's paths are read back
42
- * with it, and a scorer that used its own idea of an entry counted a preamble
43
- * sentence and a leaked `</tool_call>` as invented paths.
42
+ * Split out of `hasAnswerContent` so the same rule can decide what a line IS, not
43
+ * just how many of them there are a reader of a FILES section needs the same
44
+ * test, and its own idea of an entry would count a preamble sentence or a leaked
45
+ * `</tool_call>` as one.
44
46
  *
45
- * Prose wraps at no particular column, so it carries no two-space gap and no
46
- * spaced dash; when it does, it ends in `.` or `:` and an entry does not.
47
+ * A leading `-`, `*`, `•` or `1.`/`1)` bullet is stripped first. What remains must
48
+ * hold a two-space gap or a spaced dash and must NOT end in `.` or `:`. Prose
49
+ * wraps at no particular column, so it carries neither; when it does carry one, it
50
+ * ends in punctuation and an entry does not.
47
51
  */
48
52
  export declare function isEntryLine(raw: string): boolean;
49
53
  /**
@@ -67,7 +71,7 @@ export interface RunWorkerInput {
67
71
  /** Called for each tool execution start and text-writing event inside the worker. */
68
72
  onLine?: (line: string) => void;
69
73
  /** Called when a tool call FINISHES, with its (truncatable) result — lets a caller
70
- * log tool OUTPUTS, not just the command (mx5 run 10 item 6). */
74
+ * log tool OUTPUTS, not just the command. */
71
75
  onToolResult?: (result: {
72
76
  name: string;
73
77
  isError: boolean;
@@ -85,34 +89,31 @@ export interface RunWorkerInput {
85
89
  * The worker child's context window in tokens, or `'unknown'` when the
86
90
  * caller genuinely has none.
87
91
  *
88
- * REQUIRED, and required for the same reason `profile` below is. This was
89
- * `contextWindow?: number`, and two of the three production call sites simply
90
- * did not write it: `pi-worker.ts` and `research-worker.ts`. pi's event
91
- * stream carries no window (issue #16), so `noteContext` only ever saw 0, and
92
- * `StallDetector`'s CONTEXT CHURN rule gated on a positive window
93
- * (`stall-detector.ts`) could never fire for the ad-hoc worker or for any
94
- * of the four research workers. Nothing was red. The optional was the whole
95
- * defect: a rule that silently does not exist reads exactly like a rule that
96
- * exists and did not trip.
92
+ * REQUIRED, and required for the same reason `profile` below is. pi's event
93
+ * stream carries NO window the string `context_usage` appears nowhere in any
94
+ * installed @earendil-works package, and its only usage-bearing JSON event is
95
+ * `message_update` so this parameter is the only source there is. Left
96
+ * optional, a caller that omits it leaves `noteContext` seeing 0, and
97
+ * `StallDetector`'s CONTEXT CHURN rule is gated on a positive window, so the
98
+ * rule silently does not exist. That reads exactly like a rule that exists and
99
+ * did not trip.
97
100
  *
98
- * WHY A WORD AND NOT `0` OR `null`. Both of those are what a caller types
99
- * when it has not thought about the question, and both disarm the rule
100
- * silently — which is the state this replaces. `'unknown'` cannot be typed by
101
- * accident, is greppable, and shows up in a diff as a decision.
101
+ * WHY A WORD AND NOT `0` OR `null`. Both of those are what a caller types when
102
+ * it has not thought about the question, and both disarm the rule silently.
103
+ * `'unknown'` cannot be typed by accident, is greppable, and shows up in a diff
104
+ * as a decision.
102
105
  *
103
106
  * Two consumers read it: the churn rule, and the caller's progress bar, which
104
- * shows a bare token count without a window. Both degrade exactly as before
105
- * on `'unknown'`.
107
+ * shows a bare token count without a window. Both degrade on `'unknown'`.
106
108
  */
107
109
  contextWindow: number | 'unknown';
108
110
  /**
109
111
  * WHICH KIND of worker child this is — the whole guard policy, in one word.
110
112
  *
111
- * REQUIRED, and required on purpose. The ten guard knobs this replaces used
112
- * to sit here as independent optionals, so a caller that named none of them
113
- * still got a full policy and nobody could see which one. That is how the
114
- * ad-hoc `pi-worker` tool came to run the strictest wall clock of the three
115
- * children without anyone deciding it should. See worker-profiles.ts.
113
+ * REQUIRED, and required on purpose. As independent optionals, a caller that
114
+ * named none of the guard knobs still got a full policy and nobody could see
115
+ * which one — so a child can end up running the strictest wall clock of the
116
+ * three without anyone deciding it should. See worker-profiles.ts.
116
117
  */
117
118
  profile: WorkerProfileId;
118
119
  /**
@@ -121,10 +122,11 @@ export interface RunWorkerInput {
121
122
  */
122
123
  policyInputs?: WorkerPolicyInputs;
123
124
  /**
124
- * Whole guard rows laid over the profile's. TESTS AND A/B HARNESSES ONLY —
125
- * an override at a production call site is the hand-picked subset this
126
- * design exists to stop, and `worker-profiles.test.ts` fails the build if
127
- * one appears under src/ outside a test.
125
+ * Whole guard rows laid over the profile's. TESTS AND HARNESSES ONLY — an
126
+ * override at a production call site is the hand-picked subset this design
127
+ * exists to stop. `worker-profiles.test.ts` enforces it: its "no production
128
+ * source file passes an `override` to runWorker" test scans src/ for a leading
129
+ * `override:` and fails on any hit.
128
130
  */
129
131
  override?: WorkerGuardOverride;
130
132
  /**
@@ -133,9 +135,9 @@ export interface RunWorkerInput {
133
135
  *
134
136
  * WHY: asserting that a profile RESOLVES correctly proves nothing about
135
137
  * whether runWorker then READS it correctly — a rewiring that turns "0 means
136
- * off" into "0 means on" leaves every profile assertion green. This hook is
137
- * what lets a caller's own test (gate-child.test.ts) drive the REAL call
138
- * site and check the REAL policy, instead of re-typing the table.
138
+ * off" into "0 means on" leaves every profile assertion green. This hook lets
139
+ * a test drive the REAL call site and read back the REAL policy instead of
140
+ * re-typing the table; worker-profiles.test.ts is where those assertions live.
139
141
  */
140
142
  onPolicy?: (policy: WorkerGuardPolicy) => void;
141
143
  /** Backoff sleep, injectable so tests don't wait out the real delays. */
@@ -168,12 +170,10 @@ export interface RunWorkerInput {
168
170
  *
169
171
  * WHY: every restart branch below throws away a whole attempt's wall clock
170
172
  * along with its text, and `waitMs`/`workMs` describe the FINAL attempt only.
171
- * With no hook here those attempts were structurally invisible: mx5 run 18
172
- * burned 30 wall-clock timeouts / 120 minutes of compute that appeared in no
173
- * log and no timing widget, and 21 of the 23 affected workers reported
174
- * `exit=0` — clean successes as far as the run could tell. The discrepancy
175
- * was only recoverable by subtracting reported wait+work from the timestamps
176
- * of the `start` and `done` lines around it.
173
+ * With no hook here a discarded attempt is structurally invisible the worker
174
+ * still returns `exitCode` 0 and reads as a clean success, and the lost time is
175
+ * recoverable only by subtracting the reported wait+work from the timestamps
176
+ * around the call.
177
177
  */
178
178
  onRestart?: (restart: WorkerRestart) => void;
179
179
  }
@@ -200,13 +200,9 @@ export interface WorkerRestart {
200
200
  * thrown away.
201
201
  *
202
202
  * Recorded whatever `carryForward` says, because the DISCARD is the thing a
203
- * reader cannot otherwise see. A restart line reported how long an attempt
204
- * ran and why it died, and never what died with it so "the guards worked
205
- * and the run still returned 52 characters" and "the guards worked and the
206
- * run threw away a finished answer" print identically. Measured on the
207
- * ad-hoc `pi-worker` corpus: T024 lost an attempt to a dropped model socket
208
- * at 275s and returned 52 chars over 620s; nothing in the run said whether
209
- * those 275s held anything.
203
+ * reader cannot otherwise see. Without it, "the guards worked and the run
204
+ * returned almost nothing" and "the guards worked and the run threw away a
205
+ * finished answer" print identically.
210
206
  *
211
207
  * It is an OBSERVATION, not a decision: harvesting into `salvage` is still
212
208
  * gated on the profile, and this number changes no behaviour.
@@ -222,9 +218,9 @@ export interface RunWorkerResult {
222
218
  * The provider-reported cause when the model turn itself failed (disconnect,
223
219
  * fetch failed, 5xx after pi's own retries): pi delivers it as an assistant
224
220
  * message with stopReason "error" and EMPTY text, exit code 0. Phase children
225
- * have always surfaced this (child-runner.ts) research workers did not, so a
226
- * swallowed provider error reached the caller as an indistinguishable empty
227
- * answer and was reported as the useless "produced no output" (issue #10).
221
+ * surface this through child-runner.ts; without it a swallowed provider error
222
+ * reaches the caller as an indistinguishable empty answer and gets reported as
223
+ * the useless "produced no output".
228
224
  * Only meaningful when `text` is empty: a turn that produced text after pi
229
225
  * recovered is a success, and the first-error capture must not relabel it.
230
226
  */
@@ -349,9 +345,9 @@ export interface RunWorkerResult {
349
345
  * commandTimeoutHint, which tells the model in as many words to bound its
350
346
  * command; a SECOND hang means it ignored an explicit instruction, and a third
351
347
  * means it ignored it twice. Giving a non-complying child the full ceiling again
352
- * would put the worst case at 3 × 15 min = 45 minutes of dead time, resting
353
- * entirely on the model obeying prose. Halving bounds it at ~26 min while
354
- * costing a complying child nothing.
348
+ * makes the worst case three times the ceiling, resting entirely on the model
349
+ * obeying prose. Halving bounds it at under twice the ceiling while costing a
350
+ * complying child nothing.
355
351
  *
356
352
  * `priorHangs` counts watchdog kills specifically, NOT total restarts — the
357
353
  * restart budget is shared with loop kills, and a child restarted for LOOPING
@@ -359,8 +355,9 @@ export interface RunWorkerResult {
359
355
  * the full ceiling. Only a hang after a hang is defiance.
360
356
  *
361
357
  * Floored at 30s so repeated halving cannot shrink the ceiling to something no
362
- * real command could finish inside — but never ABOVE the configured ceiling
363
- * itself, or a caller asking for 10s would silently get 30.
358
+ * real command could finish inside — but the floor is `min(base, 30s)`, never
359
+ * above the configured ceiling, so a caller asking for 10s keeps 10s at every
360
+ * hang count. A base of 0 or less disables the watchdog and stays 0.
364
361
  */
365
362
  export declare function commandCeilingForAttempt(baseMs: number, priorHangs: number): number;
366
363
  /**
@@ -377,7 +374,8 @@ interface RestartState {
377
374
  timedOut: boolean;
378
375
  modelError?: string;
379
376
  leaked: string | null;
380
- /** The cap this attempt actually died against the SCALE arm moves it. */
377
+ /** The cap this attempt actually died against, not the configured one:
378
+ * `extend`/`progress` can push the deadline out during the attempt. */
381
379
  effectiveCapMs: number;
382
380
  /** The child's tool string, which decides whether its edits can persist. */
383
381
  tools: string;
@@ -10,15 +10,9 @@ import { detectLeakedToolCall, leakedToolCallHint, MAX_LEAK_RETRIES } from '../s
10
10
  import { discoverModelEndpoints, probeModelEndpoints } from '../shared/model-endpoint.js';
11
11
  import { streamStallHint } from '../shared/stream-watchdog.js';
12
12
  import { classifyWorkerFailure } from './worker-failure.js';
13
- import { CARRY_FORWARD_IDS } from './worker-kill.js';
13
+ import { CARRY_FORWARD_IDS, RESTART_ORDER } from './worker-kill.js';
14
14
  import { applyOverride, WORKER_PROFILES } from './worker-profiles.js';
15
- // `--mode json` makes pi emit structured events as they happen instead of
16
- // buffering the assistant text and flushing on exit. That matters for the
17
- // wait/work timing split: in text mode the first stdout chunk only arrives at
18
- // the very end, so onFirstByte fires moments before close and workMs is
19
- // effectively zero. With JSON events the first byte lands as soon as the
20
- // model starts producing — making waitMs the real queue/cold-start cost and
21
- // workMs the real generation+tool-call cost.
15
+ /** The tool whitelist a caller gets when it names none. */
22
16
  const DEFAULT_TOOLS = 'read,grep,find,ls';
23
17
  /**
24
18
  * The one place `'unknown'` becomes the 0 both consumers already treat as
@@ -33,18 +27,19 @@ function contextWindowTokens(cw) {
33
27
  * command could be cited from. `pi-worker-docs` (the primary), `read` and `grep`
34
28
  * (project source), and the web escalations `pi-worker-search`/`pi-worker-fetch`.
35
29
  *
36
- * `ls` and `find` are deliberately EXCLUDED: they return file/directory NAMES,
37
- * and APIS owns symbols by name only, never paths (RESEARCH_APIS_PROMPT). Bare
38
- * enumeration cannot verify a signature, so a worker that fabricates its section
39
- * from memory does not launder itself grounded by calling `ls` once. That
40
- * exclusion is the anti-gaming property of any gate built on this count: "one
41
- * trivial `ls` then fabricate the rest" leaves groundingRetrievalCount at 0.
30
+ * `ls` and `find` are deliberately EXCLUDED: they return file and directory
31
+ * NAMES, and APIS owns symbols by name only, never paths (see
32
+ * RESEARCH_APIS_PROMPT in prompts.ts). Bare enumeration cannot verify a
33
+ * signature, so a worker that fabricates its section from memory cannot launder
34
+ * itself grounded by calling `ls` once "one trivial `ls` then fabricate the
35
+ * rest" still leaves groundingRetrievalCount at 0.
36
+ *
37
+ * The set is DERIVED from WORKER_CHANNELS (worker-channels.ts), not hand-kept.
38
+ * Re-exported here only so worker-channels.test.ts can assert this module hands
39
+ * back the same predicate.
42
40
  */
43
- // The grounding set is derived from WORKER_CHANNELS (worker-channels.ts), not
44
- // hand-kept — this was a second copy of the four tool names. Re-exported because
45
- // several call sites and tests import it from here.
46
41
  export { isGroundingRetrieval } from './worker-channels.js';
47
- // RESEARCH_WORKER_TIMEOUT_MS and STALL_AFTER_MS live on the profile table now
42
+ // RESEARCH_WORKER_TIMEOUT_MS and STALL_AFTER_MS live on the profile table
48
43
  // (worker-profiles.ts): they are the default VALUES of two guard rows, and a
49
44
  // default that lives apart from the table stating it is a second place to look.
50
45
  /**
@@ -59,49 +54,44 @@ const WORKER_TIMEOUT_HINT = '[SYSTEM NOTE: Your previous attempt ran out of time
59
54
  /**
60
55
  * How much of a discarded attempt's answer is carried into the next one.
61
56
  *
62
- * A restart used to hand the re-spawn nothing but a hint — which is why
63
- * WORKER_TIMEOUT_HINT above can tell a worker "do not re-explore ground you have
64
- * already covered" while giving it no record of what that ground was. It could
65
- * not comply. mx5 run 18 shows the cost: on tasks with >=46 project-source
66
- * lookups, 5 of 5 workers burned the FULL restart budget, because every attempt
67
- * re-read the same files against the same clock and died in the same place.
57
+ * A restart that hands the re-spawn nothing but a hint is why WORKER_TIMEOUT_HINT
58
+ * above can tell a worker "do not re-explore ground you have already covered"
59
+ * while giving it no record of what that ground was. It cannot comply, so it
60
+ * re-reads the same files against the same clock and dies in the same place.
68
61
  *
69
- * Carrying the partial answer forward is what makes a restart converge instead
70
- * of repeat. The risk it takes is real and is the thing the A/B measures: a
71
- * half-written or speculative entry, replayed under "already established", is
72
- * exactly how a fabrication gets laundered into a final answer. That is what the
73
- * ungrounded-symbol and anti-synthesis guards are pointed at, so the carry is
74
- * framed as findings to VERIFY-or-DROP rather than as settled fact.
62
+ * Carrying the partial forward is what lets a restart converge instead of repeat.
63
+ * The risk is real: a half-written or speculative entry, replayed under "already
64
+ * established", is how a fabrication gets laundered into a final answer. That is
65
+ * why `formatCarryForward` frames it as findings to VERIFY-or-DROP rather than as
66
+ * settled fact.
75
67
  */
76
68
  const CARRY_FORWARD_LIMIT = 24_000;
77
69
  /**
78
70
  * Restart reasons whose partial output is worth keeping.
79
71
  *
80
- * A clock kill (`worker-timeout`), a hung tool (`command-timeout`), an idle
81
- * stream (`stream-stall`) and a dropped socket (`connection-error`) all discard
82
- * work the model genuinely did. A loop kill and a leaked tool call do not — the
83
- * first is by definition the same call repeated, the second is malformed
84
- * protocol text, and replaying either would feed the failure back to itself.
72
+ * Exactly four, derived from WORKER_KILLS: `command-timeout`, `stream-stall`,
73
+ * `worker-timeout` and `connection-error` all discard work the model genuinely
74
+ * did. A loop kill and a leaked tool call do not — the first is by definition the
75
+ * same call repeated, the second is malformed protocol text, and replaying either
76
+ * would feed the failure back to itself.
85
77
  */
86
78
  const CARRY_FORWARD_REASONS = CARRY_FORWARD_IDS;
87
79
  /**
88
80
  * Does this partial output carry ANSWER CONTENT, or is it the model clearing its
89
81
  * throat?
90
82
  *
91
- * Salvage originally kept the LONGEST partial, which is not the same question. On
92
- * the live carry arm, TASK_0020 and TASK_0021 both timed out on all three
93
- * attempts and salvage shipped this as the section:
83
+ * Keeping the LONGEST partial is not the same question, and it has an obvious
84
+ * failure: a preamble sentence like
94
85
  *
95
86
  * "Now let me get more details on the specific APIs and components I need:"
96
87
  *
97
- * — a preamble sentence, which beats an empty string on length and carries
98
- * nothing. Both trials scored 2 entries and DEGRADED, against 22 and 5 for the
99
- * same fixtures in baseline.
88
+ * beats an empty string on length and carries nothing.
100
89
  *
101
90
  * A research worker's answer is a list of lines that each name something and
102
- * describe it. The test is therefore structural, not lexical: at least two lines
103
- * that look like entries — a name, then a gap, then a description. Prose wraps
104
- * at no particular column and does not repeat that shape.
91
+ * describe it. The test is therefore structural, not lexical: at least TWO lines
92
+ * that look like entries — a name, then a gap, then a description. Prose wraps at
93
+ * no particular column and does not repeat that shape; the sentence above scores
94
+ * zero entry lines.
105
95
  */
106
96
  export function hasAnswerContent(text) {
107
97
  return text.split('\n').filter(isEntryLine).length >= 2;
@@ -109,13 +99,15 @@ export function hasAnswerContent(text) {
109
99
  /**
110
100
  * Is ONE line an entry — a name, a gap, then a description — rather than prose?
111
101
  *
112
- * Split out of `hasAnswerContent` so the same rule can decide what a line IS,
113
- * not just how many of them there are. A FILES section's paths are read back
114
- * with it, and a scorer that used its own idea of an entry counted a preamble
115
- * sentence and a leaked `</tool_call>` as invented paths.
102
+ * Split out of `hasAnswerContent` so the same rule can decide what a line IS, not
103
+ * just how many of them there are a reader of a FILES section needs the same
104
+ * test, and its own idea of an entry would count a preamble sentence or a leaked
105
+ * `</tool_call>` as one.
116
106
  *
117
- * Prose wraps at no particular column, so it carries no two-space gap and no
118
- * spaced dash; when it does, it ends in `.` or `:` and an entry does not.
107
+ * A leading `-`, `*`, `•` or `1.`/`1)` bullet is stripped first. What remains must
108
+ * hold a two-space gap or a spaced dash and must NOT end in `.` or `:`. Prose
109
+ * wraps at no particular column, so it carries neither; when it does carry one, it
110
+ * ends in punctuation and an entry does not.
119
111
  */
120
112
  export function isEntryLine(raw) {
121
113
  const l = raw.replace(/^\s*(?:[-*•]|\d+[.)])\s+/, '').trim();
@@ -183,10 +175,10 @@ absoluteCeilingMs) {
183
175
  return {
184
176
  signal: ctrl.signal,
185
177
  timedOut: () => timedOut,
186
- // SCALE arm of nexttask 5B, inert unless a caller calls it: push the
187
- // deadline out, never past `started + ceilingMs`. A disabled timeout
188
- // (nothing armed) stays disabled — extending "never" is meaningless — and
189
- // an already-fired timer is not resurrected.
178
+ // Push the deadline out, never past `started + ceilingMs`. Inert unless a
179
+ // caller calls it. A disabled timeout (nothing armed) stays disabled
180
+ // extending "never" is meaningless — and an already-fired timer is not
181
+ // resurrected.
190
182
  extend: (byMs, ceilingMs) => {
191
183
  if (!armed || timedOut || ctrl.signal.aborted)
192
184
  return;
@@ -232,9 +224,9 @@ absoluteCeilingMs) {
232
224
  * commandTimeoutHint, which tells the model in as many words to bound its
233
225
  * command; a SECOND hang means it ignored an explicit instruction, and a third
234
226
  * means it ignored it twice. Giving a non-complying child the full ceiling again
235
- * would put the worst case at 3 × 15 min = 45 minutes of dead time, resting
236
- * entirely on the model obeying prose. Halving bounds it at ~26 min while
237
- * costing a complying child nothing.
227
+ * makes the worst case three times the ceiling, resting entirely on the model
228
+ * obeying prose. Halving bounds it at under twice the ceiling while costing a
229
+ * complying child nothing.
238
230
  *
239
231
  * `priorHangs` counts watchdog kills specifically, NOT total restarts — the
240
232
  * restart budget is shared with loop kills, and a child restarted for LOOPING
@@ -242,8 +234,9 @@ absoluteCeilingMs) {
242
234
  * the full ceiling. Only a hang after a hang is defiance.
243
235
  *
244
236
  * Floored at 30s so repeated halving cannot shrink the ceiling to something no
245
- * real command could finish inside — but never ABOVE the configured ceiling
246
- * itself, or a caller asking for 10s would silently get 30.
237
+ * real command could finish inside — but the floor is `min(base, 30s)`, never
238
+ * above the configured ceiling, so a caller asking for 10s keeps 10s at every
239
+ * hang count. A base of 0 or less disables the watchdog and stays 0.
247
240
  */
248
241
  export function commandCeilingForAttempt(baseMs, priorHangs) {
249
242
  if (!(baseMs > 0))
@@ -323,8 +316,8 @@ export const RESTART_RULES = [
323
316
  // a loop also tripped — the loop hint above is more specific.
324
317
  reason: 'worker-timeout',
325
318
  detect: s => s.timedOut && !s.loopHit && s.restartBudgetSpent < MAX_LOOP_RESTARTS ?
326
- // The EFFECTIVE cap, which the SCALE arm moves reporting the
327
- // configured one would misname why this attempt died.
319
+ // The EFFECTIVE cap, which `extend`/`progress` can have moved
320
+ // reporting the configured one would misname why this attempt died.
328
321
  { detail: `cap ${s.effectiveCapMs}ms` }
329
322
  : null,
330
323
  hint: () => WORKER_TIMEOUT_HINT,
@@ -332,22 +325,15 @@ export const RESTART_RULES = [
332
325
  },
333
326
  {
334
327
  // A connection-class model error is restartable on the same budget, exactly
335
- // as runPhaseChild already treats it a research worker had no such
336
- // retry, so one dropped fetch failed the whole task at research while the
337
- // identical blip in refine/compose was absorbed.
328
+ // as runPhaseChild already treats it. Without it one dropped socket fails
329
+ // the whole task at research, while the identical blip in refine or compose
330
+ // is absorbed.
338
331
  //
339
- // What this can and cannot buy, measured (flaky proxy in front of the local
340
- // llama-server, dropping every connection for a fixed outage window): pi
341
- // retries a failed turn itself, 4 attempts over ~15s, and a run that
342
- // recovers no longer reports modelError at all (see JsonEventSink). So a
343
- // surfaced connection error means pi's own ~15s budget is already spent, and
344
- // a re-spawn only helps when the outage outlasts it. It does: at a 20s
345
- // outage the baseline never recovered and this policy always did, 0/8 → 8/8
346
- // (Fisher p=0.00016), and the same at 35s. Below ~15s pi absorbs it alone —
347
- // 8/8 both arms, so the retry neither helps nor costs there. Beyond ~46s
348
- // (three spawns' combined budget) both arms fail. The price is paid only on
349
- // a backend that is really gone: time-to-report goes ~15s → ~46s. Re-run:
350
- // scripts/connection-retry-ab.ts.
332
+ // pi retries a failed turn itself before reporting anything, and a run that
333
+ // recovers reports no modelError at all (see JsonEventSink). So a SURFACED
334
+ // connection error means pi's own budget is already spent, and a re-spawn
335
+ // only helps when the outage outlasts it. The price is paid only on a
336
+ // backend that is really gone: time-to-report grows by the extra spawns.
351
337
  //
352
338
  // Connection class ONLY. Auth, bad request and context overflow still fail
353
339
  // fast: re-issuing the same request cannot fix them, so spending the budget
@@ -380,11 +366,11 @@ export const RESTART_RULES = [
380
366
  * runChild turns into a process-GROUP kill — reaping the hung command itself,
381
367
  * not just the pi child holding it.
382
368
  *
383
- * LIMIT: the group kill only reaches processes still IN the group. A hung
384
- * command that detached a daemon (setsid/nohup dev server) leaves it running —
385
- * the fresh attempt can then hit a port the dead attempt's escapee still holds
386
- * (the run-9 orphan-dev-server false-EADDRINUSE shape). No cheap fix from
387
- * here; the restart hint's "check current state" line is the mitigation.
369
+ * LIMIT: the group kill only reaches processes still IN the group. A hung command
370
+ * that detached a daemon (setsid, nohup, a background dev server) leaves it
371
+ * running, so the fresh attempt can hit a port the dead attempt's escapee still
372
+ * holds. There is no cheap fix from here; the restart hint's "check current state"
373
+ * line is the mitigation.
388
374
  *
389
375
  * Returns null when the watchdog is off, so the caller keeps the plain timeout
390
376
  * signal and no per-call bookkeeping happens at all.
@@ -428,6 +414,13 @@ function commandWatch(timeoutMs) {
428
414
  }
429
415
  export async function runWorker(input) {
430
416
  const tools = input.tools ?? DEFAULT_TOOLS;
417
+ // `--mode json` makes pi emit structured events as they happen instead of
418
+ // buffering the assistant text and flushing on exit. Its print-mode source
419
+ // shows both halves: under `json` a session subscriber writes every event to
420
+ // stdout as it arrives, while under `text` NOTHING is written until after the
421
+ // prompt resolves, when the last assistant message is printed once. That is
422
+ // what makes the wait/work split real — onFirstByte would otherwise fire
423
+ // moments before close and leave workMs at nearly zero.
431
424
  const baseArgs = [
432
425
  ...childBaseArgs(input.extensions ?? []),
433
426
  ...(input.thinking ?? []),
@@ -475,11 +468,10 @@ export async function runWorker(input) {
475
468
  const salvage = { text: null };
476
469
  for (;;) {
477
470
  const carried = salvage.text === null ? null : formatCarryForward(salvage.text);
478
- // Announce the INJECTION, not just the restart. Without this, "the carry
479
- // reached the re-spawn" can only be inferred from entry countsand
480
- // inferring what a worker did from what it produced is the exact gap 5A
481
- // exists to close. The prompt goes to the child on stdin, so no log
482
- // downstream of here can show it.
471
+ // Announce the INJECTION, not just the restart. The prompt goes to the child
472
+ // on stdin, so no log downstream of here can show it without this hook
473
+ // "the carry reached the re-spawn" could only be inferred from the answer,
474
+ // which is inferring what a worker did from what it produced.
483
475
  if (carried !== null) {
484
476
  input.onCarryForward?.({
485
477
  attempt: restarts.length + 1,
@@ -506,8 +498,8 @@ export async function runWorker(input) {
506
498
  null
507
499
  : new StallDetector(guards.loop.progress.limit, guards.loop.progress.churnFactor);
508
500
  // Arm the churn rule BEFORE the first tool call. pi's stream carries no
509
- // context event (issue #16), so waiting for one leaves the rule
510
- // permanently disarmed. The parent knows the window at spawn time.
501
+ // context event at all, so waiting for one leaves the rule permanently
502
+ // disarmed. The parent knows the window at spawn time.
511
503
  stallDetector?.noteContext(contextWindowTokens(input.contextWindow));
512
504
  // Capture the hit the detector reports (it also returns it to the unified
513
505
  // runner, which kills the child on a hit). Without capturing it here the
@@ -546,8 +538,9 @@ export async function runWorker(input) {
546
538
  // A tool call is the worker working. Inert unless the
547
539
  // caller opted into a progress-based deadline.
548
540
  timeout.progress();
549
- // The generic child runner used to name ONE tool and ONE of
550
- // its parameters here. It asks the tool's own row now.
541
+ // Naming ONE tool and ONE of its parameters here would put
542
+ // that knowledge in the generic runner, so it asks the
543
+ // tool's own row in WORKER_CHANNELS instead.
551
544
  if (clock.fanout
552
545
  && workerChannel(call.name)?.isProjectSourceLookup?.(call.args ?? {}) === true) {
553
546
  timeout.extend(clock.fanout.perLookupMs, clock.fanout.ceilingMs);
@@ -570,11 +563,11 @@ export async function runWorker(input) {
570
563
  timeout.progress();
571
564
  input.onLine?.(line);
572
565
  },
573
- // Always wired now (it used to be conditional on the command
574
- // watchdog): the sink only emits tool_execution_end if a
575
- // handler exists, and a completed tool call is the clearest
576
- // progress signal there is. Without it a worker whose tool
577
- // calls all succeed would still look idle to the deadline.
566
+ // Always wired, never conditional on the command watchdog: the
567
+ // sink only emits a tool-execution-end if a handler exists, and
568
+ // a completed tool call is the clearest progress signal there
569
+ // is. Without it a worker whose tool calls all succeed would
570
+ // still look idle to the deadline.
578
571
  onToolResult: r => {
579
572
  timeout.progress();
580
573
  cmdWatch?.onEnd(r.toolCallId);
@@ -637,7 +630,7 @@ export async function runWorker(input) {
637
630
  // there would just mislabel the real failure.
638
631
  const leaked = result.exitCode === 0 && !result.aborted ? detectLeakedToolCall(text) : null;
639
632
  // THE RESTART LADDER. Precedence is RESTART_RULES' row order; this loop
640
- // owns the ritual every rule used to repeat: budget, hint, counters,
633
+ // owns the ritual every rule would otherwise repeat: budget, hint, counters,
641
634
  // record-and-announce, backoff, re-spawn.
642
635
  const state = {
643
636
  ...(loopHit ? { loopHit } : {}),
@@ -678,24 +671,22 @@ export async function runWorker(input) {
678
671
  }
679
672
  if (restarted)
680
673
  continue;
681
- // SALVAGE. The run used to return the LAST attempt's text unconditionally,
682
- // so a worker whose final attempt was killed early reported nothing at all
683
- // — even when a discarded attempt had produced a usable answer that was
684
- // still in hand at the moment it was thrown away. A restart budget is
685
- // meant to buy more chances at an answer, not to overwrite a good attempt
686
- // with a worse one.
674
+ // SALVAGE. Returning the LAST attempt's text unconditionally makes a
675
+ // worker whose final attempt was killed early report nothing at all — even
676
+ // when a discarded attempt produced a usable answer that was still in hand
677
+ // at the moment it was thrown away. A restart budget is meant to buy more
678
+ // chances at an answer, not to overwrite a good attempt with a worse one.
687
679
  //
688
680
  // Gated on the final attempt having FAILED, not on it being shorter. A
689
681
  // worker that finished cleanly has answered, and a short answer is a
690
682
  // legitimate answer — length would let a long half-finished fragment
691
683
  // override a concise correct one, which is the opposite of the fix.
692
- // ASK THE LADDER — do not restate it. This was an eight-term disjunction,
693
- // a fifth hand-written statement of the taxonomy `worker-failure.ts` exists
694
- // to own, and it had already drifted: `leakedToolCall` and a plain non-zero
695
- // `exitCode` are rows in FAILURE_RULES and were missing here. Both are cases
696
- // where an attempt that produced nothing usable counted as NOT failed, so
697
- // salvage was skipped and a good earlier partial was overwritten — the exact
698
- // outcome the comment above forbids.
684
+ // ASK THE LADDER — do not restate it. Hand-writing this test restates the
685
+ // taxonomy `worker-failure.ts` owns, and a restatement drifts: drop
686
+ // `leakedToolCall` or a plain non-zero `exitCode` from it and an attempt
687
+ // that produced nothing usable counts as NOT failed, so salvage is skipped
688
+ // and a good earlier partial is overwritten the outcome the comment above
689
+ // forbids.
699
690
  //
700
691
  // The two non-kill terms stay explicit because `worker-failure.ts`
701
692
  // deliberately excludes them as CONSUMER policy: an empty answer and a