@mjasnikovs/pi-task 0.38.28 → 0.38.30

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (370) hide show
  1. package/dist/config/config.d.ts +70 -70
  2. package/dist/config/config.js +26 -35
  3. package/dist/config/extension-list.d.ts +6 -5
  4. package/dist/config/extension-list.js +3 -2
  5. package/dist/config/reasoning-args.d.ts +9 -7
  6. package/dist/config/reasoning-args.js +12 -10
  7. package/dist/config/reasoning.d.ts +44 -105
  8. package/dist/config/reasoning.js +27 -704
  9. package/dist/config/register.d.ts +34 -48
  10. package/dist/config/register.js +41 -51
  11. package/dist/config/tool-list.d.ts +16 -16
  12. package/dist/config/tool-list.js +1 -1
  13. package/dist/remote/bridge.d.ts +19 -10
  14. package/dist/remote/bridge.js +3 -2
  15. package/dist/remote/broadcast.js +3 -1
  16. package/dist/remote/events.js +12 -11
  17. package/dist/remote/history.d.ts +1 -1
  18. package/dist/remote/protocol.d.ts +6 -3
  19. package/dist/remote/protocol.js +2 -1
  20. package/dist/remote/push.d.ts +16 -16
  21. package/dist/remote/push.js +27 -27
  22. package/dist/remote/register.d.ts +3 -3
  23. package/dist/remote/register.js +17 -19
  24. package/dist/remote/server.d.ts +9 -8
  25. package/dist/remote/server.js +15 -14
  26. package/dist/remote/session-state.d.ts +5 -4
  27. package/dist/remote/session-state.js +8 -5
  28. package/dist/remote/sw.d.ts +7 -6
  29. package/dist/remote/sw.js +7 -6
  30. package/dist/remote/tailscale.d.ts +4 -2
  31. package/dist/remote/tailscale.js +4 -2
  32. package/dist/remote/ui-highlight.js +6 -5
  33. package/dist/remote/ui-render.js +4 -4
  34. package/dist/remote/ui-script.js +24 -24
  35. package/dist/remote/ui-styles.d.ts +1 -1
  36. package/dist/remote/ui-styles.js +10 -13
  37. package/dist/remote/ui-tools.js +9 -6
  38. package/dist/shared/child-extensions.d.ts +29 -17
  39. package/dist/shared/child-extensions.js +29 -17
  40. package/dist/shared/child-output.d.ts +30 -24
  41. package/dist/shared/child-output.js +25 -17
  42. package/dist/shared/child-process.d.ts +47 -40
  43. package/dist/shared/child-process.js +50 -59
  44. package/dist/shared/command-watchdog.d.ts +22 -16
  45. package/dist/shared/command-watchdog.js +28 -21
  46. package/dist/shared/fs-text.d.ts +16 -10
  47. package/dist/shared/fs-text.js +16 -10
  48. package/dist/shared/git-runner.d.ts +25 -25
  49. package/dist/shared/git-runner.js +25 -25
  50. package/dist/shared/leaked-tool-call.d.ts +17 -11
  51. package/dist/shared/leaked-tool-call.js +23 -15
  52. package/dist/shared/model-endpoint.d.ts +29 -16
  53. package/dist/shared/model-endpoint.js +33 -21
  54. package/dist/shared/pi-invocation.d.ts +7 -4
  55. package/dist/shared/pi-invocation.js +12 -7
  56. package/dist/shared/pkg-version.d.ts +13 -5
  57. package/dist/shared/pkg-version.js +13 -5
  58. package/dist/shared/reasoning-capability.d.ts +35 -24
  59. package/dist/shared/reasoning-capability.js +35 -24
  60. package/dist/shared/stream-watchdog.d.ts +60 -44
  61. package/dist/shared/stream-watchdog.js +62 -45
  62. package/dist/task/accept-debt.d.ts +41 -43
  63. package/dist/task/accept-debt.js +73 -65
  64. package/dist/task/api-synthesis.d.ts +24 -21
  65. package/dist/task/api-synthesis.js +32 -26
  66. package/dist/task/apis-contract.d.ts +32 -64
  67. package/dist/task/apis-contract.js +32 -64
  68. package/dist/task/artifact-closure.d.ts +27 -13
  69. package/dist/task/artifact-closure.js +95 -67
  70. package/dist/task/auto-commit.d.ts +46 -35
  71. package/dist/task/auto-commit.js +51 -38
  72. package/dist/task/auto-io.d.ts +45 -25
  73. package/dist/task/auto-io.js +57 -29
  74. package/dist/task/auto-orchestrator.d.ts +26 -24
  75. package/dist/task/auto-orchestrator.js +178 -162
  76. package/dist/task/auto-prompts.d.ts +36 -24
  77. package/dist/task/auto-prompts.js +40 -26
  78. package/dist/task/autofix-ledger.d.ts +27 -25
  79. package/dist/task/autofix-ledger.js +29 -26
  80. package/dist/task/batch-test-task.d.ts +20 -12
  81. package/dist/task/batch-test-task.js +67 -60
  82. package/dist/task/boot-probe.d.ts +60 -44
  83. package/dist/task/boot-probe.js +91 -72
  84. package/dist/task/cancel-input.d.ts +30 -16
  85. package/dist/task/cancel-input.js +20 -11
  86. package/dist/task/cancel-points.d.ts +27 -20
  87. package/dist/task/cancel-points.js +30 -22
  88. package/dist/task/child-runner.d.ts +46 -51
  89. package/dist/task/child-runner.js +48 -49
  90. package/dist/task/child-status.d.ts +23 -16
  91. package/dist/task/child-status.js +23 -16
  92. package/dist/task/clamp-output.js +12 -5
  93. package/dist/task/command-run.d.ts +31 -28
  94. package/dist/task/command-run.js +44 -35
  95. package/dist/task/command-shrink.d.ts +25 -18
  96. package/dist/task/command-shrink.js +37 -31
  97. package/dist/task/command-watchdog.d.ts +9 -6
  98. package/dist/task/command-watchdog.js +21 -15
  99. package/dist/task/context-attribution.d.ts +34 -26
  100. package/dist/task/context-attribution.js +34 -26
  101. package/dist/task/context-silence.d.ts +39 -29
  102. package/dist/task/context-silence.js +35 -25
  103. package/dist/task/context-usage.d.ts +25 -7
  104. package/dist/task/context-usage.js +21 -6
  105. package/dist/task/contracts.d.ts +8 -4
  106. package/dist/task/contracts.js +25 -17
  107. package/dist/task/coverage-loop.d.ts +22 -18
  108. package/dist/task/coverage-loop.js +35 -30
  109. package/dist/task/critique-probes.d.ts +13 -14
  110. package/dist/task/critique-probes.js +50 -39
  111. package/dist/task/debug-log.d.ts +13 -5
  112. package/dist/task/debug-log.js +32 -20
  113. package/dist/task/decompose-fidelity.d.ts +11 -9
  114. package/dist/task/decompose-fidelity.js +38 -33
  115. package/dist/task/decompose-granularity.d.ts +41 -38
  116. package/dist/task/decompose-granularity.js +41 -38
  117. package/dist/task/deep-render-check.d.ts +22 -14
  118. package/dist/task/deep-render-check.js +40 -31
  119. package/dist/task/dropped-input.d.ts +12 -7
  120. package/dist/task/dropped-input.js +5 -2
  121. package/dist/task/enforce-attribution.d.ts +38 -47
  122. package/dist/task/enforce-attribution.js +46 -52
  123. package/dist/task/enforce-guidelines.d.ts +31 -20
  124. package/dist/task/enforce-guidelines.js +32 -21
  125. package/dist/task/enrichment.d.ts +7 -2
  126. package/dist/task/enrichment.js +26 -14
  127. package/dist/task/env-notes.d.ts +16 -7
  128. package/dist/task/env-notes.js +48 -31
  129. package/dist/task/env-template-closure.d.ts +4 -4
  130. package/dist/task/env-template-closure.js +42 -34
  131. package/dist/task/external-context.d.ts +28 -21
  132. package/dist/task/external-context.js +17 -12
  133. package/dist/task/failure-classifier.d.ts +4 -5
  134. package/dist/task/failure-classifier.js +6 -7
  135. package/dist/task/file-inventory.d.ts +15 -11
  136. package/dist/task/file-inventory.js +25 -22
  137. package/dist/task/final-gate-fix.d.ts +74 -86
  138. package/dist/task/final-gate-fix.js +97 -116
  139. package/dist/task/final-gate-progress.d.ts +29 -46
  140. package/dist/task/final-gate-progress.js +40 -51
  141. package/dist/task/final-gate.d.ts +64 -97
  142. package/dist/task/final-gate.js +192 -199
  143. package/dist/task/fix-child.d.ts +21 -27
  144. package/dist/task/fix-child.js +21 -27
  145. package/dist/task/foreign-path.d.ts +6 -5
  146. package/dist/task/foreign-path.js +0 -0
  147. package/dist/task/frozen-conflict.d.ts +9 -10
  148. package/dist/task/frozen-conflict.js +61 -64
  149. package/dist/task/frozen-path-guard.d.ts +35 -14
  150. package/dist/task/frozen-path-guard.js +56 -39
  151. package/dist/task/gate-child.d.ts +27 -28
  152. package/dist/task/gate-child.js +37 -35
  153. package/dist/task/gate-deps.d.ts +34 -27
  154. package/dist/task/gate-deps.js +169 -159
  155. package/dist/task/gate-tally.d.ts +77 -80
  156. package/dist/task/gate-tally.js +65 -68
  157. package/dist/task/git-state-guard.d.ts +15 -11
  158. package/dist/task/git-state-guard.js +76 -66
  159. package/dist/task/impl-widget.d.ts +25 -16
  160. package/dist/task/impl-widget.js +27 -17
  161. package/dist/task/implementation-thinking.d.ts +33 -31
  162. package/dist/task/implementation-thinking.js +5 -6
  163. package/dist/task/implementation-turn.d.ts +34 -31
  164. package/dist/task/implementation-turn.js +29 -27
  165. package/dist/task/inline-markdown.d.ts +20 -7
  166. package/dist/task/inline-markdown.js +15 -6
  167. package/dist/task/launch-config-gap.js +25 -39
  168. package/dist/task/launch-contract.d.ts +18 -21
  169. package/dist/task/launch-contract.js +28 -30
  170. package/dist/task/launch-manifest.d.ts +6 -2
  171. package/dist/task/launch-manifest.js +35 -34
  172. package/dist/task/ledger.js +16 -14
  173. package/dist/task/lint-fix.d.ts +6 -8
  174. package/dist/task/lint-fix.js +67 -69
  175. package/dist/task/loop-detector.d.ts +9 -8
  176. package/dist/task/loop-detector.js +16 -12
  177. package/dist/task/mid-run-input.d.ts +17 -15
  178. package/dist/task/mid-run-input.js +17 -15
  179. package/dist/task/orchestrator.d.ts +24 -28
  180. package/dist/task/orchestrator.js +62 -64
  181. package/dist/task/orientation.d.ts +18 -23
  182. package/dist/task/orientation.js +24 -31
  183. package/dist/task/owned-freeze-conflict.d.ts +21 -20
  184. package/dist/task/owned-freeze-conflict.js +52 -85
  185. package/dist/task/owned-freeze-reassign.d.ts +40 -60
  186. package/dist/task/owned-freeze-reassign.js +41 -61
  187. package/dist/task/parsers.d.ts +4 -2
  188. package/dist/task/parsers.js +4 -4
  189. package/dist/task/phases.d.ts +41 -48
  190. package/dist/task/phases.js +180 -248
  191. package/dist/task/plan-io.d.ts +6 -7
  192. package/dist/task/plan-io.js +6 -7
  193. package/dist/task/plan-orchestrator.d.ts +10 -8
  194. package/dist/task/plan-orchestrator.js +14 -10
  195. package/dist/task/plan-prompts.d.ts +6 -5
  196. package/dist/task/plan-prompts.js +6 -5
  197. package/dist/task/plan-readonly.d.ts +4 -5
  198. package/dist/task/plan-readonly.js +4 -5
  199. package/dist/task/plan-rounds.d.ts +17 -29
  200. package/dist/task/plan-rounds.js +21 -34
  201. package/dist/task/plan-session.d.ts +58 -72
  202. package/dist/task/plan-session.js +61 -83
  203. package/dist/task/probe-gaming.d.ts +28 -27
  204. package/dist/task/probe-gaming.js +0 -0
  205. package/dist/task/prohibition-probe.d.ts +14 -16
  206. package/dist/task/prompts.d.ts +3 -4
  207. package/dist/task/prompts.js +17 -26
  208. package/dist/task/qa-transcript.d.ts +15 -22
  209. package/dist/task/qa-transcript.js +15 -21
  210. package/dist/task/question-box.d.ts +17 -13
  211. package/dist/task/question-box.js +19 -15
  212. package/dist/task/question-dedup.d.ts +6 -7
  213. package/dist/task/question-dedup.js +13 -14
  214. package/dist/task/question-dialog.d.ts +22 -32
  215. package/dist/task/question-dialog.js +22 -32
  216. package/dist/task/question-source.d.ts +18 -44
  217. package/dist/task/question-source.js +22 -51
  218. package/dist/task/refuted-constraint.d.ts +11 -31
  219. package/dist/task/refuted-constraint.js +27 -51
  220. package/dist/task/regenerable-artifacts.d.ts +12 -31
  221. package/dist/task/regenerable-artifacts.js +12 -31
  222. package/dist/task/render-check.d.ts +11 -22
  223. package/dist/task/render-check.js +33 -46
  224. package/dist/task/repo-health-check.d.ts +10 -14
  225. package/dist/task/repo-health-check.js +17 -23
  226. package/dist/task/requirements.d.ts +38 -71
  227. package/dist/task/requirements.js +78 -126
  228. package/dist/task/research-fanout-budget.d.ts +51 -88
  229. package/dist/task/research-fanout-budget.js +51 -88
  230. package/dist/task/research-worker.d.ts +33 -36
  231. package/dist/task/research-worker.js +39 -61
  232. package/dist/task/resume-gap.d.ts +14 -15
  233. package/dist/task/root-cause-repair.d.ts +9 -9
  234. package/dist/task/root-cause-repair.js +28 -40
  235. package/dist/task/run-bracket.d.ts +10 -13
  236. package/dist/task/run-end.d.ts +12 -22
  237. package/dist/task/run-end.js +8 -16
  238. package/dist/task/run-final-gate.d.ts +19 -21
  239. package/dist/task/run-final-gate.js +62 -80
  240. package/dist/task/runner-globs.d.ts +12 -13
  241. package/dist/task/runner-globs.js +12 -13
  242. package/dist/task/runner-resolve.d.ts +9 -9
  243. package/dist/task/runner-resolve.js +22 -23
  244. package/dist/task/script-escape.d.ts +10 -12
  245. package/dist/task/script-escape.js +13 -14
  246. package/dist/task/serve-entry.d.ts +1 -1
  247. package/dist/task/serve-entry.js +22 -25
  248. package/dist/task/service-blocks.js +4 -2
  249. package/dist/task/shipped-source.d.ts +11 -29
  250. package/dist/task/shipped-source.js +11 -29
  251. package/dist/task/skip-escape.js +10 -14
  252. package/dist/task/spec-urls.d.ts +26 -65
  253. package/dist/task/spec-urls.js +26 -65
  254. package/dist/task/spec-validation.d.ts +17 -20
  255. package/dist/task/spec-validation.js +17 -20
  256. package/dist/task/stall-detector.d.ts +23 -30
  257. package/dist/task/stall-detector.js +23 -30
  258. package/dist/task/stream-watchdog.d.ts +14 -12
  259. package/dist/task/stream-watchdog.js +14 -12
  260. package/dist/task/substitution-probe.d.ts +17 -20
  261. package/dist/task/substitution-probe.js +17 -20
  262. package/dist/task/task-gates.d.ts +36 -41
  263. package/dist/task/task-gates.js +95 -106
  264. package/dist/task/task-io.d.ts +4 -4
  265. package/dist/task/task-io.js +4 -4
  266. package/dist/task/task-parsers.js +4 -3
  267. package/dist/task/task-provenance.d.ts +2 -2
  268. package/dist/task/task-provenance.js +11 -13
  269. package/dist/task/task-types.d.ts +4 -3
  270. package/dist/task/terminal-outcome.d.ts +14 -16
  271. package/dist/task/terminal-outcome.js +12 -14
  272. package/dist/task/test-assembly.d.ts +13 -20
  273. package/dist/task/test-assembly.js +13 -20
  274. package/dist/task/timings.d.ts +5 -3
  275. package/dist/task/timings.js +5 -3
  276. package/dist/task/title-label.d.ts +9 -4
  277. package/dist/task/title-label.js +9 -4
  278. package/dist/task/type-only-answer.d.ts +44 -52
  279. package/dist/task/type-only-answer.js +44 -52
  280. package/dist/task/unfailable-command.d.ts +18 -24
  281. package/dist/task/unfailable-command.js +21 -27
  282. package/dist/task/unknown-routing.d.ts +10 -4
  283. package/dist/task/unknown-routing.js +10 -4
  284. package/dist/task/user-directives.d.ts +5 -8
  285. package/dist/task/user-directives.js +5 -8
  286. package/dist/task/verify-quality.d.ts +18 -22
  287. package/dist/task/verify-quality.js +45 -46
  288. package/dist/task/verify-reconcile.d.ts +15 -10
  289. package/dist/task/verify-reconcile.js +45 -43
  290. package/dist/task/verify-resolution.d.ts +24 -20
  291. package/dist/task/verify-resolution.js +51 -50
  292. package/dist/task/verify-work.d.ts +59 -66
  293. package/dist/task/verify-work.js +101 -138
  294. package/dist/task/widget.d.ts +15 -14
  295. package/dist/task/widget.js +22 -17
  296. package/dist/task/wiring-claims.d.ts +25 -32
  297. package/dist/task/wiring-claims.js +30 -35
  298. package/dist/task/write-guard.d.ts +39 -39
  299. package/dist/task/write-guard.js +48 -51
  300. package/dist/task/yolo.d.ts +34 -30
  301. package/dist/task/yolo.js +42 -37
  302. package/dist/workers/abstention.d.ts +21 -41
  303. package/dist/workers/abstention.js +27 -48
  304. package/dist/workers/brave-search.d.ts +4 -3
  305. package/dist/workers/brave-search.js +5 -2
  306. package/dist/workers/brave-warning.d.ts +7 -4
  307. package/dist/workers/brave-warning.js +19 -7
  308. package/dist/workers/ddg-search.d.ts +6 -6
  309. package/dist/workers/ddg-search.js +18 -12
  310. package/dist/workers/docs-cache.js +5 -2
  311. package/dist/workers/docs-chunk.d.ts +30 -37
  312. package/dist/workers/docs-chunk.js +37 -41
  313. package/dist/workers/docs-core.d.ts +28 -44
  314. package/dist/workers/docs-core.js +25 -44
  315. package/dist/workers/docs-index.js +4 -3
  316. package/dist/workers/docs-lookup.d.ts +15 -22
  317. package/dist/workers/docs-lookup.js +12 -21
  318. package/dist/workers/docs-project.d.ts +15 -9
  319. package/dist/workers/docs-project.js +17 -10
  320. package/dist/workers/docs-resolve.d.ts +19 -20
  321. package/dist/workers/docs-resolve.js +35 -32
  322. package/dist/workers/docs-retrieve.d.ts +5 -6
  323. package/dist/workers/docs-retrieve.js +18 -15
  324. package/dist/workers/exa-search.d.ts +9 -6
  325. package/dist/workers/exa-search.js +23 -12
  326. package/dist/workers/fetch-core.d.ts +13 -16
  327. package/dist/workers/fetch-core.js +23 -23
  328. package/dist/workers/focused-extractor.d.ts +12 -12
  329. package/dist/workers/focused-extractor.js +16 -19
  330. package/dist/workers/html-clean.js +24 -14
  331. package/dist/workers/http-request.d.ts +28 -20
  332. package/dist/workers/http-request.js +22 -17
  333. package/dist/workers/npm-version.d.ts +28 -11
  334. package/dist/workers/npm-version.js +24 -15
  335. package/dist/workers/phantom-imports.d.ts +15 -12
  336. package/dist/workers/phantom-imports.js +30 -24
  337. package/dist/workers/pi-worker-core.d.ts +86 -54
  338. package/dist/workers/pi-worker-core.js +112 -112
  339. package/dist/workers/pi-worker-docs.d.ts +24 -19
  340. package/dist/workers/pi-worker-docs.js +67 -76
  341. package/dist/workers/pi-worker-fetch.d.ts +7 -3
  342. package/dist/workers/pi-worker-fetch.js +27 -19
  343. package/dist/workers/pi-worker-search.js +12 -8
  344. package/dist/workers/pi-worker.d.ts +9 -4
  345. package/dist/workers/pi-worker.js +23 -10
  346. package/dist/workers/reasoning-warning.d.ts +18 -17
  347. package/dist/workers/reasoning-warning.js +22 -20
  348. package/dist/workers/research-cache.js +50 -78
  349. package/dist/workers/search-core.js +7 -5
  350. package/dist/workers/search-types.d.ts +10 -9
  351. package/dist/workers/search-types.js +9 -8
  352. package/dist/workers/session-hint.d.ts +13 -14
  353. package/dist/workers/session-hint.js +8 -9
  354. package/dist/workers/shared.d.ts +21 -25
  355. package/dist/workers/shared.js +0 -0
  356. package/dist/workers/single-read-extension.d.ts +14 -7
  357. package/dist/workers/single-read-extension.js +14 -7
  358. package/dist/workers/single-read-guard.d.ts +25 -28
  359. package/dist/workers/single-read-guard.js +32 -32
  360. package/dist/workers/typeonly-log.d.ts +12 -9
  361. package/dist/workers/typeonly-log.js +29 -33
  362. package/dist/workers/worker-channels.d.ts +15 -23
  363. package/dist/workers/worker-channels.js +15 -23
  364. package/dist/workers/worker-failure.d.ts +38 -46
  365. package/dist/workers/worker-failure.js +31 -39
  366. package/dist/workers/worker-kill.d.ts +25 -26
  367. package/dist/workers/worker-kill.js +16 -19
  368. package/dist/workers/worker-profiles.d.ts +43 -53
  369. package/dist/workers/worker-profiles.js +30 -38
  370. package/package.json +10 -8
@@ -6,44 +6,48 @@ import { type WorkerGuardOverride, type WorkerGuardPolicy, type WorkerPolicyInpu
6
6
  * command could be cited from. `pi-worker-docs` (the primary), `read` and `grep`
7
7
  * (project source), and the web escalations `pi-worker-search`/`pi-worker-fetch`.
8
8
  *
9
- * `ls` and `find` are deliberately EXCLUDED: they return file/directory NAMES,
10
- * and APIS owns symbols by name only, never paths (RESEARCH_APIS_PROMPT). Bare
11
- * enumeration cannot verify a signature, so a worker that fabricates its section
12
- * from memory does not launder itself grounded by calling `ls` once. That
13
- * exclusion is the anti-gaming property of any gate built on this count: "one
14
- * trivial `ls` then fabricate the rest" leaves groundingRetrievalCount at 0.
9
+ * `ls` and `find` are deliberately EXCLUDED: they return file and directory
10
+ * NAMES, and APIS owns symbols by name only, never paths (see
11
+ * RESEARCH_APIS_PROMPT in prompts.ts). Bare enumeration cannot verify a
12
+ * signature, so a worker that fabricates its section from memory cannot launder
13
+ * itself grounded by calling `ls` once "one trivial `ls` then fabricate the
14
+ * rest" still leaves groundingRetrievalCount at 0.
15
+ *
16
+ * The set is DERIVED from WORKER_CHANNELS (worker-channels.ts), not hand-kept.
17
+ * Re-exported here only so worker-channels.test.ts can assert this module hands
18
+ * back the same predicate.
15
19
  */
16
20
  export { isGroundingRetrieval } from './worker-channels.js';
17
21
  /**
18
22
  * Does this partial output carry ANSWER CONTENT, or is it the model clearing its
19
23
  * throat?
20
24
  *
21
- * Salvage originally kept the LONGEST partial, which is not the same question. On
22
- * the live carry arm, TASK_0020 and TASK_0021 both timed out on all three
23
- * attempts and salvage shipped this as the section:
25
+ * Keeping the LONGEST partial is not the same question, and it has an obvious
26
+ * failure: a preamble sentence like
24
27
  *
25
28
  * "Now let me get more details on the specific APIs and components I need:"
26
29
  *
27
- * — a preamble sentence, which beats an empty string on length and carries
28
- * nothing. Both trials scored 2 entries and DEGRADED, against 22 and 5 for the
29
- * same fixtures in baseline.
30
+ * beats an empty string on length and carries nothing.
30
31
  *
31
32
  * A research worker's answer is a list of lines that each name something and
32
- * describe it. The test is therefore structural, not lexical: at least two lines
33
- * that look like entries — a name, then a gap, then a description. Prose wraps
34
- * at no particular column and does not repeat that shape.
33
+ * describe it. The test is therefore structural, not lexical: at least TWO lines
34
+ * that look like entries — a name, then a gap, then a description. Prose wraps at
35
+ * no particular column and does not repeat that shape; the sentence above scores
36
+ * zero entry lines.
35
37
  */
36
38
  export declare function hasAnswerContent(text: string): boolean;
37
39
  /**
38
40
  * Is ONE line an entry — a name, a gap, then a description — rather than prose?
39
41
  *
40
- * Split out of `hasAnswerContent` so the same rule can decide what a line IS,
41
- * not just how many of them there are. A FILES section's paths are read back
42
- * with it, and a scorer that used its own idea of an entry counted a preamble
43
- * sentence and a leaked `</tool_call>` as invented paths.
42
+ * Split out of `hasAnswerContent` so the same rule can decide what a line IS, not
43
+ * just how many of them there are a reader of a FILES section needs the same
44
+ * test, and its own idea of an entry would count a preamble sentence or a leaked
45
+ * `</tool_call>` as one.
44
46
  *
45
- * Prose wraps at no particular column, so it carries no two-space gap and no
46
- * spaced dash; when it does, it ends in `.` or `:` and an entry does not.
47
+ * A leading `-`, `*`, `•` or `1.`/`1)` bullet is stripped first. What remains must
48
+ * hold a two-space gap or a spaced dash and must NOT end in `.` or `:`. Prose
49
+ * wraps at no particular column, so it carries neither; when it does carry one, it
50
+ * ends in punctuation and an entry does not.
47
51
  */
48
52
  export declare function isEntryLine(raw: string): boolean;
49
53
  /**
@@ -67,7 +71,7 @@ export interface RunWorkerInput {
67
71
  /** Called for each tool execution start and text-writing event inside the worker. */
68
72
  onLine?: (line: string) => void;
69
73
  /** Called when a tool call FINISHES, with its (truncatable) result — lets a caller
70
- * log tool OUTPUTS, not just the command (mx5 run 10 item 6). */
74
+ * log tool OUTPUTS, not just the command. */
71
75
  onToolResult?: (result: {
72
76
  name: string;
73
77
  isError: boolean;
@@ -82,20 +86,34 @@ export interface RunWorkerInput {
82
86
  */
83
87
  onContextUsage?: (snapshot: ContextSnapshot) => void;
84
88
  /**
85
- * The worker child's context window in tokens. pi's event stream carries no
86
- * window (issue #16), so a caller that wants a progress bar rather than a
87
- * bare token count has to supply the parent session's — which is the child's
88
- * too, since workers are spawned without `-m`.
89
+ * The worker child's context window in tokens, or `'unknown'` when the
90
+ * caller genuinely has none.
91
+ *
92
+ * REQUIRED, and required for the same reason `profile` below is. pi's event
93
+ * stream carries NO window — the string `context_usage` appears nowhere in any
94
+ * installed @earendil-works package, and its only usage-bearing JSON event is
95
+ * `message_update` — so this parameter is the only source there is. Left
96
+ * optional, a caller that omits it leaves `noteContext` seeing 0, and
97
+ * `StallDetector`'s CONTEXT CHURN rule is gated on a positive window, so the
98
+ * rule silently does not exist. That reads exactly like a rule that exists and
99
+ * did not trip.
100
+ *
101
+ * WHY A WORD AND NOT `0` OR `null`. Both of those are what a caller types when
102
+ * it has not thought about the question, and both disarm the rule silently.
103
+ * `'unknown'` cannot be typed by accident, is greppable, and shows up in a diff
104
+ * as a decision.
105
+ *
106
+ * Two consumers read it: the churn rule, and the caller's progress bar, which
107
+ * shows a bare token count without a window. Both degrade on `'unknown'`.
89
108
  */
90
- contextWindow?: number;
109
+ contextWindow: number | 'unknown';
91
110
  /**
92
111
  * WHICH KIND of worker child this is — the whole guard policy, in one word.
93
112
  *
94
- * REQUIRED, and required on purpose. The ten guard knobs this replaces used
95
- * to sit here as independent optionals, so a caller that named none of them
96
- * still got a full policy and nobody could see which one. That is how the
97
- * ad-hoc `pi-worker` tool came to run the strictest wall clock of the three
98
- * children without anyone deciding it should. See worker-profiles.ts.
113
+ * REQUIRED, and required on purpose. As independent optionals, a caller that
114
+ * named none of the guard knobs still got a full policy and nobody could see
115
+ * which one — so a child can end up running the strictest wall clock of the
116
+ * three without anyone deciding it should. See worker-profiles.ts.
99
117
  */
100
118
  profile: WorkerProfileId;
101
119
  /**
@@ -104,10 +122,11 @@ export interface RunWorkerInput {
104
122
  */
105
123
  policyInputs?: WorkerPolicyInputs;
106
124
  /**
107
- * Whole guard rows laid over the profile's. TESTS AND A/B HARNESSES ONLY —
108
- * an override at a production call site is the hand-picked subset this
109
- * design exists to stop, and `worker-profiles.test.ts` fails the build if
110
- * one appears under src/ outside a test.
125
+ * Whole guard rows laid over the profile's. TESTS AND HARNESSES ONLY — an
126
+ * override at a production call site is the hand-picked subset this design
127
+ * exists to stop. `worker-profiles.test.ts` enforces it: its "no production
128
+ * source file passes an `override` to runWorker" test scans src/ for a leading
129
+ * `override:` and fails on any hit.
111
130
  */
112
131
  override?: WorkerGuardOverride;
113
132
  /**
@@ -116,9 +135,9 @@ export interface RunWorkerInput {
116
135
  *
117
136
  * WHY: asserting that a profile RESOLVES correctly proves nothing about
118
137
  * whether runWorker then READS it correctly — a rewiring that turns "0 means
119
- * off" into "0 means on" leaves every profile assertion green. This hook is
120
- * what lets a caller's own test (gate-child.test.ts) drive the REAL call
121
- * site and check the REAL policy, instead of re-typing the table.
138
+ * off" into "0 means on" leaves every profile assertion green. This hook lets
139
+ * a test drive the REAL call site and read back the REAL policy instead of
140
+ * re-typing the table; worker-profiles.test.ts is where those assertions live.
122
141
  */
123
142
  onPolicy?: (policy: WorkerGuardPolicy) => void;
124
143
  /** Backoff sleep, injectable so tests don't wait out the real delays. */
@@ -151,12 +170,10 @@ export interface RunWorkerInput {
151
170
  *
152
171
  * WHY: every restart branch below throws away a whole attempt's wall clock
153
172
  * along with its text, and `waitMs`/`workMs` describe the FINAL attempt only.
154
- * With no hook here those attempts were structurally invisible: mx5 run 18
155
- * burned 30 wall-clock timeouts / 120 minutes of compute that appeared in no
156
- * log and no timing widget, and 21 of the 23 affected workers reported
157
- * `exit=0` — clean successes as far as the run could tell. The discrepancy
158
- * was only recoverable by subtracting reported wait+work from the timestamps
159
- * of the `start` and `done` lines around it.
173
+ * With no hook here a discarded attempt is structurally invisible the worker
174
+ * still returns `exitCode` 0 and reads as a clean success, and the lost time is
175
+ * recoverable only by subtracting the reported wait+work from the timestamps
176
+ * around the call.
160
177
  */
161
178
  onRestart?: (restart: WorkerRestart) => void;
162
179
  }
@@ -178,6 +195,19 @@ export interface WorkerRestart {
178
195
  workMs: number;
179
196
  /** Reason-specific diagnosis: the looping call, the hung tool, the error text. */
180
197
  detail?: string;
198
+ /**
199
+ * Characters of ANSWER TEXT this attempt had produced at the moment it was
200
+ * thrown away.
201
+ *
202
+ * Recorded whatever `carryForward` says, because the DISCARD is the thing a
203
+ * reader cannot otherwise see. Without it, "the guards worked and the run
204
+ * returned almost nothing" and "the guards worked and the run threw away a
205
+ * finished answer" print identically.
206
+ *
207
+ * It is an OBSERVATION, not a decision: harvesting into `salvage` is still
208
+ * gated on the profile, and this number changes no behaviour.
209
+ */
210
+ partialChars: number;
181
211
  }
182
212
  export interface RunWorkerResult {
183
213
  text: string;
@@ -188,9 +218,9 @@ export interface RunWorkerResult {
188
218
  * The provider-reported cause when the model turn itself failed (disconnect,
189
219
  * fetch failed, 5xx after pi's own retries): pi delivers it as an assistant
190
220
  * message with stopReason "error" and EMPTY text, exit code 0. Phase children
191
- * have always surfaced this (child-runner.ts) research workers did not, so a
192
- * swallowed provider error reached the caller as an indistinguishable empty
193
- * answer and was reported as the useless "produced no output" (issue #10).
221
+ * surface this through child-runner.ts; without it a swallowed provider error
222
+ * reaches the caller as an indistinguishable empty answer and gets reported as
223
+ * the useless "produced no output".
194
224
  * Only meaningful when `text` is empty: a turn that produced text after pi
195
225
  * recovered is a success, and the first-error capture must not relabel it.
196
226
  */
@@ -315,9 +345,9 @@ export interface RunWorkerResult {
315
345
  * commandTimeoutHint, which tells the model in as many words to bound its
316
346
  * command; a SECOND hang means it ignored an explicit instruction, and a third
317
347
  * means it ignored it twice. Giving a non-complying child the full ceiling again
318
- * would put the worst case at 3 × 15 min = 45 minutes of dead time, resting
319
- * entirely on the model obeying prose. Halving bounds it at ~26 min while
320
- * costing a complying child nothing.
348
+ * makes the worst case three times the ceiling, resting entirely on the model
349
+ * obeying prose. Halving bounds it at under twice the ceiling while costing a
350
+ * complying child nothing.
321
351
  *
322
352
  * `priorHangs` counts watchdog kills specifically, NOT total restarts — the
323
353
  * restart budget is shared with loop kills, and a child restarted for LOOPING
@@ -325,8 +355,9 @@ export interface RunWorkerResult {
325
355
  * the full ceiling. Only a hang after a hang is defiance.
326
356
  *
327
357
  * Floored at 30s so repeated halving cannot shrink the ceiling to something no
328
- * real command could finish inside — but never ABOVE the configured ceiling
329
- * itself, or a caller asking for 10s would silently get 30.
358
+ * real command could finish inside — but the floor is `min(base, 30s)`, never
359
+ * above the configured ceiling, so a caller asking for 10s keeps 10s at every
360
+ * hang count. A base of 0 or less disables the watchdog and stays 0.
330
361
  */
331
362
  export declare function commandCeilingForAttempt(baseMs: number, priorHangs: number): number;
332
363
  /**
@@ -343,7 +374,8 @@ interface RestartState {
343
374
  timedOut: boolean;
344
375
  modelError?: string;
345
376
  leaked: string | null;
346
- /** The cap this attempt actually died against the SCALE arm moves it. */
377
+ /** The cap this attempt actually died against, not the configured one:
378
+ * `extend`/`progress` can push the deadline out during the attempt. */
347
379
  effectiveCapMs: number;
348
380
  /** The child's tool string, which decides whether its edits can persist. */
349
381
  tools: string;
@@ -10,33 +10,36 @@ import { detectLeakedToolCall, leakedToolCallHint, MAX_LEAK_RETRIES } from '../s
10
10
  import { discoverModelEndpoints, probeModelEndpoints } from '../shared/model-endpoint.js';
11
11
  import { streamStallHint } from '../shared/stream-watchdog.js';
12
12
  import { classifyWorkerFailure } from './worker-failure.js';
13
- import { CARRY_FORWARD_IDS } from './worker-kill.js';
13
+ import { CARRY_FORWARD_IDS, RESTART_ORDER } from './worker-kill.js';
14
14
  import { applyOverride, WORKER_PROFILES } from './worker-profiles.js';
15
- // `--mode json` makes pi emit structured events as they happen instead of
16
- // buffering the assistant text and flushing on exit. That matters for the
17
- // wait/work timing split: in text mode the first stdout chunk only arrives at
18
- // the very end, so onFirstByte fires moments before close and workMs is
19
- // effectively zero. With JSON events the first byte lands as soon as the
20
- // model starts producing — making waitMs the real queue/cold-start cost and
21
- // workMs the real generation+tool-call cost.
15
+ /** The tool whitelist a caller gets when it names none. */
22
16
  const DEFAULT_TOOLS = 'read,grep,find,ls';
17
+ /**
18
+ * The one place `'unknown'` becomes the 0 both consumers already treat as
19
+ * "no window". Written once so a future reader cannot re-introduce the optional
20
+ * by handling the union at only one of the two sites that read it.
21
+ */
22
+ function contextWindowTokens(cw) {
23
+ return cw === 'unknown' || cw <= 0 ? 0 : cw;
24
+ }
23
25
  /**
24
26
  * Tool calls that can GROUND an APIS claim — i.e. return content a signature or
25
27
  * command could be cited from. `pi-worker-docs` (the primary), `read` and `grep`
26
28
  * (project source), and the web escalations `pi-worker-search`/`pi-worker-fetch`.
27
29
  *
28
- * `ls` and `find` are deliberately EXCLUDED: they return file/directory NAMES,
29
- * and APIS owns symbols by name only, never paths (RESEARCH_APIS_PROMPT). Bare
30
- * enumeration cannot verify a signature, so a worker that fabricates its section
31
- * from memory does not launder itself grounded by calling `ls` once. That
32
- * exclusion is the anti-gaming property of any gate built on this count: "one
33
- * trivial `ls` then fabricate the rest" leaves groundingRetrievalCount at 0.
30
+ * `ls` and `find` are deliberately EXCLUDED: they return file and directory
31
+ * NAMES, and APIS owns symbols by name only, never paths (see
32
+ * RESEARCH_APIS_PROMPT in prompts.ts). Bare enumeration cannot verify a
33
+ * signature, so a worker that fabricates its section from memory cannot launder
34
+ * itself grounded by calling `ls` once "one trivial `ls` then fabricate the
35
+ * rest" still leaves groundingRetrievalCount at 0.
36
+ *
37
+ * The set is DERIVED from WORKER_CHANNELS (worker-channels.ts), not hand-kept.
38
+ * Re-exported here only so worker-channels.test.ts can assert this module hands
39
+ * back the same predicate.
34
40
  */
35
- // The grounding set is derived from WORKER_CHANNELS (worker-channels.ts), not
36
- // hand-kept — this was a second copy of the four tool names. Re-exported because
37
- // several call sites and tests import it from here.
38
41
  export { isGroundingRetrieval } from './worker-channels.js';
39
- // RESEARCH_WORKER_TIMEOUT_MS and STALL_AFTER_MS live on the profile table now
42
+ // RESEARCH_WORKER_TIMEOUT_MS and STALL_AFTER_MS live on the profile table
40
43
  // (worker-profiles.ts): they are the default VALUES of two guard rows, and a
41
44
  // default that lives apart from the table stating it is a second place to look.
42
45
  /**
@@ -51,49 +54,44 @@ const WORKER_TIMEOUT_HINT = '[SYSTEM NOTE: Your previous attempt ran out of time
51
54
  /**
52
55
  * How much of a discarded attempt's answer is carried into the next one.
53
56
  *
54
- * A restart used to hand the re-spawn nothing but a hint — which is why
55
- * WORKER_TIMEOUT_HINT above can tell a worker "do not re-explore ground you have
56
- * already covered" while giving it no record of what that ground was. It could
57
- * not comply. mx5 run 18 shows the cost: on tasks with >=46 project-source
58
- * lookups, 5 of 5 workers burned the FULL restart budget, because every attempt
59
- * re-read the same files against the same clock and died in the same place.
57
+ * A restart that hands the re-spawn nothing but a hint is why WORKER_TIMEOUT_HINT
58
+ * above can tell a worker "do not re-explore ground you have already covered"
59
+ * while giving it no record of what that ground was. It cannot comply, so it
60
+ * re-reads the same files against the same clock and dies in the same place.
60
61
  *
61
- * Carrying the partial answer forward is what makes a restart converge instead
62
- * of repeat. The risk it takes is real and is the thing the A/B measures: a
63
- * half-written or speculative entry, replayed under "already established", is
64
- * exactly how a fabrication gets laundered into a final answer. That is what the
65
- * ungrounded-symbol and anti-synthesis guards are pointed at, so the carry is
66
- * framed as findings to VERIFY-or-DROP rather than as settled fact.
62
+ * Carrying the partial forward is what lets a restart converge instead of repeat.
63
+ * The risk is real: a half-written or speculative entry, replayed under "already
64
+ * established", is how a fabrication gets laundered into a final answer. That is
65
+ * why `formatCarryForward` frames it as findings to VERIFY-or-DROP rather than as
66
+ * settled fact.
67
67
  */
68
68
  const CARRY_FORWARD_LIMIT = 24_000;
69
69
  /**
70
70
  * Restart reasons whose partial output is worth keeping.
71
71
  *
72
- * A clock kill (`worker-timeout`), a hung tool (`command-timeout`), an idle
73
- * stream (`stream-stall`) and a dropped socket (`connection-error`) all discard
74
- * work the model genuinely did. A loop kill and a leaked tool call do not — the
75
- * first is by definition the same call repeated, the second is malformed
76
- * protocol text, and replaying either would feed the failure back to itself.
72
+ * Exactly four, derived from WORKER_KILLS: `command-timeout`, `stream-stall`,
73
+ * `worker-timeout` and `connection-error` all discard work the model genuinely
74
+ * did. A loop kill and a leaked tool call do not — the first is by definition the
75
+ * same call repeated, the second is malformed protocol text, and replaying either
76
+ * would feed the failure back to itself.
77
77
  */
78
78
  const CARRY_FORWARD_REASONS = CARRY_FORWARD_IDS;
79
79
  /**
80
80
  * Does this partial output carry ANSWER CONTENT, or is it the model clearing its
81
81
  * throat?
82
82
  *
83
- * Salvage originally kept the LONGEST partial, which is not the same question. On
84
- * the live carry arm, TASK_0020 and TASK_0021 both timed out on all three
85
- * attempts and salvage shipped this as the section:
83
+ * Keeping the LONGEST partial is not the same question, and it has an obvious
84
+ * failure: a preamble sentence like
86
85
  *
87
86
  * "Now let me get more details on the specific APIs and components I need:"
88
87
  *
89
- * — a preamble sentence, which beats an empty string on length and carries
90
- * nothing. Both trials scored 2 entries and DEGRADED, against 22 and 5 for the
91
- * same fixtures in baseline.
88
+ * beats an empty string on length and carries nothing.
92
89
  *
93
90
  * A research worker's answer is a list of lines that each name something and
94
- * describe it. The test is therefore structural, not lexical: at least two lines
95
- * that look like entries — a name, then a gap, then a description. Prose wraps
96
- * at no particular column and does not repeat that shape.
91
+ * describe it. The test is therefore structural, not lexical: at least TWO lines
92
+ * that look like entries — a name, then a gap, then a description. Prose wraps at
93
+ * no particular column and does not repeat that shape; the sentence above scores
94
+ * zero entry lines.
97
95
  */
98
96
  export function hasAnswerContent(text) {
99
97
  return text.split('\n').filter(isEntryLine).length >= 2;
@@ -101,13 +99,15 @@ export function hasAnswerContent(text) {
101
99
  /**
102
100
  * Is ONE line an entry — a name, a gap, then a description — rather than prose?
103
101
  *
104
- * Split out of `hasAnswerContent` so the same rule can decide what a line IS,
105
- * not just how many of them there are. A FILES section's paths are read back
106
- * with it, and a scorer that used its own idea of an entry counted a preamble
107
- * sentence and a leaked `</tool_call>` as invented paths.
102
+ * Split out of `hasAnswerContent` so the same rule can decide what a line IS, not
103
+ * just how many of them there are a reader of a FILES section needs the same
104
+ * test, and its own idea of an entry would count a preamble sentence or a leaked
105
+ * `</tool_call>` as one.
108
106
  *
109
- * Prose wraps at no particular column, so it carries no two-space gap and no
110
- * spaced dash; when it does, it ends in `.` or `:` and an entry does not.
107
+ * A leading `-`, `*`, `•` or `1.`/`1)` bullet is stripped first. What remains must
108
+ * hold a two-space gap or a spaced dash and must NOT end in `.` or `:`. Prose
109
+ * wraps at no particular column, so it carries neither; when it does carry one, it
110
+ * ends in punctuation and an entry does not.
111
111
  */
112
112
  export function isEntryLine(raw) {
113
113
  const l = raw.replace(/^\s*(?:[-*•]|\d+[.)])\s+/, '').trim();
@@ -175,10 +175,10 @@ absoluteCeilingMs) {
175
175
  return {
176
176
  signal: ctrl.signal,
177
177
  timedOut: () => timedOut,
178
- // SCALE arm of nexttask 5B, inert unless a caller calls it: push the
179
- // deadline out, never past `started + ceilingMs`. A disabled timeout
180
- // (nothing armed) stays disabled — extending "never" is meaningless — and
181
- // an already-fired timer is not resurrected.
178
+ // Push the deadline out, never past `started + ceilingMs`. Inert unless a
179
+ // caller calls it. A disabled timeout (nothing armed) stays disabled
180
+ // extending "never" is meaningless — and an already-fired timer is not
181
+ // resurrected.
182
182
  extend: (byMs, ceilingMs) => {
183
183
  if (!armed || timedOut || ctrl.signal.aborted)
184
184
  return;
@@ -224,9 +224,9 @@ absoluteCeilingMs) {
224
224
  * commandTimeoutHint, which tells the model in as many words to bound its
225
225
  * command; a SECOND hang means it ignored an explicit instruction, and a third
226
226
  * means it ignored it twice. Giving a non-complying child the full ceiling again
227
- * would put the worst case at 3 × 15 min = 45 minutes of dead time, resting
228
- * entirely on the model obeying prose. Halving bounds it at ~26 min while
229
- * costing a complying child nothing.
227
+ * makes the worst case three times the ceiling, resting entirely on the model
228
+ * obeying prose. Halving bounds it at under twice the ceiling while costing a
229
+ * complying child nothing.
230
230
  *
231
231
  * `priorHangs` counts watchdog kills specifically, NOT total restarts — the
232
232
  * restart budget is shared with loop kills, and a child restarted for LOOPING
@@ -234,8 +234,9 @@ absoluteCeilingMs) {
234
234
  * the full ceiling. Only a hang after a hang is defiance.
235
235
  *
236
236
  * Floored at 30s so repeated halving cannot shrink the ceiling to something no
237
- * real command could finish inside — but never ABOVE the configured ceiling
238
- * itself, or a caller asking for 10s would silently get 30.
237
+ * real command could finish inside — but the floor is `min(base, 30s)`, never
238
+ * above the configured ceiling, so a caller asking for 10s keeps 10s at every
239
+ * hang count. A base of 0 or less disables the watchdog and stays 0.
239
240
  */
240
241
  export function commandCeilingForAttempt(baseMs, priorHangs) {
241
242
  if (!(baseMs > 0))
@@ -315,8 +316,8 @@ export const RESTART_RULES = [
315
316
  // a loop also tripped — the loop hint above is more specific.
316
317
  reason: 'worker-timeout',
317
318
  detect: s => s.timedOut && !s.loopHit && s.restartBudgetSpent < MAX_LOOP_RESTARTS ?
318
- // The EFFECTIVE cap, which the SCALE arm moves reporting the
319
- // configured one would misname why this attempt died.
319
+ // The EFFECTIVE cap, which `extend`/`progress` can have moved
320
+ // reporting the configured one would misname why this attempt died.
320
321
  { detail: `cap ${s.effectiveCapMs}ms` }
321
322
  : null,
322
323
  hint: () => WORKER_TIMEOUT_HINT,
@@ -324,22 +325,15 @@ export const RESTART_RULES = [
324
325
  },
325
326
  {
326
327
  // A connection-class model error is restartable on the same budget, exactly
327
- // as runPhaseChild already treats it a research worker had no such
328
- // retry, so one dropped fetch failed the whole task at research while the
329
- // identical blip in refine/compose was absorbed.
328
+ // as runPhaseChild already treats it. Without it one dropped socket fails
329
+ // the whole task at research, while the identical blip in refine or compose
330
+ // is absorbed.
330
331
  //
331
- // What this can and cannot buy, measured (flaky proxy in front of the local
332
- // llama-server, dropping every connection for a fixed outage window): pi
333
- // retries a failed turn itself, 4 attempts over ~15s, and a run that
334
- // recovers no longer reports modelError at all (see JsonEventSink). So a
335
- // surfaced connection error means pi's own ~15s budget is already spent, and
336
- // a re-spawn only helps when the outage outlasts it. It does: at a 20s
337
- // outage the baseline never recovered and this policy always did, 0/8 → 8/8
338
- // (Fisher p=0.00016), and the same at 35s. Below ~15s pi absorbs it alone —
339
- // 8/8 both arms, so the retry neither helps nor costs there. Beyond ~46s
340
- // (three spawns' combined budget) both arms fail. The price is paid only on
341
- // a backend that is really gone: time-to-report goes ~15s → ~46s. Re-run:
342
- // scripts/connection-retry-ab.ts.
332
+ // pi retries a failed turn itself before reporting anything, and a run that
333
+ // recovers reports no modelError at all (see JsonEventSink). So a SURFACED
334
+ // connection error means pi's own budget is already spent, and a re-spawn
335
+ // only helps when the outage outlasts it. The price is paid only on a
336
+ // backend that is really gone: time-to-report grows by the extra spawns.
343
337
  //
344
338
  // Connection class ONLY. Auth, bad request and context overflow still fail
345
339
  // fast: re-issuing the same request cannot fix them, so spending the budget
@@ -372,11 +366,11 @@ export const RESTART_RULES = [
372
366
  * runChild turns into a process-GROUP kill — reaping the hung command itself,
373
367
  * not just the pi child holding it.
374
368
  *
375
- * LIMIT: the group kill only reaches processes still IN the group. A hung
376
- * command that detached a daemon (setsid/nohup dev server) leaves it running —
377
- * the fresh attempt can then hit a port the dead attempt's escapee still holds
378
- * (the run-9 orphan-dev-server false-EADDRINUSE shape). No cheap fix from
379
- * here; the restart hint's "check current state" line is the mitigation.
369
+ * LIMIT: the group kill only reaches processes still IN the group. A hung command
370
+ * that detached a daemon (setsid, nohup, a background dev server) leaves it
371
+ * running, so the fresh attempt can hit a port the dead attempt's escapee still
372
+ * holds. There is no cheap fix from here; the restart hint's "check current state"
373
+ * line is the mitigation.
380
374
  *
381
375
  * Returns null when the watchdog is off, so the caller keeps the plain timeout
382
376
  * signal and no per-call bookkeeping happens at all.
@@ -420,6 +414,13 @@ function commandWatch(timeoutMs) {
420
414
  }
421
415
  export async function runWorker(input) {
422
416
  const tools = input.tools ?? DEFAULT_TOOLS;
417
+ // `--mode json` makes pi emit structured events as they happen instead of
418
+ // buffering the assistant text and flushing on exit. Its print-mode source
419
+ // shows both halves: under `json` a session subscriber writes every event to
420
+ // stdout as it arrives, while under `text` NOTHING is written until after the
421
+ // prompt resolves, when the last assistant message is printed once. That is
422
+ // what makes the wait/work split real — onFirstByte would otherwise fire
423
+ // moments before close and leave workMs at nearly zero.
423
424
  const baseArgs = [
424
425
  ...childBaseArgs(input.extensions ?? []),
425
426
  ...(input.thinking ?? []),
@@ -467,11 +468,10 @@ export async function runWorker(input) {
467
468
  const salvage = { text: null };
468
469
  for (;;) {
469
470
  const carried = salvage.text === null ? null : formatCarryForward(salvage.text);
470
- // Announce the INJECTION, not just the restart. Without this, "the carry
471
- // reached the re-spawn" can only be inferred from entry countsand
472
- // inferring what a worker did from what it produced is the exact gap 5A
473
- // exists to close. The prompt goes to the child on stdin, so no log
474
- // downstream of here can show it.
471
+ // Announce the INJECTION, not just the restart. The prompt goes to the child
472
+ // on stdin, so no log downstream of here can show it without this hook
473
+ // "the carry reached the re-spawn" could only be inferred from the answer,
474
+ // which is inferring what a worker did from what it produced.
475
475
  if (carried !== null) {
476
476
  input.onCarryForward?.({
477
477
  attempt: restarts.length + 1,
@@ -498,9 +498,9 @@ export async function runWorker(input) {
498
498
  null
499
499
  : new StallDetector(guards.loop.progress.limit, guards.loop.progress.churnFactor);
500
500
  // Arm the churn rule BEFORE the first tool call. pi's stream carries no
501
- // context event (issue #16), so waiting for one leaves the rule
502
- // permanently disarmed. The parent knows the window at spawn time.
503
- stallDetector?.noteContext(input.contextWindow ?? 0);
501
+ // context event at all, so waiting for one leaves the rule permanently
502
+ // disarmed. The parent knows the window at spawn time.
503
+ stallDetector?.noteContext(contextWindowTokens(input.contextWindow));
504
504
  // Capture the hit the detector reports (it also returns it to the unified
505
505
  // runner, which kills the child on a hit). Without capturing it here the
506
506
  // SIGTERM that kill produces would surface as a bare non-zero exit the
@@ -538,8 +538,9 @@ export async function runWorker(input) {
538
538
  // A tool call is the worker working. Inert unless the
539
539
  // caller opted into a progress-based deadline.
540
540
  timeout.progress();
541
- // The generic child runner used to name ONE tool and ONE of
542
- // its parameters here. It asks the tool's own row now.
541
+ // Naming ONE tool and ONE of its parameters here would put
542
+ // that knowledge in the generic runner, so it asks the
543
+ // tool's own row in WORKER_CHANNELS instead.
543
544
  if (clock.fanout
544
545
  && workerChannel(call.name)?.isProjectSourceLookup?.(call.args ?? {}) === true) {
545
546
  timeout.extend(clock.fanout.perLookupMs, clock.fanout.ceilingMs);
@@ -562,11 +563,11 @@ export async function runWorker(input) {
562
563
  timeout.progress();
563
564
  input.onLine?.(line);
564
565
  },
565
- // Always wired now (it used to be conditional on the command
566
- // watchdog): the sink only emits tool_execution_end if a
567
- // handler exists, and a completed tool call is the clearest
568
- // progress signal there is. Without it a worker whose tool
569
- // calls all succeed would still look idle to the deadline.
566
+ // Always wired, never conditional on the command watchdog: the
567
+ // sink only emits a tool-execution-end if a handler exists, and
568
+ // a completed tool call is the clearest progress signal there
569
+ // is. Without it a worker whose tool calls all succeed would
570
+ // still look idle to the deadline.
570
571
  onToolResult: r => {
571
572
  timeout.progress();
572
573
  cmdWatch?.onEnd(r.toolCallId);
@@ -579,8 +580,8 @@ export async function runWorker(input) {
579
580
  stallDetector?.noteContext(snapshot.contextWindow);
580
581
  input.onContextUsage?.(snapshot);
581
582
  },
582
- ...(input.contextWindow && input.contextWindow > 0 ?
583
- { contextWindow: input.contextWindow }
583
+ ...(contextWindowTokens(input.contextWindow) > 0 ?
584
+ { contextWindow: contextWindowTokens(input.contextWindow) }
584
585
  : {})
585
586
  }, input.spawn);
586
587
  }
@@ -601,6 +602,7 @@ export async function runWorker(input) {
601
602
  wallMs: tEnd - tAttemptStart,
602
603
  waitMs,
603
604
  workMs,
605
+ partialChars: text.trim().length,
604
606
  ...(detail ? { detail } : {})
605
607
  };
606
608
  restarts.push(record);
@@ -628,7 +630,7 @@ export async function runWorker(input) {
628
630
  // there would just mislabel the real failure.
629
631
  const leaked = result.exitCode === 0 && !result.aborted ? detectLeakedToolCall(text) : null;
630
632
  // THE RESTART LADDER. Precedence is RESTART_RULES' row order; this loop
631
- // owns the ritual every rule used to repeat: budget, hint, counters,
633
+ // owns the ritual every rule would otherwise repeat: budget, hint, counters,
632
634
  // record-and-announce, backoff, re-spawn.
633
635
  const state = {
634
636
  ...(loopHit ? { loopHit } : {}),
@@ -669,24 +671,22 @@ export async function runWorker(input) {
669
671
  }
670
672
  if (restarted)
671
673
  continue;
672
- // SALVAGE. The run used to return the LAST attempt's text unconditionally,
673
- // so a worker whose final attempt was killed early reported nothing at all
674
- // — even when a discarded attempt had produced a usable answer that was
675
- // still in hand at the moment it was thrown away. A restart budget is
676
- // meant to buy more chances at an answer, not to overwrite a good attempt
677
- // with a worse one.
674
+ // SALVAGE. Returning the LAST attempt's text unconditionally makes a
675
+ // worker whose final attempt was killed early report nothing at all — even
676
+ // when a discarded attempt produced a usable answer that was still in hand
677
+ // at the moment it was thrown away. A restart budget is meant to buy more
678
+ // chances at an answer, not to overwrite a good attempt with a worse one.
678
679
  //
679
680
  // Gated on the final attempt having FAILED, not on it being shorter. A
680
681
  // worker that finished cleanly has answered, and a short answer is a
681
682
  // legitimate answer — length would let a long half-finished fragment
682
683
  // override a concise correct one, which is the opposite of the fix.
683
- // ASK THE LADDER — do not restate it. This was an eight-term disjunction,
684
- // a fifth hand-written statement of the taxonomy `worker-failure.ts` exists
685
- // to own, and it had already drifted: `leakedToolCall` and a plain non-zero
686
- // `exitCode` are rows in FAILURE_RULES and were missing here. Both are cases
687
- // where an attempt that produced nothing usable counted as NOT failed, so
688
- // salvage was skipped and a good earlier partial was overwritten — the exact
689
- // outcome the comment above forbids.
684
+ // ASK THE LADDER — do not restate it. Hand-writing this test restates the
685
+ // taxonomy `worker-failure.ts` owns, and a restatement drifts: drop
686
+ // `leakedToolCall` or a plain non-zero `exitCode` from it and an attempt
687
+ // that produced nothing usable counts as NOT failed, so salvage is skipped
688
+ // and a good earlier partial is overwritten the outcome the comment above
689
+ // forbids.
690
690
  //
691
691
  // The two non-kill terms stay explicit because `worker-failure.ts`
692
692
  // deliberately excludes them as CONSUMER policy: an empty answer and a