@mjasnikovs/pi-task 0.38.29 → 0.38.30

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (370) hide show
  1. package/dist/config/config.d.ts +70 -70
  2. package/dist/config/config.js +26 -35
  3. package/dist/config/extension-list.d.ts +6 -5
  4. package/dist/config/extension-list.js +3 -2
  5. package/dist/config/reasoning-args.d.ts +9 -7
  6. package/dist/config/reasoning-args.js +12 -10
  7. package/dist/config/reasoning.d.ts +44 -105
  8. package/dist/config/reasoning.js +27 -704
  9. package/dist/config/register.d.ts +34 -48
  10. package/dist/config/register.js +41 -51
  11. package/dist/config/tool-list.d.ts +16 -16
  12. package/dist/config/tool-list.js +1 -1
  13. package/dist/remote/bridge.d.ts +19 -10
  14. package/dist/remote/bridge.js +3 -2
  15. package/dist/remote/broadcast.js +3 -1
  16. package/dist/remote/events.js +12 -11
  17. package/dist/remote/history.d.ts +1 -1
  18. package/dist/remote/protocol.d.ts +6 -3
  19. package/dist/remote/protocol.js +2 -1
  20. package/dist/remote/push.d.ts +16 -16
  21. package/dist/remote/push.js +27 -27
  22. package/dist/remote/register.d.ts +3 -3
  23. package/dist/remote/register.js +17 -19
  24. package/dist/remote/server.d.ts +9 -8
  25. package/dist/remote/server.js +15 -14
  26. package/dist/remote/session-state.d.ts +5 -4
  27. package/dist/remote/session-state.js +8 -5
  28. package/dist/remote/sw.d.ts +7 -6
  29. package/dist/remote/sw.js +7 -6
  30. package/dist/remote/tailscale.d.ts +4 -2
  31. package/dist/remote/tailscale.js +4 -2
  32. package/dist/remote/ui-highlight.js +6 -5
  33. package/dist/remote/ui-render.js +4 -4
  34. package/dist/remote/ui-script.js +24 -24
  35. package/dist/remote/ui-styles.d.ts +1 -1
  36. package/dist/remote/ui-styles.js +10 -13
  37. package/dist/remote/ui-tools.js +9 -6
  38. package/dist/shared/child-extensions.d.ts +29 -17
  39. package/dist/shared/child-extensions.js +29 -17
  40. package/dist/shared/child-output.d.ts +30 -24
  41. package/dist/shared/child-output.js +25 -17
  42. package/dist/shared/child-process.d.ts +47 -40
  43. package/dist/shared/child-process.js +50 -59
  44. package/dist/shared/command-watchdog.d.ts +22 -16
  45. package/dist/shared/command-watchdog.js +28 -21
  46. package/dist/shared/fs-text.d.ts +16 -10
  47. package/dist/shared/fs-text.js +16 -10
  48. package/dist/shared/git-runner.d.ts +25 -25
  49. package/dist/shared/git-runner.js +25 -25
  50. package/dist/shared/leaked-tool-call.d.ts +17 -11
  51. package/dist/shared/leaked-tool-call.js +23 -15
  52. package/dist/shared/model-endpoint.d.ts +29 -16
  53. package/dist/shared/model-endpoint.js +33 -21
  54. package/dist/shared/pi-invocation.d.ts +7 -4
  55. package/dist/shared/pi-invocation.js +12 -7
  56. package/dist/shared/pkg-version.d.ts +13 -5
  57. package/dist/shared/pkg-version.js +13 -5
  58. package/dist/shared/reasoning-capability.d.ts +35 -24
  59. package/dist/shared/reasoning-capability.js +35 -24
  60. package/dist/shared/stream-watchdog.d.ts +60 -44
  61. package/dist/shared/stream-watchdog.js +62 -45
  62. package/dist/task/accept-debt.d.ts +41 -43
  63. package/dist/task/accept-debt.js +73 -65
  64. package/dist/task/api-synthesis.d.ts +24 -21
  65. package/dist/task/api-synthesis.js +32 -26
  66. package/dist/task/apis-contract.d.ts +32 -64
  67. package/dist/task/apis-contract.js +32 -64
  68. package/dist/task/artifact-closure.d.ts +27 -13
  69. package/dist/task/artifact-closure.js +95 -67
  70. package/dist/task/auto-commit.d.ts +46 -35
  71. package/dist/task/auto-commit.js +51 -38
  72. package/dist/task/auto-io.d.ts +45 -25
  73. package/dist/task/auto-io.js +57 -29
  74. package/dist/task/auto-orchestrator.d.ts +26 -24
  75. package/dist/task/auto-orchestrator.js +178 -162
  76. package/dist/task/auto-prompts.d.ts +36 -24
  77. package/dist/task/auto-prompts.js +40 -26
  78. package/dist/task/autofix-ledger.d.ts +27 -25
  79. package/dist/task/autofix-ledger.js +29 -26
  80. package/dist/task/batch-test-task.d.ts +20 -12
  81. package/dist/task/batch-test-task.js +67 -60
  82. package/dist/task/boot-probe.d.ts +60 -44
  83. package/dist/task/boot-probe.js +91 -72
  84. package/dist/task/cancel-input.d.ts +30 -16
  85. package/dist/task/cancel-input.js +20 -11
  86. package/dist/task/cancel-points.d.ts +27 -20
  87. package/dist/task/cancel-points.js +30 -22
  88. package/dist/task/child-runner.d.ts +46 -51
  89. package/dist/task/child-runner.js +48 -49
  90. package/dist/task/child-status.d.ts +23 -16
  91. package/dist/task/child-status.js +23 -16
  92. package/dist/task/clamp-output.js +12 -5
  93. package/dist/task/command-run.d.ts +31 -28
  94. package/dist/task/command-run.js +44 -35
  95. package/dist/task/command-shrink.d.ts +25 -18
  96. package/dist/task/command-shrink.js +37 -31
  97. package/dist/task/command-watchdog.d.ts +9 -6
  98. package/dist/task/command-watchdog.js +21 -15
  99. package/dist/task/context-attribution.d.ts +34 -26
  100. package/dist/task/context-attribution.js +34 -26
  101. package/dist/task/context-silence.d.ts +39 -29
  102. package/dist/task/context-silence.js +35 -25
  103. package/dist/task/context-usage.d.ts +16 -9
  104. package/dist/task/context-usage.js +16 -9
  105. package/dist/task/contracts.d.ts +8 -4
  106. package/dist/task/contracts.js +25 -17
  107. package/dist/task/coverage-loop.d.ts +22 -18
  108. package/dist/task/coverage-loop.js +35 -30
  109. package/dist/task/critique-probes.d.ts +13 -14
  110. package/dist/task/critique-probes.js +50 -39
  111. package/dist/task/debug-log.d.ts +13 -5
  112. package/dist/task/debug-log.js +32 -20
  113. package/dist/task/decompose-fidelity.d.ts +11 -9
  114. package/dist/task/decompose-fidelity.js +38 -33
  115. package/dist/task/decompose-granularity.d.ts +41 -38
  116. package/dist/task/decompose-granularity.js +41 -38
  117. package/dist/task/deep-render-check.d.ts +22 -14
  118. package/dist/task/deep-render-check.js +40 -31
  119. package/dist/task/dropped-input.d.ts +12 -7
  120. package/dist/task/dropped-input.js +5 -2
  121. package/dist/task/enforce-attribution.d.ts +38 -47
  122. package/dist/task/enforce-attribution.js +46 -52
  123. package/dist/task/enforce-guidelines.d.ts +31 -20
  124. package/dist/task/enforce-guidelines.js +32 -21
  125. package/dist/task/enrichment.d.ts +7 -2
  126. package/dist/task/enrichment.js +26 -14
  127. package/dist/task/env-notes.d.ts +16 -7
  128. package/dist/task/env-notes.js +48 -31
  129. package/dist/task/env-template-closure.d.ts +4 -4
  130. package/dist/task/env-template-closure.js +42 -34
  131. package/dist/task/external-context.d.ts +28 -21
  132. package/dist/task/external-context.js +17 -12
  133. package/dist/task/failure-classifier.d.ts +4 -5
  134. package/dist/task/failure-classifier.js +6 -7
  135. package/dist/task/file-inventory.d.ts +15 -11
  136. package/dist/task/file-inventory.js +25 -22
  137. package/dist/task/final-gate-fix.d.ts +74 -86
  138. package/dist/task/final-gate-fix.js +97 -116
  139. package/dist/task/final-gate-progress.d.ts +29 -46
  140. package/dist/task/final-gate-progress.js +40 -51
  141. package/dist/task/final-gate.d.ts +64 -97
  142. package/dist/task/final-gate.js +192 -199
  143. package/dist/task/fix-child.d.ts +21 -27
  144. package/dist/task/fix-child.js +21 -27
  145. package/dist/task/foreign-path.d.ts +6 -5
  146. package/dist/task/foreign-path.js +0 -0
  147. package/dist/task/frozen-conflict.d.ts +9 -10
  148. package/dist/task/frozen-conflict.js +61 -64
  149. package/dist/task/frozen-path-guard.d.ts +35 -14
  150. package/dist/task/frozen-path-guard.js +56 -39
  151. package/dist/task/gate-child.d.ts +27 -28
  152. package/dist/task/gate-child.js +36 -35
  153. package/dist/task/gate-deps.d.ts +34 -27
  154. package/dist/task/gate-deps.js +169 -159
  155. package/dist/task/gate-tally.d.ts +77 -80
  156. package/dist/task/gate-tally.js +65 -68
  157. package/dist/task/git-state-guard.d.ts +15 -11
  158. package/dist/task/git-state-guard.js +76 -66
  159. package/dist/task/impl-widget.d.ts +25 -16
  160. package/dist/task/impl-widget.js +27 -17
  161. package/dist/task/implementation-thinking.d.ts +33 -31
  162. package/dist/task/implementation-thinking.js +5 -6
  163. package/dist/task/implementation-turn.d.ts +34 -31
  164. package/dist/task/implementation-turn.js +29 -27
  165. package/dist/task/inline-markdown.d.ts +20 -7
  166. package/dist/task/inline-markdown.js +15 -6
  167. package/dist/task/launch-config-gap.js +25 -39
  168. package/dist/task/launch-contract.d.ts +18 -21
  169. package/dist/task/launch-contract.js +28 -30
  170. package/dist/task/launch-manifest.d.ts +6 -2
  171. package/dist/task/launch-manifest.js +35 -34
  172. package/dist/task/ledger.js +16 -14
  173. package/dist/task/lint-fix.d.ts +6 -8
  174. package/dist/task/lint-fix.js +67 -69
  175. package/dist/task/loop-detector.d.ts +9 -8
  176. package/dist/task/loop-detector.js +16 -12
  177. package/dist/task/mid-run-input.d.ts +17 -15
  178. package/dist/task/mid-run-input.js +17 -15
  179. package/dist/task/orchestrator.d.ts +24 -28
  180. package/dist/task/orchestrator.js +62 -64
  181. package/dist/task/orientation.d.ts +18 -23
  182. package/dist/task/orientation.js +24 -31
  183. package/dist/task/owned-freeze-conflict.d.ts +21 -20
  184. package/dist/task/owned-freeze-conflict.js +52 -85
  185. package/dist/task/owned-freeze-reassign.d.ts +40 -60
  186. package/dist/task/owned-freeze-reassign.js +41 -61
  187. package/dist/task/parsers.d.ts +4 -2
  188. package/dist/task/parsers.js +4 -4
  189. package/dist/task/phases.d.ts +41 -48
  190. package/dist/task/phases.js +179 -248
  191. package/dist/task/plan-io.d.ts +6 -7
  192. package/dist/task/plan-io.js +6 -7
  193. package/dist/task/plan-orchestrator.d.ts +10 -8
  194. package/dist/task/plan-orchestrator.js +14 -10
  195. package/dist/task/plan-prompts.d.ts +6 -5
  196. package/dist/task/plan-prompts.js +6 -5
  197. package/dist/task/plan-readonly.d.ts +4 -5
  198. package/dist/task/plan-readonly.js +4 -5
  199. package/dist/task/plan-rounds.d.ts +17 -29
  200. package/dist/task/plan-rounds.js +21 -34
  201. package/dist/task/plan-session.d.ts +58 -72
  202. package/dist/task/plan-session.js +61 -83
  203. package/dist/task/probe-gaming.d.ts +28 -27
  204. package/dist/task/probe-gaming.js +0 -0
  205. package/dist/task/prohibition-probe.d.ts +14 -16
  206. package/dist/task/prompts.d.ts +3 -4
  207. package/dist/task/prompts.js +17 -26
  208. package/dist/task/qa-transcript.d.ts +15 -22
  209. package/dist/task/qa-transcript.js +15 -21
  210. package/dist/task/question-box.d.ts +17 -13
  211. package/dist/task/question-box.js +19 -15
  212. package/dist/task/question-dedup.d.ts +6 -7
  213. package/dist/task/question-dedup.js +13 -14
  214. package/dist/task/question-dialog.d.ts +22 -32
  215. package/dist/task/question-dialog.js +22 -32
  216. package/dist/task/question-source.d.ts +18 -44
  217. package/dist/task/question-source.js +22 -51
  218. package/dist/task/refuted-constraint.d.ts +11 -31
  219. package/dist/task/refuted-constraint.js +27 -51
  220. package/dist/task/regenerable-artifacts.d.ts +12 -31
  221. package/dist/task/regenerable-artifacts.js +12 -31
  222. package/dist/task/render-check.d.ts +11 -22
  223. package/dist/task/render-check.js +33 -46
  224. package/dist/task/repo-health-check.d.ts +10 -14
  225. package/dist/task/repo-health-check.js +17 -23
  226. package/dist/task/requirements.d.ts +38 -71
  227. package/dist/task/requirements.js +78 -126
  228. package/dist/task/research-fanout-budget.d.ts +51 -88
  229. package/dist/task/research-fanout-budget.js +51 -88
  230. package/dist/task/research-worker.d.ts +29 -39
  231. package/dist/task/research-worker.js +37 -61
  232. package/dist/task/resume-gap.d.ts +14 -15
  233. package/dist/task/root-cause-repair.d.ts +9 -9
  234. package/dist/task/root-cause-repair.js +28 -40
  235. package/dist/task/run-bracket.d.ts +10 -13
  236. package/dist/task/run-end.d.ts +12 -22
  237. package/dist/task/run-end.js +8 -16
  238. package/dist/task/run-final-gate.d.ts +19 -21
  239. package/dist/task/run-final-gate.js +62 -80
  240. package/dist/task/runner-globs.d.ts +12 -13
  241. package/dist/task/runner-globs.js +12 -13
  242. package/dist/task/runner-resolve.d.ts +9 -9
  243. package/dist/task/runner-resolve.js +22 -23
  244. package/dist/task/script-escape.d.ts +10 -12
  245. package/dist/task/script-escape.js +13 -14
  246. package/dist/task/serve-entry.d.ts +1 -1
  247. package/dist/task/serve-entry.js +22 -25
  248. package/dist/task/service-blocks.js +4 -2
  249. package/dist/task/shipped-source.d.ts +11 -29
  250. package/dist/task/shipped-source.js +11 -29
  251. package/dist/task/skip-escape.js +10 -14
  252. package/dist/task/spec-urls.d.ts +26 -65
  253. package/dist/task/spec-urls.js +26 -65
  254. package/dist/task/spec-validation.d.ts +17 -20
  255. package/dist/task/spec-validation.js +17 -20
  256. package/dist/task/stall-detector.d.ts +23 -30
  257. package/dist/task/stall-detector.js +23 -30
  258. package/dist/task/stream-watchdog.d.ts +14 -12
  259. package/dist/task/stream-watchdog.js +14 -12
  260. package/dist/task/substitution-probe.d.ts +17 -20
  261. package/dist/task/substitution-probe.js +17 -20
  262. package/dist/task/task-gates.d.ts +36 -41
  263. package/dist/task/task-gates.js +95 -106
  264. package/dist/task/task-io.d.ts +4 -4
  265. package/dist/task/task-io.js +4 -4
  266. package/dist/task/task-parsers.js +4 -3
  267. package/dist/task/task-provenance.d.ts +2 -2
  268. package/dist/task/task-provenance.js +11 -13
  269. package/dist/task/task-types.d.ts +4 -3
  270. package/dist/task/terminal-outcome.d.ts +14 -16
  271. package/dist/task/terminal-outcome.js +12 -14
  272. package/dist/task/test-assembly.d.ts +13 -20
  273. package/dist/task/test-assembly.js +13 -20
  274. package/dist/task/timings.d.ts +5 -3
  275. package/dist/task/timings.js +5 -3
  276. package/dist/task/title-label.d.ts +9 -4
  277. package/dist/task/title-label.js +9 -4
  278. package/dist/task/type-only-answer.d.ts +44 -52
  279. package/dist/task/type-only-answer.js +44 -52
  280. package/dist/task/unfailable-command.d.ts +18 -24
  281. package/dist/task/unfailable-command.js +21 -27
  282. package/dist/task/unknown-routing.d.ts +10 -4
  283. package/dist/task/unknown-routing.js +10 -4
  284. package/dist/task/user-directives.d.ts +5 -8
  285. package/dist/task/user-directives.js +5 -8
  286. package/dist/task/verify-quality.d.ts +18 -22
  287. package/dist/task/verify-quality.js +45 -46
  288. package/dist/task/verify-reconcile.d.ts +15 -10
  289. package/dist/task/verify-reconcile.js +45 -43
  290. package/dist/task/verify-resolution.d.ts +24 -20
  291. package/dist/task/verify-resolution.js +51 -50
  292. package/dist/task/verify-work.d.ts +59 -66
  293. package/dist/task/verify-work.js +101 -138
  294. package/dist/task/widget.d.ts +15 -14
  295. package/dist/task/widget.js +22 -17
  296. package/dist/task/wiring-claims.d.ts +25 -32
  297. package/dist/task/wiring-claims.js +30 -35
  298. package/dist/task/write-guard.d.ts +39 -39
  299. package/dist/task/write-guard.js +48 -51
  300. package/dist/task/yolo.d.ts +34 -30
  301. package/dist/task/yolo.js +42 -37
  302. package/dist/workers/abstention.d.ts +21 -41
  303. package/dist/workers/abstention.js +27 -48
  304. package/dist/workers/brave-search.d.ts +4 -3
  305. package/dist/workers/brave-search.js +5 -2
  306. package/dist/workers/brave-warning.d.ts +7 -4
  307. package/dist/workers/brave-warning.js +19 -7
  308. package/dist/workers/ddg-search.d.ts +6 -6
  309. package/dist/workers/ddg-search.js +18 -12
  310. package/dist/workers/docs-cache.js +5 -2
  311. package/dist/workers/docs-chunk.d.ts +30 -37
  312. package/dist/workers/docs-chunk.js +37 -41
  313. package/dist/workers/docs-core.d.ts +28 -44
  314. package/dist/workers/docs-core.js +25 -44
  315. package/dist/workers/docs-index.js +4 -3
  316. package/dist/workers/docs-lookup.d.ts +15 -22
  317. package/dist/workers/docs-lookup.js +12 -21
  318. package/dist/workers/docs-project.d.ts +15 -9
  319. package/dist/workers/docs-project.js +17 -10
  320. package/dist/workers/docs-resolve.d.ts +19 -20
  321. package/dist/workers/docs-resolve.js +35 -32
  322. package/dist/workers/docs-retrieve.d.ts +5 -6
  323. package/dist/workers/docs-retrieve.js +18 -15
  324. package/dist/workers/exa-search.d.ts +9 -6
  325. package/dist/workers/exa-search.js +23 -12
  326. package/dist/workers/fetch-core.d.ts +13 -16
  327. package/dist/workers/fetch-core.js +23 -23
  328. package/dist/workers/focused-extractor.d.ts +12 -12
  329. package/dist/workers/focused-extractor.js +16 -19
  330. package/dist/workers/html-clean.js +24 -14
  331. package/dist/workers/http-request.d.ts +28 -20
  332. package/dist/workers/http-request.js +22 -17
  333. package/dist/workers/npm-version.d.ts +28 -11
  334. package/dist/workers/npm-version.js +24 -15
  335. package/dist/workers/phantom-imports.d.ts +15 -12
  336. package/dist/workers/phantom-imports.js +30 -24
  337. package/dist/workers/pi-worker-core.d.ts +69 -71
  338. package/dist/workers/pi-worker-core.js +100 -109
  339. package/dist/workers/pi-worker-docs.d.ts +24 -19
  340. package/dist/workers/pi-worker-docs.js +67 -76
  341. package/dist/workers/pi-worker-fetch.d.ts +7 -3
  342. package/dist/workers/pi-worker-fetch.js +27 -19
  343. package/dist/workers/pi-worker-search.js +12 -8
  344. package/dist/workers/pi-worker.d.ts +9 -4
  345. package/dist/workers/pi-worker.js +21 -14
  346. package/dist/workers/reasoning-warning.d.ts +18 -17
  347. package/dist/workers/reasoning-warning.js +22 -20
  348. package/dist/workers/research-cache.js +50 -78
  349. package/dist/workers/search-core.js +7 -5
  350. package/dist/workers/search-types.d.ts +10 -9
  351. package/dist/workers/search-types.js +9 -8
  352. package/dist/workers/session-hint.d.ts +13 -14
  353. package/dist/workers/session-hint.js +8 -9
  354. package/dist/workers/shared.d.ts +21 -25
  355. package/dist/workers/shared.js +0 -0
  356. package/dist/workers/single-read-extension.d.ts +14 -7
  357. package/dist/workers/single-read-extension.js +14 -7
  358. package/dist/workers/single-read-guard.d.ts +25 -28
  359. package/dist/workers/single-read-guard.js +32 -32
  360. package/dist/workers/typeonly-log.d.ts +12 -9
  361. package/dist/workers/typeonly-log.js +29 -33
  362. package/dist/workers/worker-channels.d.ts +15 -23
  363. package/dist/workers/worker-channels.js +15 -23
  364. package/dist/workers/worker-failure.d.ts +38 -46
  365. package/dist/workers/worker-failure.js +31 -39
  366. package/dist/workers/worker-kill.d.ts +25 -26
  367. package/dist/workers/worker-kill.js +16 -19
  368. package/dist/workers/worker-profiles.d.ts +43 -53
  369. package/dist/workers/worker-profiles.js +30 -38
  370. package/package.json +10 -8
@@ -13,15 +13,15 @@ export function parseVerifyBlock(spec) {
13
13
  /**
14
14
  * parseVerifyBlock, but only when the fenced block is actually CLOSED.
15
15
  *
16
- * An unterminated fence makes the lenient parser swallow the rest of the file:
17
- * mx5 run 19's `TASK_0001.md` opens ```sh and never closes it, so its "VERIFY
18
- * commands" include the phase-timings table and every appended gate-trail line.
16
+ * An unterminated fence makes the lenient parser swallow the rest of the file: a
17
+ * task file that opens ```sh and never closes it yields "VERIFY commands" that
18
+ * include every line appended after the spec — a timings table, a gate-trail line.
19
19
  * That is harmless where the parser only asks "is there something runnable here",
20
- * and NOT harmless where a parsed line is treated as provenance a debt reason
21
- * quoting `bun run lint` would match a gate-trail sentence and mint a stored,
20
+ * and NOT harmless where a parsed line is treated as PROVENANCE: a debt reason
21
+ * quoting `bun run lint` would then match a trail sentence and mint a stored,
22
22
  * re-runnable command the spec never asked for (accept-debt.ts
23
- * verifyCommandFromReason, `inv-command-provenance`). Callers that need the block
24
- * to MEAN something use this one: an unclosed fence is no block at all.
23
+ * `verifyCommandFromReason`, `inv-command-provenance`). Callers that need the
24
+ * block to MEAN something use this one: an unclosed fence is no block at all.
25
25
  */
26
26
  export function parseVerifyBlockStrict(spec) {
27
27
  const scan = scanVerifyBlock(spec);
@@ -119,24 +119,21 @@ export const REFINE_SECTIONS = [
119
119
  /**
120
120
  * Is a refine child's output shaped like a refined prompt?
121
121
  *
122
- * Refine shipped with NO shape check at all `phaseRefine` passes no validator,
123
- * unlike compose and critique and every downstream reader of a refined prompt
124
- * is a PARTIAL parser that tolerates a missing section SILENTLY:
125
- * `extractCapsSection` (refuted-constraint.ts) returns null, `scopedToolingGoal`
126
- * (phases.ts) returns the whole text, `deriveTitle` and `extractEnrichTargets`
127
- * fall back. So a refine answer that dropped a heading degrades four features at
128
- * once and says nothing. This names the contract in one place.
122
+ * Every downstream reader of a refined prompt is a PARTIAL parser that tolerates a
123
+ * missing section SILENTLY: `extractCapsSection` (refuted-constraint.ts) returns
124
+ * null, `scopedToolingGoal` (phases.ts) returns the whole text, and `deriveTitle`
125
+ * (parsers.ts) and `extractEnrichTargets` (enrichment.ts) fall back. So a refine
126
+ * answer that dropped a heading degrades four features at once and says nothing.
127
+ * This names the contract in one place; phases.ts calls it after refine.
129
128
  *
130
129
  * WHAT IT CHECKS, and why it is only this. Every one of those consumers looks
131
130
  * for a BARE ALL-CAPS heading alone on its own line — `l.trim() === heading`,
132
131
  * `/^GOAL[ \t]*\n/m`. That is the operative contract, so that is the test.
133
132
  *
134
- * It does NOT require the text to START with GOAL, even though REFINE_PROMPT
135
- * says "four sections, exact headings, in this order" and forbids a preamble.
136
- * Measured over the 57-task mx5 corpus: 55/56 non-empty refined prompts carry
137
- * all four bare headings, but only 25/56 open with one a preamble is what real
138
- * refine output usually looks like, and production has always consumed it fine.
139
- * A validator stricter than its consumers would reject work that works.
133
+ * It does NOT require the text to START with GOAL, even though REFINE_PROMPT asks
134
+ * for the four headings in order and forbids a preamble. None of the four
135
+ * consumers above cares where the heading sits, so a validator that did would
136
+ * reject work every one of them handles.
140
137
  *
141
138
  * Returns a problem string, or null when the shape is good — same contract as
142
139
  * `validateSpecShape` above.
@@ -2,16 +2,11 @@
2
2
  * Progress-based runaway guard for phase children — the replacement for a
3
3
  * wall-clock cap.
4
4
  *
5
- * WHY NOT SECONDS. PHASE_CHILD_TIMEOUT_MS was sized against "measured HEALTHY
6
- * planning children" on one local backend: decompose 89s, so 600s looked like a
7
- * 6x margin. That sizing is not a property of the pathology, it is a property of
8
- * that day's model, that day's samplers and that day's design doc. Measured on
9
- * the same 27B backend with reasoning ON (2026-08-17, n=10 replays of one
10
- * captured auto-decompose request, everything else byte-identical): every single
11
- * healthy run took 610-927s and produced 26-42 correct titles. The cap would
12
- * have killed 10 out of 10 GOOD runs. A slower model, a bigger design doc or a
13
- * longer reasoning budget moves that number again — so any constant in seconds
14
- * is wrong for someone.
5
+ * WHY NOT SECONDS. A wall clock has to be sized against how long a HEALTHY child
6
+ * takes, and that is not a property of the pathology it is a property of the
7
+ * model, its sampler settings, the reasoning budget and the size of the document
8
+ * being read. Any of those moving turns the margin into a killer of good runs.
9
+ * `PHASE_CHILD_TIMEOUT_MS` is 0 (off) for exactly that reason.
15
10
  *
16
11
  * WHAT REPLACES IT. Two bounds, both dimensionless — invariant to model speed,
17
12
  * project size and reasoning budget:
@@ -22,26 +17,25 @@
22
17
  * never trips, however slow it is. A child re-opening the same four files
23
18
  * trips after NO_PROGRESS_LIMIT of them, however fast it is.
24
19
  *
25
- * Judged on the RESULT, not the arguments, because the arguments lie. The
26
- * thrash measured on 2026-08-17 was 197 of 200 calls REFUSED by the
27
- * single-read guard, each at a different rising offset so by arguments it
28
- * looked like textbook forward paging, and both an offset rule and the loop
29
- * detector's path rule waved all 200 through. By result it is 197 identical
30
- * refusals in a row, which is what it actually was.
20
+ * Judged on the RESULT, not the arguments, because the arguments lie. A child
21
+ * whose reads are all REFUSED by the single-read guard, say — can issue
22
+ * them at steadily rising offsets, so by arguments it looks like textbook
23
+ * forward paging and an offset rule waves every one through. By result it is
24
+ * one identical refusal after another.
31
25
  *
32
26
  * 2. CONTEXT CHURN. Sum the bytes of tool RESULTS the child has pulled in. Once
33
27
  * that exceeds CONTEXT_CHURN_FACTOR times its own context window and it
34
28
  * still has not answered, it has necessarily forgotten what it read first
35
- * and is re-reading to fill a window pi keeps compacting. That is the
36
- * mx5-n 2026-08-14 shape: 16m23s at 117,370 of a 120,064-token window,
37
- * ~56k tokens of tool output per minute, forward-paging the whole time so
38
- * rule 1 alone would not have caught it. The bound scales with the model's
29
+ * and is re-reading to fill a window pi keeps compacting. This catches the
30
+ * child rule 1 cannot: one that pages FORWARD the whole time, pulling in new
31
+ * bytes on every call, and never converges. The bound scales with the model's
39
32
  * OWN window, so a 1M-context model gets a 1M-context allowance.
40
33
  *
41
34
  * The window is supplied by the PARENT at spawn (`PhaseDeps.contextWindow`),
42
- * not read off the child's event stream: pi emits no context event at all,
43
- * which is why this rule sat disarmed from the day it was written until
44
- * GitHub issue #16.
35
+ * not read off the child's event stream: pi's `--mode json` stream carries no
36
+ * context event at all. A detector that waited to be told one sits at 0, and
37
+ * `churnTripped` returns false on a non-positive window — so the rule would
38
+ * never fire.
45
39
  *
46
40
  * Neither rule can fire on a child that is thinking rather than calling tools:
47
41
  * that case is bounded by the model's max tokens (server-enforced) and by the
@@ -58,11 +52,10 @@ import type { LoopHit, ToolCall } from '../shared/child-process.js';
58
52
  *
59
53
  * Eight, because the honest reasons to get back something you have already seen
60
54
  * are few and bounded: re-checking a file after an edit, a grep that lands in a
61
- * file already read, a retry after a malformed call, a missing path. A child
62
- * doing real work interleaves those with progress and resets the counter. In the
63
- * replayed thrash the counter never resets at all the observed runs made
64
- * 188-201 consecutive dead calls. The gap between "a handful" and "two hundred"
65
- * is wide enough that the exact value is not load-bearing.
55
+ * file already read, a retry after a malformed call, a missing path. A child doing
56
+ * real work interleaves those with progress and RESETS the counter, so the streak
57
+ * only survives when nothing at all is being learned. The exact value is not
58
+ * load-bearing: a genuine thrash never resets and runs on indefinitely.
66
59
  */
67
60
  export declare const NO_PROGRESS_LIMIT = 8;
68
61
  /**
@@ -109,7 +102,7 @@ export declare class StallDetector {
109
102
  /**
110
103
  * Restart hint for a child killed by the stall detector. Names the specific
111
104
  * mistake — re-reading covered ground vs pulling in more than it can hold —
112
- * because "you ran out of time" (the old wall-clock hint) told a model that was
113
- * working correctly but slowly to truncate its work for no reason.
105
+ * because "you ran out of time" tells a model that was working correctly but
106
+ * slowly to truncate its work for no reason.
114
107
  */
115
108
  export declare function formatStallHint(kind: StallKind): string;
@@ -2,16 +2,11 @@
2
2
  * Progress-based runaway guard for phase children — the replacement for a
3
3
  * wall-clock cap.
4
4
  *
5
- * WHY NOT SECONDS. PHASE_CHILD_TIMEOUT_MS was sized against "measured HEALTHY
6
- * planning children" on one local backend: decompose 89s, so 600s looked like a
7
- * 6x margin. That sizing is not a property of the pathology, it is a property of
8
- * that day's model, that day's samplers and that day's design doc. Measured on
9
- * the same 27B backend with reasoning ON (2026-08-17, n=10 replays of one
10
- * captured auto-decompose request, everything else byte-identical): every single
11
- * healthy run took 610-927s and produced 26-42 correct titles. The cap would
12
- * have killed 10 out of 10 GOOD runs. A slower model, a bigger design doc or a
13
- * longer reasoning budget moves that number again — so any constant in seconds
14
- * is wrong for someone.
5
+ * WHY NOT SECONDS. A wall clock has to be sized against how long a HEALTHY child
6
+ * takes, and that is not a property of the pathology it is a property of the
7
+ * model, its sampler settings, the reasoning budget and the size of the document
8
+ * being read. Any of those moving turns the margin into a killer of good runs.
9
+ * `PHASE_CHILD_TIMEOUT_MS` is 0 (off) for exactly that reason.
15
10
  *
16
11
  * WHAT REPLACES IT. Two bounds, both dimensionless — invariant to model speed,
17
12
  * project size and reasoning budget:
@@ -22,26 +17,25 @@
22
17
  * never trips, however slow it is. A child re-opening the same four files
23
18
  * trips after NO_PROGRESS_LIMIT of them, however fast it is.
24
19
  *
25
- * Judged on the RESULT, not the arguments, because the arguments lie. The
26
- * thrash measured on 2026-08-17 was 197 of 200 calls REFUSED by the
27
- * single-read guard, each at a different rising offset so by arguments it
28
- * looked like textbook forward paging, and both an offset rule and the loop
29
- * detector's path rule waved all 200 through. By result it is 197 identical
30
- * refusals in a row, which is what it actually was.
20
+ * Judged on the RESULT, not the arguments, because the arguments lie. A child
21
+ * whose reads are all REFUSED by the single-read guard, say — can issue
22
+ * them at steadily rising offsets, so by arguments it looks like textbook
23
+ * forward paging and an offset rule waves every one through. By result it is
24
+ * one identical refusal after another.
31
25
  *
32
26
  * 2. CONTEXT CHURN. Sum the bytes of tool RESULTS the child has pulled in. Once
33
27
  * that exceeds CONTEXT_CHURN_FACTOR times its own context window and it
34
28
  * still has not answered, it has necessarily forgotten what it read first
35
- * and is re-reading to fill a window pi keeps compacting. That is the
36
- * mx5-n 2026-08-14 shape: 16m23s at 117,370 of a 120,064-token window,
37
- * ~56k tokens of tool output per minute, forward-paging the whole time so
38
- * rule 1 alone would not have caught it. The bound scales with the model's
29
+ * and is re-reading to fill a window pi keeps compacting. This catches the
30
+ * child rule 1 cannot: one that pages FORWARD the whole time, pulling in new
31
+ * bytes on every call, and never converges. The bound scales with the model's
39
32
  * OWN window, so a 1M-context model gets a 1M-context allowance.
40
33
  *
41
34
  * The window is supplied by the PARENT at spawn (`PhaseDeps.contextWindow`),
42
- * not read off the child's event stream: pi emits no context event at all,
43
- * which is why this rule sat disarmed from the day it was written until
44
- * GitHub issue #16.
35
+ * not read off the child's event stream: pi's `--mode json` stream carries no
36
+ * context event at all. A detector that waited to be told one sits at 0, and
37
+ * `churnTripped` returns false on a non-positive window — so the rule would
38
+ * never fire.
45
39
  *
46
40
  * Neither rule can fire on a child that is thinking rather than calling tools:
47
41
  * that case is bounded by the model's max tokens (server-enforced) and by the
@@ -58,11 +52,10 @@ import { stableStringify } from './loop-detector.js';
58
52
  *
59
53
  * Eight, because the honest reasons to get back something you have already seen
60
54
  * are few and bounded: re-checking a file after an edit, a grep that lands in a
61
- * file already read, a retry after a malformed call, a missing path. A child
62
- * doing real work interleaves those with progress and resets the counter. In the
63
- * replayed thrash the counter never resets at all the observed runs made
64
- * 188-201 consecutive dead calls. The gap between "a handful" and "two hundred"
65
- * is wide enough that the exact value is not load-bearing.
55
+ * file already read, a retry after a malformed call, a missing path. A child doing
56
+ * real work interleaves those with progress and RESETS the counter, so the streak
57
+ * only survives when nothing at all is being learned. The exact value is not
58
+ * load-bearing: a genuine thrash never resets and runs on indefinitely.
66
59
  */
67
60
  export const NO_PROGRESS_LIMIT = 8;
68
61
  /**
@@ -146,8 +139,8 @@ export class StallDetector {
146
139
  /**
147
140
  * Restart hint for a child killed by the stall detector. Names the specific
148
141
  * mistake — re-reading covered ground vs pulling in more than it can hold —
149
- * because "you ran out of time" (the old wall-clock hint) told a model that was
150
- * working correctly but slowly to truncate its work for no reason.
142
+ * because "you ran out of time" tells a model that was working correctly but
143
+ * slowly to truncate its work for no reason.
151
144
  */
152
145
  export function formatStallHint(kind) {
153
146
  if (kind === 'context-churn') {
@@ -2,12 +2,12 @@ import type { ExtensionAPI } from '@earendil-works/pi-coding-agent';
2
2
  /**
3
3
  * MAIN-SESSION adapter for the model-stream watchdog.
4
4
  *
5
- * WHY (mx5 run 14): three implementation turns died mid-turn — the session jsonl's
6
- * last record is an ordinary assistant message, then silence forever, while the
7
- * model container stayed Up(healthy). No error is ever thrown for this shape, so
8
- * the connection-error retry (which needs a reported ModelError) cannot fire and
9
- * the command watchdog, which only covers tool executions, never arms. The run sat
10
- * dead for ~2.9h across the three until a human restarted it.
5
+ * WHY: a turn can die mid-stream — the last thing recorded is an ordinary
6
+ * assistant message, then silence forever, with the model endpoint still up.
7
+ * Nothing throws for that shape, so the connection-error retry (which needs a
8
+ * reported ModelError) cannot fire, and the command watchdog only covers tool
9
+ * executions, so it never arms. Without this the session sits dead until a human
10
+ * notices.
11
11
  *
12
12
  * HOW: pi's extension events ARE the stream. Any of them — a token delta, a
13
13
  * thinking delta, a tool-call delta, the provider's response headers — resets the
@@ -17,16 +17,18 @@ import type { ExtensionAPI } from '@earendil-works/pi-coding-agent';
17
17
  * blind re-send would re-run them).
18
18
  *
19
19
  * ONE ABORT CHANNEL: the fire path goes through the command watchdog's existing
20
- * {@link noteWatchdogAbort} flag and its WATCHDOG_CANCEL_MARKER, so
21
- * steerUntilDone's already-fixed abort/steer race (b543d15) covers this watchdog
22
- * too instead of racing a second, parallel abort mechanism.
20
+ * {@link noteWatchdogAbort} flag and its WATCHDOG_CANCEL_MARKER, so the abort/steer
21
+ * race steerUntilDone already handles covers this watchdog too, instead of racing a
22
+ * second, parallel abort mechanism.
23
23
  *
24
24
  * SUSPENDED DURING TOOLS: while a tool executes the model stream is legitimately
25
- * idle — a 12-minute build emits nothing. That window belongs to the command
25
+ * idle — a long build emits nothing on it. That window belongs to the command
26
26
  * watchdog (requestTimeoutMs); this one pauses between tool_execution_start and
27
27
  * tool_execution_end so the two can never double-fire on the same silence.
28
28
  *
29
- * SCOPE: main session only. Children run `--no-extensions`, so their equivalent
30
- * guard lives in runChild (shared/child-process.ts) and shares the same machine.
29
+ * SCOPE: main session only. Children run `--no-extensions` (CHILD_BASE_ARGS), so
30
+ * none of these events reaches them; their equivalent guard is a second
31
+ * `StreamWatchdog` inside runChild (shared/child-process.ts), on the same
32
+ * machine.
31
33
  */
32
34
  export declare function registerStreamWatchdog(pi: ExtensionAPI): void;
@@ -4,12 +4,12 @@ import { noteWatchdogAbort, WATCHDOG_CANCEL_MARKER } from './command-watchdog.js
4
4
  /**
5
5
  * MAIN-SESSION adapter for the model-stream watchdog.
6
6
  *
7
- * WHY (mx5 run 14): three implementation turns died mid-turn — the session jsonl's
8
- * last record is an ordinary assistant message, then silence forever, while the
9
- * model container stayed Up(healthy). No error is ever thrown for this shape, so
10
- * the connection-error retry (which needs a reported ModelError) cannot fire and
11
- * the command watchdog, which only covers tool executions, never arms. The run sat
12
- * dead for ~2.9h across the three until a human restarted it.
7
+ * WHY: a turn can die mid-stream — the last thing recorded is an ordinary
8
+ * assistant message, then silence forever, with the model endpoint still up.
9
+ * Nothing throws for that shape, so the connection-error retry (which needs a
10
+ * reported ModelError) cannot fire, and the command watchdog only covers tool
11
+ * executions, so it never arms. Without this the session sits dead until a human
12
+ * notices.
13
13
  *
14
14
  * HOW: pi's extension events ARE the stream. Any of them — a token delta, a
15
15
  * thinking delta, a tool-call delta, the provider's response headers — resets the
@@ -19,17 +19,19 @@ import { noteWatchdogAbort, WATCHDOG_CANCEL_MARKER } from './command-watchdog.js
19
19
  * blind re-send would re-run them).
20
20
  *
21
21
  * ONE ABORT CHANNEL: the fire path goes through the command watchdog's existing
22
- * {@link noteWatchdogAbort} flag and its WATCHDOG_CANCEL_MARKER, so
23
- * steerUntilDone's already-fixed abort/steer race (b543d15) covers this watchdog
24
- * too instead of racing a second, parallel abort mechanism.
22
+ * {@link noteWatchdogAbort} flag and its WATCHDOG_CANCEL_MARKER, so the abort/steer
23
+ * race steerUntilDone already handles covers this watchdog too, instead of racing a
24
+ * second, parallel abort mechanism.
25
25
  *
26
26
  * SUSPENDED DURING TOOLS: while a tool executes the model stream is legitimately
27
- * idle — a 12-minute build emits nothing. That window belongs to the command
27
+ * idle — a long build emits nothing on it. That window belongs to the command
28
28
  * watchdog (requestTimeoutMs); this one pauses between tool_execution_start and
29
29
  * tool_execution_end so the two can never double-fire on the same silence.
30
30
  *
31
- * SCOPE: main session only. Children run `--no-extensions`, so their equivalent
32
- * guard lives in runChild (shared/child-process.ts) and shares the same machine.
31
+ * SCOPE: main session only. Children run `--no-extensions` (CHILD_BASE_ARGS), so
32
+ * none of these events reaches them; their equivalent guard is a second
33
+ * `StreamWatchdog` inside runChild (shared/child-process.ts), on the same
34
+ * machine.
33
35
  */
34
36
  export function registerStreamWatchdog(pi) {
35
37
  // The ctx whose abort() ends the in-flight turn, refreshed on every event so
@@ -2,19 +2,15 @@
2
2
  * substitution-probe — deterministic detection of SELF-VERIFIED work, feeding the
3
3
  * verify gate's prompt.
4
4
  *
5
- * The failure class (mx5 run 3, TASK_0016/0017): the implementation hit a real bug
6
- * in the shipped app ("Context is not finalized" from a mis-signatured middleware),
7
- * blamed the framework, and shipped "integration tests" that re-implement the API
8
- * inline an own server intercepting protected routes, and a 943-line hand-rolled
9
- * duplicate in a test-support file that imports the real app and never calls it.
10
- * The suite runs 26/26 green while every protected route of the real server 500s.
11
- * The verify child judged "do the tests pass", saw green, and PASSed.
5
+ * The failure class: an implementation that hits a real bug in the shipped app
6
+ * can ship "integration tests" that re-implement the API inline instead — a test
7
+ * file that stands up its own server, or imports the real app and never calls it.
8
+ * The suite then runs green while every route of the real server fails, and a
9
+ * verify child that asks "do the tests pass" sees green and PASSes.
12
10
  *
13
- * A/B on the live local model (5 runs/arm, the real mx5 tree as fixture) proved the
14
- * fix must be a DETERMINISTIC, CONCRETE finding in the prompt, not prompt language
15
- * alone: the bare substitution rule caught 2/5 (right verdicts, 40% attention); the
16
- * rule PLUS a probe finding naming the suspect file caught 5/5, each verdict naming
17
- * the exact inline handlers and confirming the real app crashes.
11
+ * The prompt rule alone is weak against that, because the child has to notice the
12
+ * substitution unprompted. This probe hands it a concrete finding naming the
13
+ * suspect file, so the rule fires on a line rather than on self-discovery.
18
14
  *
19
15
  * WHAT THE PROBE ASSERTS is deliberately the class INVARIANT, not any framework
20
16
  * shape: when a task's own diff includes the very tests whose green result would
@@ -23,15 +19,16 @@
23
19
  * That is computable from pure git diff shape (which changed files live in test
24
20
  * paths, and how many lines the task added there): no language parsing, no server-
25
21
  * constructor lists, no import syntax — a Python, Go, or Rust project produces the
26
- * same finding the same way. (An earlier draft pattern-matched JS server
27
- * constructors; that only re-encoded the one incident. A behavioral canary — rerun
28
- * the suite with shipped sources removed, still-green ⇒ copy — was also rejected:
29
- * the real mx5 copy still imported the shipped db module, so poisoning sources
30
- * breaks the copy's tests too and the conviction never fires.)
22
+ * same finding the same way.
31
23
  *
32
- * Findings are advisory: they mandate the spot-check, they never auto-FAIL. Honest
33
- * self-authored tests survive the spot-check (control fixtures PASS with the
34
- * mandate present); only tests that bypass the artifact get named in a FAIL.
24
+ * A behavioural canary re-run the suite with the shipped sources removed, and
25
+ * treat still-green as proof of a copy does NOT work here: a substituted test
26
+ * usually still imports SOMETHING real (a db module, a type), so poisoning the
27
+ * sources breaks it too and the canary never convicts.
28
+ *
29
+ * Findings are advisory: they mandate the spot-check, they never auto-FAIL. An
30
+ * honest self-authored test survives that spot-check; only one that bypasses the
31
+ * artifact gets named in a FAIL.
35
32
  */
36
33
  /** One changed file as the git-shape collector reports it. */
37
34
  export interface ChangedFile {
@@ -2,19 +2,15 @@
2
2
  * substitution-probe — deterministic detection of SELF-VERIFIED work, feeding the
3
3
  * verify gate's prompt.
4
4
  *
5
- * The failure class (mx5 run 3, TASK_0016/0017): the implementation hit a real bug
6
- * in the shipped app ("Context is not finalized" from a mis-signatured middleware),
7
- * blamed the framework, and shipped "integration tests" that re-implement the API
8
- * inline an own server intercepting protected routes, and a 943-line hand-rolled
9
- * duplicate in a test-support file that imports the real app and never calls it.
10
- * The suite runs 26/26 green while every protected route of the real server 500s.
11
- * The verify child judged "do the tests pass", saw green, and PASSed.
5
+ * The failure class: an implementation that hits a real bug in the shipped app
6
+ * can ship "integration tests" that re-implement the API inline instead — a test
7
+ * file that stands up its own server, or imports the real app and never calls it.
8
+ * The suite then runs green while every route of the real server fails, and a
9
+ * verify child that asks "do the tests pass" sees green and PASSes.
12
10
  *
13
- * A/B on the live local model (5 runs/arm, the real mx5 tree as fixture) proved the
14
- * fix must be a DETERMINISTIC, CONCRETE finding in the prompt, not prompt language
15
- * alone: the bare substitution rule caught 2/5 (right verdicts, 40% attention); the
16
- * rule PLUS a probe finding naming the suspect file caught 5/5, each verdict naming
17
- * the exact inline handlers and confirming the real app crashes.
11
+ * The prompt rule alone is weak against that, because the child has to notice the
12
+ * substitution unprompted. This probe hands it a concrete finding naming the
13
+ * suspect file, so the rule fires on a line rather than on self-discovery.
18
14
  *
19
15
  * WHAT THE PROBE ASSERTS is deliberately the class INVARIANT, not any framework
20
16
  * shape: when a task's own diff includes the very tests whose green result would
@@ -23,15 +19,16 @@
23
19
  * That is computable from pure git diff shape (which changed files live in test
24
20
  * paths, and how many lines the task added there): no language parsing, no server-
25
21
  * constructor lists, no import syntax — a Python, Go, or Rust project produces the
26
- * same finding the same way. (An earlier draft pattern-matched JS server
27
- * constructors; that only re-encoded the one incident. A behavioral canary — rerun
28
- * the suite with shipped sources removed, still-green ⇒ copy — was also rejected:
29
- * the real mx5 copy still imported the shipped db module, so poisoning sources
30
- * breaks the copy's tests too and the conviction never fires.)
22
+ * same finding the same way.
31
23
  *
32
- * Findings are advisory: they mandate the spot-check, they never auto-FAIL. Honest
33
- * self-authored tests survive the spot-check (control fixtures PASS with the
34
- * mandate present); only tests that bypass the artifact get named in a FAIL.
24
+ * A behavioural canary re-run the suite with the shipped sources removed, and
25
+ * treat still-green as proof of a copy does NOT work here: a substituted test
26
+ * usually still imports SOMETHING real (a db module, a type), so poisoning the
27
+ * sources breaks it too and the canary never convicts.
28
+ *
29
+ * Findings are advisory: they mandate the spot-check, they never auto-FAIL. An
30
+ * honest self-authored test survives that spot-check; only one that bypasses the
31
+ * artifact gets named in a FAIL.
35
32
  */
36
33
  /** Is this a file whose job is testing — by suffix or by living in a test dir? */
37
34
  export function isTestFile(p) {
@@ -69,10 +69,8 @@ export interface GateDeps {
69
69
  * it has always been, so production wiring is untouched.
70
70
  *
71
71
  * Its own field because it answers a different question from the gate above.
72
- * While the two shared one field the only way to answer them differently was to
73
- * count invocations a `verifyCalls` state machine whose FIRST return existed
74
- * solely to unlock `mode === 'edit'`, re-invented in the suite and again in
75
- * scripts/enforce-revert-attribution-replay-ab.ts.
72
+ * Sharing one field would leave a caller or a test — no way to answer the two
73
+ * differently except by counting invocations.
76
74
  */
77
75
  reVerify?: (ctx: ExtensionCommandContext, cwd: string, taskTitle: string, taskId: string) => Promise<VerifyOutcome>;
78
76
  /**
@@ -96,10 +94,10 @@ export interface GateDeps {
96
94
  /**
97
95
  * BOUNDED fix for a repo-health verify FAIL: a small read,edit,bash child fixes
98
96
  * exactly the static findings (revert-guarded — see lint-fix.ts), instead of the
99
- * full implementation re-run AUTOFIX reaches for. Validated live: 64s/106s to
100
- * lint-clean where the impl re-run burned 36–56 min and did not converge. Runs
101
- * at most once per gate sequence; not-applied falls through to the picker.
102
- * Absent → the loop goes straight to recommend/picker as before.
97
+ * full implementation re-run AUTOFIX reaches for. Smallest tool first: a static
98
+ * finding does not need the whole turn re-run to fix it. Runs at most once per
99
+ * gate sequence; not-applied falls through to the picker. Absent → the loop goes
100
+ * straight to recommend/picker.
103
101
  */
104
102
  lintFix?: (ctx: ExtensionCommandContext, cwd: string, taskTitle: string, taskId: string, failReason: string) => Promise<{
105
103
  ok: boolean;
@@ -107,15 +105,15 @@ export interface GateDeps {
107
105
  }>;
108
106
  /**
109
107
  * Deterministic whole-repo static check (repo-health), used as the PRE-COMMIT
110
- * gate on an edit-mode enforce pass: live data shows enforce edits broke the
111
- * repo's own lint in 11 of 16 tasks, each costing a commit + model re-verify +
112
- * revert cycle. Checking before committing skips that cycle. Absent → the old
113
- * commit-then-differential path runs unchanged.
108
+ * gate on an edit-mode enforce pass: an enforce edit that breaks the project's
109
+ * own lint would otherwise cost a commit, a model re-verify and a revert before
110
+ * anything noticed. Checking first skips that cycle. Absent → the
111
+ * commit-then-differential path runs on its own.
112
+ *
113
+ * Takes the live ctx and a label so the implementation can render a status line
114
+ * while it runs: it is as slow as the project's own lint, and a gate step that
115
+ * long with no widget is indistinguishable from a hang.
114
116
  */
115
- /** Deterministic whole-repo static check for the enforce pre-commit gate. Takes
116
- * the live ctx and a label so the implementation can render a status line while
117
- * it runs — it is as slow as the project's own lint (15–69s measured), and a
118
- * gate step that long with no widget is indistinguishable from a hang. */
119
117
  repoHealth?: (ctx: ExtensionCommandContext, cwd: string, label: string) => Promise<{
120
118
  ok: boolean;
121
119
  reason: string;
@@ -131,10 +129,10 @@ export interface GateDeps {
131
129
  * Append one line to the task's durable gate trail (`## gates` in the task
132
130
  * file). Every gate outcome — each verify verdict, the user's FAIL resolution,
133
131
  * the commit result, enforce mode + verdict, the differential guard's decision —
134
- * is recorded so the sequence is auditable from artifacts alone (the mx5 audit
135
- * could not tell WHY 10 of 18 tasks show no enforce run: verdicts lived only in
136
- * terminal notifies). Best-effort: absent in tests → skipped; failures are
137
- * swallowed by the implementation, never by this sequence.
132
+ * is recorded so the sequence is auditable from artifacts alone. A verdict that
133
+ * lives only in a terminal notify cannot answer "why did enforce not run here?"
134
+ * after the fact. Best-effort: absent in tests → skipped; failures are swallowed
135
+ * by the implementation, never by this sequence.
138
136
  */
139
137
  record?: (cwd: string, taskId: string, line: string) => Promise<void>;
140
138
  /**
@@ -172,8 +170,8 @@ export interface GateDeps {
172
170
  * verify site); `committed` = the files the task snapshot + the ENFORCE commit
173
171
  * changed (the post-commit enforce site); `enforce-commit` = the ENFORCE COMMIT
174
172
  * ALONE, which is the only correct authorship question at the enforce
175
- * differential (mx5 run 18 / nexttask 4 — that differential decides whether to
176
- * discard the enforce commit, so what the TASK touched is irrelevant to it).
173
+ * differential — that differential decides whether to discard the ENFORCE
174
+ * COMMIT, so what the TASK touched is irrelevant to it.
177
175
  * `null` means UNKNOWN (git unavailable) and stands the whole channel down —
178
176
  * inconclusive is never evidence, so an unreadable tree can only cost a repair
179
177
  * task, never spawn a wrong one or wrongly keep a regression.
@@ -200,8 +198,9 @@ export interface GateDeps {
200
198
  /**
201
199
  * Restore the given frozen paths to their committed (HEAD) state, discarding a
202
200
  * gate child's edits to them, and return the files actually reverted. Prompt
203
- * framing is A/B-proven insufficient for this class, so the deny is mechanical:
204
- * the write is undone, not merely warned about. Absent → the guard warns only.
201
+ * framing is insufficient for this class (see frozen-path-guard.ts), so the deny
202
+ * is mechanical: the write is undone, not merely warned about. Absent → the
203
+ * guard warns only.
205
204
  */
206
205
  revertFrozenPaths?: (cwd: string, paths: string[]) => Promise<string[]>;
207
206
  }
@@ -286,13 +285,12 @@ export interface YoloAcceptContext {
286
285
  /**
287
286
  * The reason an auto-ACCEPT is being written — NAMED, not assumed.
288
287
  *
289
- * The line this replaces asserted "autofix budget spent" on every branch. It was
290
- * measured false in 2709 of 2709 recorded accepts (scripts/yolo-accept-baserate.ts):
291
- * the budget has never once reached MAX_AUTO_AUTOFIX anywhere in the corpus — 30%
292
- * of accepts are UNOBSERVED (the research is never even consulted) and 70% are an
293
- * ACCEPT recommendation with the budget fully untouched. A durable trail that
294
- * misstates why a defect shipped is worse than no trail: mx5 run 19's TASK_0009
295
- * reads as an exhausted fixer when nothing was ever attempted.
288
+ * Four disjoint branches reach the terminal auto-ACCEPT, and only ONE of them has
289
+ * spent the autofix budget. An UNOBSERVED FAIL never consults the research; a
290
+ * frozen-blocked one is a contradiction no re-run can resolve; an ACCEPT
291
+ * recommendation can arrive with the budget untouched. A single "autofix budget
292
+ * spent" line would be false on three of the four, and a durable trail that
293
+ * misstates why a defect shipped reads as an exhausted fixer that never tried.
296
294
  */
297
295
  export declare function yoloAcceptReason(c: YoloAcceptContext): string;
298
296
  /**
@@ -306,8 +304,8 @@ export declare function askVerifyResolution(ctx: ExtensionCommandContext, title:
306
304
  /**
307
305
  * Run the verify + enforce gates against a task's just-finished implementation.
308
306
  *
309
- * Lifted verbatim from /task-auto's per-task loop so the two commands gate
310
- * identically. Returns a GateResult; `done` means the caller should proceed (the
307
+ * Both commands gate identically because both call this. Returns a GateResult;
308
+ * `done` means the caller should proceed (the
311
309
  * work is verified-or-accepted, checked off, committed, and enforced), every other
312
310
  * kind is a terminal stop the caller announces. Never throws for a gate outcome —
313
311
  * only a user cancel inside a gate child propagates (handled by the caller's
@@ -346,12 +344,10 @@ type VerifyGateStep = {
346
344
  * recommendation → unattended autofix → picker) until it verifies, is accepted,
347
345
  * or terminates.
348
346
  *
349
- * Split out of `runGatesForTask` at the single boolean that crosses to the
350
- * ENFORCE half. It carries 8 mutable locals over ~290 lines and has four terminal
351
- * exits; enforce carries one local and always falls through. Joining them meant a
352
- * test of the enforce differential had to traverse this whole loop first, which is
353
- * why `deps.verify` was driven by an invocation counter whose first return existed
354
- * only to unlock `mode === 'edit'`.
347
+ * Split from `runGatesForTask` at the single boolean that crosses to the ENFORCE
348
+ * half (`cleanPass`). This loop has four terminal exits and carries the whole
349
+ * negotiation; enforce always falls through. Joined, a test of the enforce
350
+ * differential would have to traverse this entire loop first.
355
351
  */
356
352
  export declare function resolveVerifyGate(ctxIn: ExtensionCommandContext, deps: GateDeps, p: GateParams, rec: Recorder, routeRootCause: RootCauseRouter): Promise<VerifyGateStep>;
357
353
  /**
@@ -359,8 +355,7 @@ export declare function resolveVerifyGate(ctxIn: ExtensionCommandContext, deps:
359
355
  * then decide whether the pass\'s own commit survives.
360
356
  *
361
357
  * `reVerify` is deliberately NOT `deps.verify`. They answer two different
362
- * questions — the gate above, and this differential — and while they shared one
363
- * field the only way to answer them differently was to count invocations.
358
+ * questions — the gate above, and this differential — so they are two fields.
364
359
  *
365
360
  * Reads `active` and never reassigns it: nothing here can replace the live
366
361
  * session, unlike the autofix in the verify half.