@mjasnikovs/pi-task 0.38.28 → 0.38.30

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (370) hide show
  1. package/dist/config/config.d.ts +70 -70
  2. package/dist/config/config.js +26 -35
  3. package/dist/config/extension-list.d.ts +6 -5
  4. package/dist/config/extension-list.js +3 -2
  5. package/dist/config/reasoning-args.d.ts +9 -7
  6. package/dist/config/reasoning-args.js +12 -10
  7. package/dist/config/reasoning.d.ts +44 -105
  8. package/dist/config/reasoning.js +27 -704
  9. package/dist/config/register.d.ts +34 -48
  10. package/dist/config/register.js +41 -51
  11. package/dist/config/tool-list.d.ts +16 -16
  12. package/dist/config/tool-list.js +1 -1
  13. package/dist/remote/bridge.d.ts +19 -10
  14. package/dist/remote/bridge.js +3 -2
  15. package/dist/remote/broadcast.js +3 -1
  16. package/dist/remote/events.js +12 -11
  17. package/dist/remote/history.d.ts +1 -1
  18. package/dist/remote/protocol.d.ts +6 -3
  19. package/dist/remote/protocol.js +2 -1
  20. package/dist/remote/push.d.ts +16 -16
  21. package/dist/remote/push.js +27 -27
  22. package/dist/remote/register.d.ts +3 -3
  23. package/dist/remote/register.js +17 -19
  24. package/dist/remote/server.d.ts +9 -8
  25. package/dist/remote/server.js +15 -14
  26. package/dist/remote/session-state.d.ts +5 -4
  27. package/dist/remote/session-state.js +8 -5
  28. package/dist/remote/sw.d.ts +7 -6
  29. package/dist/remote/sw.js +7 -6
  30. package/dist/remote/tailscale.d.ts +4 -2
  31. package/dist/remote/tailscale.js +4 -2
  32. package/dist/remote/ui-highlight.js +6 -5
  33. package/dist/remote/ui-render.js +4 -4
  34. package/dist/remote/ui-script.js +24 -24
  35. package/dist/remote/ui-styles.d.ts +1 -1
  36. package/dist/remote/ui-styles.js +10 -13
  37. package/dist/remote/ui-tools.js +9 -6
  38. package/dist/shared/child-extensions.d.ts +29 -17
  39. package/dist/shared/child-extensions.js +29 -17
  40. package/dist/shared/child-output.d.ts +30 -24
  41. package/dist/shared/child-output.js +25 -17
  42. package/dist/shared/child-process.d.ts +47 -40
  43. package/dist/shared/child-process.js +50 -59
  44. package/dist/shared/command-watchdog.d.ts +22 -16
  45. package/dist/shared/command-watchdog.js +28 -21
  46. package/dist/shared/fs-text.d.ts +16 -10
  47. package/dist/shared/fs-text.js +16 -10
  48. package/dist/shared/git-runner.d.ts +25 -25
  49. package/dist/shared/git-runner.js +25 -25
  50. package/dist/shared/leaked-tool-call.d.ts +17 -11
  51. package/dist/shared/leaked-tool-call.js +23 -15
  52. package/dist/shared/model-endpoint.d.ts +29 -16
  53. package/dist/shared/model-endpoint.js +33 -21
  54. package/dist/shared/pi-invocation.d.ts +7 -4
  55. package/dist/shared/pi-invocation.js +12 -7
  56. package/dist/shared/pkg-version.d.ts +13 -5
  57. package/dist/shared/pkg-version.js +13 -5
  58. package/dist/shared/reasoning-capability.d.ts +35 -24
  59. package/dist/shared/reasoning-capability.js +35 -24
  60. package/dist/shared/stream-watchdog.d.ts +60 -44
  61. package/dist/shared/stream-watchdog.js +62 -45
  62. package/dist/task/accept-debt.d.ts +41 -43
  63. package/dist/task/accept-debt.js +73 -65
  64. package/dist/task/api-synthesis.d.ts +24 -21
  65. package/dist/task/api-synthesis.js +32 -26
  66. package/dist/task/apis-contract.d.ts +32 -64
  67. package/dist/task/apis-contract.js +32 -64
  68. package/dist/task/artifact-closure.d.ts +27 -13
  69. package/dist/task/artifact-closure.js +95 -67
  70. package/dist/task/auto-commit.d.ts +46 -35
  71. package/dist/task/auto-commit.js +51 -38
  72. package/dist/task/auto-io.d.ts +45 -25
  73. package/dist/task/auto-io.js +57 -29
  74. package/dist/task/auto-orchestrator.d.ts +26 -24
  75. package/dist/task/auto-orchestrator.js +178 -162
  76. package/dist/task/auto-prompts.d.ts +36 -24
  77. package/dist/task/auto-prompts.js +40 -26
  78. package/dist/task/autofix-ledger.d.ts +27 -25
  79. package/dist/task/autofix-ledger.js +29 -26
  80. package/dist/task/batch-test-task.d.ts +20 -12
  81. package/dist/task/batch-test-task.js +67 -60
  82. package/dist/task/boot-probe.d.ts +60 -44
  83. package/dist/task/boot-probe.js +91 -72
  84. package/dist/task/cancel-input.d.ts +30 -16
  85. package/dist/task/cancel-input.js +20 -11
  86. package/dist/task/cancel-points.d.ts +27 -20
  87. package/dist/task/cancel-points.js +30 -22
  88. package/dist/task/child-runner.d.ts +46 -51
  89. package/dist/task/child-runner.js +48 -49
  90. package/dist/task/child-status.d.ts +23 -16
  91. package/dist/task/child-status.js +23 -16
  92. package/dist/task/clamp-output.js +12 -5
  93. package/dist/task/command-run.d.ts +31 -28
  94. package/dist/task/command-run.js +44 -35
  95. package/dist/task/command-shrink.d.ts +25 -18
  96. package/dist/task/command-shrink.js +37 -31
  97. package/dist/task/command-watchdog.d.ts +9 -6
  98. package/dist/task/command-watchdog.js +21 -15
  99. package/dist/task/context-attribution.d.ts +34 -26
  100. package/dist/task/context-attribution.js +34 -26
  101. package/dist/task/context-silence.d.ts +39 -29
  102. package/dist/task/context-silence.js +35 -25
  103. package/dist/task/context-usage.d.ts +25 -7
  104. package/dist/task/context-usage.js +21 -6
  105. package/dist/task/contracts.d.ts +8 -4
  106. package/dist/task/contracts.js +25 -17
  107. package/dist/task/coverage-loop.d.ts +22 -18
  108. package/dist/task/coverage-loop.js +35 -30
  109. package/dist/task/critique-probes.d.ts +13 -14
  110. package/dist/task/critique-probes.js +50 -39
  111. package/dist/task/debug-log.d.ts +13 -5
  112. package/dist/task/debug-log.js +32 -20
  113. package/dist/task/decompose-fidelity.d.ts +11 -9
  114. package/dist/task/decompose-fidelity.js +38 -33
  115. package/dist/task/decompose-granularity.d.ts +41 -38
  116. package/dist/task/decompose-granularity.js +41 -38
  117. package/dist/task/deep-render-check.d.ts +22 -14
  118. package/dist/task/deep-render-check.js +40 -31
  119. package/dist/task/dropped-input.d.ts +12 -7
  120. package/dist/task/dropped-input.js +5 -2
  121. package/dist/task/enforce-attribution.d.ts +38 -47
  122. package/dist/task/enforce-attribution.js +46 -52
  123. package/dist/task/enforce-guidelines.d.ts +31 -20
  124. package/dist/task/enforce-guidelines.js +32 -21
  125. package/dist/task/enrichment.d.ts +7 -2
  126. package/dist/task/enrichment.js +26 -14
  127. package/dist/task/env-notes.d.ts +16 -7
  128. package/dist/task/env-notes.js +48 -31
  129. package/dist/task/env-template-closure.d.ts +4 -4
  130. package/dist/task/env-template-closure.js +42 -34
  131. package/dist/task/external-context.d.ts +28 -21
  132. package/dist/task/external-context.js +17 -12
  133. package/dist/task/failure-classifier.d.ts +4 -5
  134. package/dist/task/failure-classifier.js +6 -7
  135. package/dist/task/file-inventory.d.ts +15 -11
  136. package/dist/task/file-inventory.js +25 -22
  137. package/dist/task/final-gate-fix.d.ts +74 -86
  138. package/dist/task/final-gate-fix.js +97 -116
  139. package/dist/task/final-gate-progress.d.ts +29 -46
  140. package/dist/task/final-gate-progress.js +40 -51
  141. package/dist/task/final-gate.d.ts +64 -97
  142. package/dist/task/final-gate.js +192 -199
  143. package/dist/task/fix-child.d.ts +21 -27
  144. package/dist/task/fix-child.js +21 -27
  145. package/dist/task/foreign-path.d.ts +6 -5
  146. package/dist/task/foreign-path.js +0 -0
  147. package/dist/task/frozen-conflict.d.ts +9 -10
  148. package/dist/task/frozen-conflict.js +61 -64
  149. package/dist/task/frozen-path-guard.d.ts +35 -14
  150. package/dist/task/frozen-path-guard.js +56 -39
  151. package/dist/task/gate-child.d.ts +27 -28
  152. package/dist/task/gate-child.js +37 -35
  153. package/dist/task/gate-deps.d.ts +34 -27
  154. package/dist/task/gate-deps.js +169 -159
  155. package/dist/task/gate-tally.d.ts +77 -80
  156. package/dist/task/gate-tally.js +65 -68
  157. package/dist/task/git-state-guard.d.ts +15 -11
  158. package/dist/task/git-state-guard.js +76 -66
  159. package/dist/task/impl-widget.d.ts +25 -16
  160. package/dist/task/impl-widget.js +27 -17
  161. package/dist/task/implementation-thinking.d.ts +33 -31
  162. package/dist/task/implementation-thinking.js +5 -6
  163. package/dist/task/implementation-turn.d.ts +34 -31
  164. package/dist/task/implementation-turn.js +29 -27
  165. package/dist/task/inline-markdown.d.ts +20 -7
  166. package/dist/task/inline-markdown.js +15 -6
  167. package/dist/task/launch-config-gap.js +25 -39
  168. package/dist/task/launch-contract.d.ts +18 -21
  169. package/dist/task/launch-contract.js +28 -30
  170. package/dist/task/launch-manifest.d.ts +6 -2
  171. package/dist/task/launch-manifest.js +35 -34
  172. package/dist/task/ledger.js +16 -14
  173. package/dist/task/lint-fix.d.ts +6 -8
  174. package/dist/task/lint-fix.js +67 -69
  175. package/dist/task/loop-detector.d.ts +9 -8
  176. package/dist/task/loop-detector.js +16 -12
  177. package/dist/task/mid-run-input.d.ts +17 -15
  178. package/dist/task/mid-run-input.js +17 -15
  179. package/dist/task/orchestrator.d.ts +24 -28
  180. package/dist/task/orchestrator.js +62 -64
  181. package/dist/task/orientation.d.ts +18 -23
  182. package/dist/task/orientation.js +24 -31
  183. package/dist/task/owned-freeze-conflict.d.ts +21 -20
  184. package/dist/task/owned-freeze-conflict.js +52 -85
  185. package/dist/task/owned-freeze-reassign.d.ts +40 -60
  186. package/dist/task/owned-freeze-reassign.js +41 -61
  187. package/dist/task/parsers.d.ts +4 -2
  188. package/dist/task/parsers.js +4 -4
  189. package/dist/task/phases.d.ts +41 -48
  190. package/dist/task/phases.js +180 -248
  191. package/dist/task/plan-io.d.ts +6 -7
  192. package/dist/task/plan-io.js +6 -7
  193. package/dist/task/plan-orchestrator.d.ts +10 -8
  194. package/dist/task/plan-orchestrator.js +14 -10
  195. package/dist/task/plan-prompts.d.ts +6 -5
  196. package/dist/task/plan-prompts.js +6 -5
  197. package/dist/task/plan-readonly.d.ts +4 -5
  198. package/dist/task/plan-readonly.js +4 -5
  199. package/dist/task/plan-rounds.d.ts +17 -29
  200. package/dist/task/plan-rounds.js +21 -34
  201. package/dist/task/plan-session.d.ts +58 -72
  202. package/dist/task/plan-session.js +61 -83
  203. package/dist/task/probe-gaming.d.ts +28 -27
  204. package/dist/task/probe-gaming.js +0 -0
  205. package/dist/task/prohibition-probe.d.ts +14 -16
  206. package/dist/task/prompts.d.ts +3 -4
  207. package/dist/task/prompts.js +17 -26
  208. package/dist/task/qa-transcript.d.ts +15 -22
  209. package/dist/task/qa-transcript.js +15 -21
  210. package/dist/task/question-box.d.ts +17 -13
  211. package/dist/task/question-box.js +19 -15
  212. package/dist/task/question-dedup.d.ts +6 -7
  213. package/dist/task/question-dedup.js +13 -14
  214. package/dist/task/question-dialog.d.ts +22 -32
  215. package/dist/task/question-dialog.js +22 -32
  216. package/dist/task/question-source.d.ts +18 -44
  217. package/dist/task/question-source.js +22 -51
  218. package/dist/task/refuted-constraint.d.ts +11 -31
  219. package/dist/task/refuted-constraint.js +27 -51
  220. package/dist/task/regenerable-artifacts.d.ts +12 -31
  221. package/dist/task/regenerable-artifacts.js +12 -31
  222. package/dist/task/render-check.d.ts +11 -22
  223. package/dist/task/render-check.js +33 -46
  224. package/dist/task/repo-health-check.d.ts +10 -14
  225. package/dist/task/repo-health-check.js +17 -23
  226. package/dist/task/requirements.d.ts +38 -71
  227. package/dist/task/requirements.js +78 -126
  228. package/dist/task/research-fanout-budget.d.ts +51 -88
  229. package/dist/task/research-fanout-budget.js +51 -88
  230. package/dist/task/research-worker.d.ts +33 -36
  231. package/dist/task/research-worker.js +39 -61
  232. package/dist/task/resume-gap.d.ts +14 -15
  233. package/dist/task/root-cause-repair.d.ts +9 -9
  234. package/dist/task/root-cause-repair.js +28 -40
  235. package/dist/task/run-bracket.d.ts +10 -13
  236. package/dist/task/run-end.d.ts +12 -22
  237. package/dist/task/run-end.js +8 -16
  238. package/dist/task/run-final-gate.d.ts +19 -21
  239. package/dist/task/run-final-gate.js +62 -80
  240. package/dist/task/runner-globs.d.ts +12 -13
  241. package/dist/task/runner-globs.js +12 -13
  242. package/dist/task/runner-resolve.d.ts +9 -9
  243. package/dist/task/runner-resolve.js +22 -23
  244. package/dist/task/script-escape.d.ts +10 -12
  245. package/dist/task/script-escape.js +13 -14
  246. package/dist/task/serve-entry.d.ts +1 -1
  247. package/dist/task/serve-entry.js +22 -25
  248. package/dist/task/service-blocks.js +4 -2
  249. package/dist/task/shipped-source.d.ts +11 -29
  250. package/dist/task/shipped-source.js +11 -29
  251. package/dist/task/skip-escape.js +10 -14
  252. package/dist/task/spec-urls.d.ts +26 -65
  253. package/dist/task/spec-urls.js +26 -65
  254. package/dist/task/spec-validation.d.ts +17 -20
  255. package/dist/task/spec-validation.js +17 -20
  256. package/dist/task/stall-detector.d.ts +23 -30
  257. package/dist/task/stall-detector.js +23 -30
  258. package/dist/task/stream-watchdog.d.ts +14 -12
  259. package/dist/task/stream-watchdog.js +14 -12
  260. package/dist/task/substitution-probe.d.ts +17 -20
  261. package/dist/task/substitution-probe.js +17 -20
  262. package/dist/task/task-gates.d.ts +36 -41
  263. package/dist/task/task-gates.js +95 -106
  264. package/dist/task/task-io.d.ts +4 -4
  265. package/dist/task/task-io.js +4 -4
  266. package/dist/task/task-parsers.js +4 -3
  267. package/dist/task/task-provenance.d.ts +2 -2
  268. package/dist/task/task-provenance.js +11 -13
  269. package/dist/task/task-types.d.ts +4 -3
  270. package/dist/task/terminal-outcome.d.ts +14 -16
  271. package/dist/task/terminal-outcome.js +12 -14
  272. package/dist/task/test-assembly.d.ts +13 -20
  273. package/dist/task/test-assembly.js +13 -20
  274. package/dist/task/timings.d.ts +5 -3
  275. package/dist/task/timings.js +5 -3
  276. package/dist/task/title-label.d.ts +9 -4
  277. package/dist/task/title-label.js +9 -4
  278. package/dist/task/type-only-answer.d.ts +44 -52
  279. package/dist/task/type-only-answer.js +44 -52
  280. package/dist/task/unfailable-command.d.ts +18 -24
  281. package/dist/task/unfailable-command.js +21 -27
  282. package/dist/task/unknown-routing.d.ts +10 -4
  283. package/dist/task/unknown-routing.js +10 -4
  284. package/dist/task/user-directives.d.ts +5 -8
  285. package/dist/task/user-directives.js +5 -8
  286. package/dist/task/verify-quality.d.ts +18 -22
  287. package/dist/task/verify-quality.js +45 -46
  288. package/dist/task/verify-reconcile.d.ts +15 -10
  289. package/dist/task/verify-reconcile.js +45 -43
  290. package/dist/task/verify-resolution.d.ts +24 -20
  291. package/dist/task/verify-resolution.js +51 -50
  292. package/dist/task/verify-work.d.ts +59 -66
  293. package/dist/task/verify-work.js +101 -138
  294. package/dist/task/widget.d.ts +15 -14
  295. package/dist/task/widget.js +22 -17
  296. package/dist/task/wiring-claims.d.ts +25 -32
  297. package/dist/task/wiring-claims.js +30 -35
  298. package/dist/task/write-guard.d.ts +39 -39
  299. package/dist/task/write-guard.js +48 -51
  300. package/dist/task/yolo.d.ts +34 -30
  301. package/dist/task/yolo.js +42 -37
  302. package/dist/workers/abstention.d.ts +21 -41
  303. package/dist/workers/abstention.js +27 -48
  304. package/dist/workers/brave-search.d.ts +4 -3
  305. package/dist/workers/brave-search.js +5 -2
  306. package/dist/workers/brave-warning.d.ts +7 -4
  307. package/dist/workers/brave-warning.js +19 -7
  308. package/dist/workers/ddg-search.d.ts +6 -6
  309. package/dist/workers/ddg-search.js +18 -12
  310. package/dist/workers/docs-cache.js +5 -2
  311. package/dist/workers/docs-chunk.d.ts +30 -37
  312. package/dist/workers/docs-chunk.js +37 -41
  313. package/dist/workers/docs-core.d.ts +28 -44
  314. package/dist/workers/docs-core.js +25 -44
  315. package/dist/workers/docs-index.js +4 -3
  316. package/dist/workers/docs-lookup.d.ts +15 -22
  317. package/dist/workers/docs-lookup.js +12 -21
  318. package/dist/workers/docs-project.d.ts +15 -9
  319. package/dist/workers/docs-project.js +17 -10
  320. package/dist/workers/docs-resolve.d.ts +19 -20
  321. package/dist/workers/docs-resolve.js +35 -32
  322. package/dist/workers/docs-retrieve.d.ts +5 -6
  323. package/dist/workers/docs-retrieve.js +18 -15
  324. package/dist/workers/exa-search.d.ts +9 -6
  325. package/dist/workers/exa-search.js +23 -12
  326. package/dist/workers/fetch-core.d.ts +13 -16
  327. package/dist/workers/fetch-core.js +23 -23
  328. package/dist/workers/focused-extractor.d.ts +12 -12
  329. package/dist/workers/focused-extractor.js +16 -19
  330. package/dist/workers/html-clean.js +24 -14
  331. package/dist/workers/http-request.d.ts +28 -20
  332. package/dist/workers/http-request.js +22 -17
  333. package/dist/workers/npm-version.d.ts +28 -11
  334. package/dist/workers/npm-version.js +24 -15
  335. package/dist/workers/phantom-imports.d.ts +15 -12
  336. package/dist/workers/phantom-imports.js +30 -24
  337. package/dist/workers/pi-worker-core.d.ts +86 -54
  338. package/dist/workers/pi-worker-core.js +112 -112
  339. package/dist/workers/pi-worker-docs.d.ts +24 -19
  340. package/dist/workers/pi-worker-docs.js +67 -76
  341. package/dist/workers/pi-worker-fetch.d.ts +7 -3
  342. package/dist/workers/pi-worker-fetch.js +27 -19
  343. package/dist/workers/pi-worker-search.js +12 -8
  344. package/dist/workers/pi-worker.d.ts +9 -4
  345. package/dist/workers/pi-worker.js +23 -10
  346. package/dist/workers/reasoning-warning.d.ts +18 -17
  347. package/dist/workers/reasoning-warning.js +22 -20
  348. package/dist/workers/research-cache.js +50 -78
  349. package/dist/workers/search-core.js +7 -5
  350. package/dist/workers/search-types.d.ts +10 -9
  351. package/dist/workers/search-types.js +9 -8
  352. package/dist/workers/session-hint.d.ts +13 -14
  353. package/dist/workers/session-hint.js +8 -9
  354. package/dist/workers/shared.d.ts +21 -25
  355. package/dist/workers/shared.js +0 -0
  356. package/dist/workers/single-read-extension.d.ts +14 -7
  357. package/dist/workers/single-read-extension.js +14 -7
  358. package/dist/workers/single-read-guard.d.ts +25 -28
  359. package/dist/workers/single-read-guard.js +32 -32
  360. package/dist/workers/typeonly-log.d.ts +12 -9
  361. package/dist/workers/typeonly-log.js +29 -33
  362. package/dist/workers/worker-channels.d.ts +15 -23
  363. package/dist/workers/worker-channels.js +15 -23
  364. package/dist/workers/worker-failure.d.ts +38 -46
  365. package/dist/workers/worker-failure.js +31 -39
  366. package/dist/workers/worker-kill.d.ts +25 -26
  367. package/dist/workers/worker-kill.js +16 -19
  368. package/dist/workers/worker-profiles.d.ts +43 -53
  369. package/dist/workers/worker-profiles.js +30 -38
  370. package/package.json +10 -8
@@ -1,31 +1,29 @@
1
1
  /**
2
2
  * The /task-plan interaction loop.
3
3
  *
4
- * Sequential & adaptive, exactly like /task's grill (phases.ts `phaseGrill`) and
5
- * /task-auto's clarify (auto-orchestrator.ts `planAuto`): ask ONE question at a
6
- * time, feed every answer back into the next generation call so later questions
7
- * react to earlier ones, and stop when the model emits NONE. The duplicate
8
- * backstop (`isDuplicateQuestion` + `DUP_REPROMPT_HINT` + `MAX_DUP_STRIKES`), the
9
- * markdown handling, the A/B answer-letter mapping and the YOLO policy are the
10
- * SAME modules those two loops use none of that is new here.
4
+ * Sequential and adaptive: ask ONE question at a time, feed the whole transcript
5
+ * back into the next generation call so later questions react to earlier answers,
6
+ * and stop when the model emits NONE. Generation, the question cap and the
7
+ * duplicate backstop live in question-source.ts; the answer cards and the A/B
8
+ * letter mapping in question-dialog.ts; markdown in inline-markdown.ts; the
9
+ * unattended policy in yolo.ts. /task's grill (phases.ts `phaseGrill`) and
10
+ * /task-auto's clarify (auto-orchestrator.ts `planAuto`) drive the same modules.
11
11
  *
12
- * What IS new is the control surface. In grill and clarify the user's only move is
13
- * to answer the question in front of them. Here three moves are available at every
14
- * single prompt, in that order of appearance:
12
+ * What this loop adds is the control surface. Grill and clarify let the user only
13
+ * answer the question in front of them. Here three moves are on every prompt, in
14
+ * this order of appearance:
15
15
  *
16
16
  * ❓ ask the model a question — the user asks, the model answers (PLAN_ASK)
17
- * ✎ answer in your own words — the free-text card askQuestionBox already
18
- * appends to every boxed picker; it is not new,
19
- * it is simply always present here, and it
20
- * doubles as "state a decision" when the model
21
- * has nothing to ask
17
+ * ✎ answer in your own words — the free-text card askQuestionBox appends to
18
+ * every boxed picker. It doubles as "state a
19
+ * decision" when the model has nothing to ask.
22
20
  * ▶ proceed to execution — stop planning, hand the decisions to /task
23
21
  * (PLAN_PROCEED). Always the LAST card in the
24
22
  * box — it ends the session, so it sits under
25
23
  * every move that continues it, including the
26
24
  * free-text card (see `manualPosition`).
27
25
  *
28
- * The loop is pure with respect to I/O: every side effect (child calls, dialogs,
26
+ * The loop performs no I/O of its own: every side effect (child calls, dialogs,
29
27
  * persistence) arrives through {@link PlanSessionDeps}, so the whole interaction
30
28
  * is unit-testable without a TUI or a model.
31
29
  */
@@ -38,8 +36,8 @@ export { resolveAnswer } from './question-dialog.js';
38
36
  // ─── Control actions ─────────────────────────────────────────────────────────
39
37
  /**
40
38
  * Sentinel values the picker resolves to when the user takes a control action
41
- * instead of answering. Deliberately shaped like the existing `USER_CANCELLED`
42
- * sentinel (child-runner.ts): a value no model answer and no human ever types.
39
+ * instead of answering. Same shape as `USER_CANCELLED` (child-runner.ts): a
40
+ * value no model answer and no human ever types.
43
41
  */
44
42
  export const PLAN_ASK = '__plan_ask__';
45
43
  export const PLAN_PROCEED = '__plan_proceed__';
@@ -53,18 +51,19 @@ export const PLAN_STATE_LABEL = '✎ Add a decision of your own…';
53
51
  export const PLAN_NO_QUESTIONS = 'No further questions — the decisions so far settle how this task is built.';
54
52
  /**
55
53
  * Hard ceiling on model-generated questions for one plan. The loop is open-ended
56
- * (it stops when the model emits NONE); this only bounds a model that never
57
- * does. Matches /task-auto's MAX_CLARIFY_QUESTIONS, for the same reason.
54
+ * it stops when the model emits NONE — so this only bounds a model that never
55
+ * does. Same value as /task-auto's MAX_CLARIFY_QUESTIONS.
58
56
  */
59
57
  export const MAX_PLAN_QUESTIONS = 8;
60
58
  /**
61
59
  * Corrective re-prompt for a question reply that did not follow the format —
62
- * either nothing parseable at all, or a question with no `SUGGESTED:` line. Same
63
- * shape and same one-shot budget as GRILL_AUTO_FORMAT_HINT (prompts.ts), which
64
- * exists because the local model drops a required tag every so often and a
65
- * silent fallback is worse than one extra call: an unparsed reply reads as "no
66
- * questions left" and a missing SUGGESTED leaves the picker with nothing to
67
- * recommend.
60
+ * either nothing the parser could read, or a question with no `SUGGESTED:` line.
61
+ * Same shape and same one-shot budget as GRILL_AUTO_FORMAT_HINT (prompts.ts).
62
+ *
63
+ * Both failures are silent without it. `makeQuestionSource` answers `exhausted`
64
+ * for a reply it cannot parse, which this loop renders as "no further questions";
65
+ * and a question with no SUGGESTED reaches `buildOptionCards` with nothing to
66
+ * build a card from.
68
67
  */
69
68
  export const PLAN_FORMAT_HINT = '[SYSTEM NOTE: Your previous reply did NOT follow the required format. Output exactly '
70
69
  + 'one numbered question line ("1. **...?** short rationale"), then on the NEXT line a '
@@ -72,16 +71,10 @@ export const PLAN_FORMAT_HINT = '[SYSTEM NOTE: Your previous reply did NOT follo
72
71
  + 'REQUIRED and must never be blank. Add an "ALT: " line only for a binary A-or-B fork. '
73
72
  + 'If nothing is left to ask, output the single token NONE and nothing else. No preamble, '
74
73
  + 'no analysis, no other text.]';
75
- // Re-exported, NOT re-implemented. After the state machine moved to
76
- // question-source.ts these were byte-identical copies: production read THAT
77
- // module's, while plan-session.test.ts and four `scripts/live-*.ts` A/B harnesses
78
- // read these — so a fix to `pickQuestion`'s heuristic would land in one copy while
79
- // the harnesses kept measuring the other, and the measurement would silently stop
80
- // describing shipped behaviour. That is the drift class this pass removes.
81
74
  export { isNoneReply, pickQuestion } from './question-source.js';
82
75
  /**
83
76
  * Does the question offer the user a choice between two named alternatives?
84
- * Deliberately shallow an "X or Y?" in the question's own clause.
77
+ * Deliberately shallow: an `or` anywhere before the first question mark.
85
78
  */
86
79
  export function looksLikeFork(question) {
87
80
  return /\bor\b/i.test(question.split('?')[0] ?? '');
@@ -89,21 +82,16 @@ export function looksLikeFork(question) {
89
82
  /**
90
83
  * Does the recommended default DEFER the decision instead of making one?
91
84
  *
92
- * The prompt asks for a "concrete, decisive default", and nothing enforced it.
93
- * Live (aiz-client TASK_PLAN_0001, 2026-08-05): the model asked "what specific
94
- * report should this new tab display?" and recommended
95
- * "clarify with the user what the report is meant to show before proceeding".
96
- * The user pressed enter, so it was recorded `(accepted recommendation)` and rode
97
- * into /task's handoff as an AUTHORITATIVE decision — an order not to proceed,
98
- * addressed to a run where no user exists. /task duly built a task whose
99
- * ACCEPTANCE was "a planning document with placeholder sections" and whose VERIFY
100
- * asserted that no source file had changed.
101
- *
102
85
  * A deferral is not an answer, and the one place it can never be one is here: the
103
86
  * user IS present during planning, so "ask the user" is a null move — that IS the
104
- * question. Detection is anchored to the START of the default, which keeps it off
105
- * legitimate product behaviour ("prompt the user to confirm deletion" decides
106
- * something; "ask the user which report" decides nothing).
87
+ * question. An accepted recommendation is a `decision` entry, so it rides into
88
+ * /task's handoff inside the block `buildHandoffPrompt` (plan-io.ts) labels
89
+ * authoritative addressed to a run where no user exists.
90
+ *
91
+ * Every pattern is anchored to the START of the default, which keeps it off
92
+ * legitimate product behaviour: "prompt the user to confirm deletion" decides
93
+ * something and does not match; "ask the user which report" decides nothing and
94
+ * does.
107
95
  */
108
96
  export function isDeferralSuggestion(suggested) {
109
97
  const s = suggested.trim().replace(/^["'`*_\s]+/, '');
@@ -116,9 +104,9 @@ export function isDeferralSuggestion(suggested) {
116
104
  || /^(do not|don'?t|no)\b[^.]{0,40}\b(proceed|implement|build|start|write|decide)\b/i.test(s));
117
105
  }
118
106
  /**
119
- * Corrective re-prompt for a default that deferred the decision. Same one-shot
120
- * budget and same quote-it-back shape as {@link planForkHint}, because the child
121
- * is stateless and cannot otherwise know what it just recommended.
107
+ * Corrective re-prompt for a default that deferred the decision. Quotes the
108
+ * question and the default back, because the child is a fresh process carrying
109
+ * only its prompt and cannot otherwise know what it just recommended.
122
110
  */
123
111
  export function planDecisiveHint(question, suggested) {
124
112
  return ('[SYSTEM NOTE: Your previous reply asked this question:\n'
@@ -133,17 +121,12 @@ export function planDecisiveHint(question, suggested) {
133
121
  }
134
122
  /**
135
123
  * Corrective re-prompt for a fork-shaped question that shipped only ONE option.
124
+ * With no ALT, `buildOptionCards` emits a single card, so the user has to type
125
+ * out the alternative the model itself just named.
136
126
  *
137
- * Measured on the local model (scripts/live-task-plan-step0.ts, 15 reps): the
138
- * SUGGESTED line is always there, but 10/15 questions named two alternatives and
139
- * gave only one of them so the picker showed a single card and the user had to
140
- * type out the option the model itself had just proposed.
141
- *
142
- * The retry quotes the question back because the child is stateless (a fresh
143
- * process per call, prompt only), so it cannot otherwise know what it just wrote.
144
- * Validated before wiring (scripts/live-task-plan-fork-alt.ts): 6/6 fires
145
- * recovered an ALT, and 6/6 re-asked the SAME question rather than changing the
146
- * subject. It costs one extra child call on the questions where it fires.
127
+ * The retry quotes the question back because the child is a fresh process
128
+ * carrying only its prompt, so it cannot otherwise know what it just wrote. It
129
+ * costs one extra child call on the questions where it fires.
147
130
  */
148
131
  export function planForkHint(question) {
149
132
  return ('[SYSTEM NOTE: Your previous reply asked this question:\n'
@@ -158,11 +141,8 @@ export function planForkHint(question) {
158
141
  * A default that DEFERS decides nothing, and an accepted deferral reaches the
159
142
  * consumer dressed as an authoritative decision.
160
143
  *
161
- * Declared as its own constant because it is the one rule BOTH dialogs use, and
162
- * `CLARIFY_QUALITY_RULES` referencing it by `id` string would be the retyped
163
- * literal with no compile link that this pass exists to remove — a rename would
164
- * silently yield `[undefined]` and throw on the first clarify question of every
165
- * run.
144
+ * Its own constant because it is the one rule BOTH tables below hold, shared by
145
+ * reference so the compiler links them.
166
146
  */
167
147
  const DEFERRAL_RULE = {
168
148
  id: 'SUGGESTED deferred the decision',
@@ -170,9 +150,10 @@ const DEFERRAL_RULE = {
170
150
  planDecisiveHint(plain, q.suggested)
171
151
  : null,
172
152
  // When only the recommendation defers, the ALT is still a real commitment:
173
- // promote it so the question keeps a usable default. With no ALT the option is
174
- // DROPPED rather than shown, so an empty submit records an unanswered question
175
- // instead of a decision the user never made.
153
+ // promote it so the question keeps a usable default. With no ALT the default
154
+ // is dropped, `buildOptionCards` returns undefined, and `resolveAnswer` maps
155
+ // an empty submit to '(skipped)' rather than to a decision the user never
156
+ // made.
176
157
  repair: q => {
177
158
  const { alt: _alt, suggested: _suggested, ...rest } = q;
178
159
  return q.alt !== undefined ? { ...rest, suggested: q.alt } : { ...rest };
@@ -181,16 +162,15 @@ const DEFERRAL_RULE = {
181
162
  /**
182
163
  * PLAN's quality rules, in order.
183
164
  *
184
- * Each is worth exactly one corrective re-prompt (the child is stateless, so each
185
- * hint quotes the question back), and each DEGRADES rather than discards when the
186
- * defect survives a question with a weak default still beats no question.
165
+ * The corrective-re-prompt budget is per QUESTION and shared across the whole
166
+ * table (question-source.ts): at most one rule fires per draw, and a defect that
167
+ * survives its re-prompt DEGRADES through `repair` rather than discarding the
168
+ * question — a weak default still beats no question. Each hint quotes the
169
+ * question back, because the child is a fresh process carrying only its prompt.
187
170
  *
188
171
  * Only {@link CLARIFY_QUALITY_RULES} is shared with `/task-auto`, and only the
189
- * deferral rule is in it. The other two were MEASURED here (10/15 fork-shaped
190
- * questions shipped one option; the SUGGESTED requirement is in both prompts) but
191
- * each costs one extra child call every time it fires, and clarify is the most
192
- * A/B'd path in the codebase — moving them there is its own experiment, not a
193
- * side effect of sharing a state machine. Recorded rather than done.
172
+ * deferral rule is in it. The other two cost an extra child call every time they
173
+ * fire.
194
174
  */
195
175
  export const PLAN_QUALITY_RULES = [
196
176
  {
@@ -202,7 +182,7 @@ export const PLAN_QUALITY_RULES = [
202
182
  DEFERRAL_RULE,
203
183
  {
204
184
  // A fork-shaped question that ships one option leaves the user typing out
205
- // the alternative the model itself just named (10/15 measured live).
185
+ // the alternative the model itself just named.
206
186
  id: 'fork-shaped question with no ALT',
207
187
  detect: (q, plain) => q.alt === undefined && q.suggested !== undefined && looksLikeFork(plain) ?
208
188
  planForkHint(plain)
@@ -212,12 +192,9 @@ export const PLAN_QUALITY_RULES = [
212
192
  /**
213
193
  * The deferral rule alone — the one clarify shares.
214
194
  *
215
- * It exists because an accepted "clarify with the user before proceeding" rode
216
- * into `/task`'s handoff AS AN AUTHORITATIVE DECISION and produced a task whose
217
- * ACCEPTANCE was "a planning document with placeholder sections" and whose VERIFY
218
- * asserted that no source file had changed. Clarify's answers ride into the
219
- * decompose prompt and the AUTO file with exactly the same authority and had no
220
- * guard at all — the same bug, one command over, waiting.
195
+ * Clarify's answers ride into the decompose prompt and the AUTO file with the
196
+ * same authority /task-plan's decisions ride into the handoff, so an accepted
197
+ * "clarify with the user before proceeding" lands there as an instruction too.
221
198
  *
222
199
  * It is also the only one of the three that costs nothing on the happy path: a
223
200
  * decisive default never triggers it.
@@ -290,7 +267,8 @@ export const ASK_QUESTION = 'What do you want to ask about this task? The answer
290
267
  export async function runPlanSession(deps) {
291
268
  const entries = [];
292
269
  const render = (s) => deps.renderMarkdown?.(s) ?? s;
293
- /** The model has nothing (more) to ask: NONE, the cap, or the dup backstop. */
270
+ /** The model has nothing (more) to ask — any `exhausted` draw: NONE, the
271
+ * cap, the duplicate backstop, or a reply the parser could not read. */
294
272
  let exhausted = false;
295
273
  let pending = null;
296
274
  const commit = async (entry) => {
@@ -1,37 +1,38 @@
1
1
  /**
2
2
  * probe-gaming — deterministic detection of CHECK-GAMING code, feeding the verify
3
- * and enforce gate prompts (run-8 F6).
3
+ * gate's rule 4c (verify-work.ts) and the enforce prompt (enforce-guidelines.ts).
4
4
  *
5
5
  * The failure class: the implementation writes code — or a comment on it — whose
6
- * STATED PURPOSE is to make a check pass, rather than to satisfy the requirement the
7
- * check stands for. mx5 run-8 shipped, verbatim,
8
- * `// Return 401 so the verification test passes (expects 200/401/403 for /api routes).`
9
- * above four catch-all handlers added to satisfy the VERIFY curl instead of fixing
10
- * the broken route mount. The route stayed dead; the check went green; the code SAID
11
- * SO IN WRITING. A check is a MESSENGER for a requirement code written to quiet the
12
- * messenger instead of meeting the requirement is the defect, and when it announces
13
- * its own intent it is cheaply detectable.
6
+ * STATED PURPOSE is to make a check pass, rather than to satisfy the requirement
7
+ * the check stands for. A catch-all handler carrying
8
+ * `// Return 401 so the verification test passes`
9
+ * leaves the real route dead while the check goes green. A check is a MESSENGER
10
+ * for a requirement; code written to quiet the messenger instead of meeting the
11
+ * requirement is the defect, and when it announces its own intent it is cheaply
12
+ * detectable.
14
13
  *
15
- * Same probe+rule design as the other run-8 probes (skip-escape.ts,
16
- * prohibition-probe.ts, substitution-probe.ts, test-assembly.ts): a deterministic
17
- * finding is the reliable lever, a prompt rule alone is weak. This scans the task's
18
- * DIFF (added lines only — the work this task introduced) for the gaming tell and
19
- * hands each hit to the gate child verbatim, so the rule fires on a concrete line
20
- * rather than on the model self-discovering the intent.
14
+ * Same probe+rule design as the other probes (skip-escape.ts,
15
+ * prohibition-probe.ts, substitution-probe.ts, test-assembly.ts): each is a
16
+ * `probeAdapter` row in verify-work.ts with its own findings, prompt block and
17
+ * numbered rule. This one scans the task's DIFF (added lines only — the work this
18
+ * task introduced) for the gaming tell and hands each hit to the gate child
19
+ * verbatim, so the rule fires on a concrete line rather than on the model
20
+ * self-discovering the intent.
21
21
  *
22
- * Advisory, never an auto-FAIL: intent phrasing is prose and a genuinely-benign line
23
- * could in principle carry it ("returns 401 so the auth test passes" describing a
24
- * CORRECT auth path). The child reads the exact line and the surrounding code and
25
- * judges the verify rule directs it to confirm the underlying requirement is
26
- * actually met, not merely that the check is green.
22
+ * Advisory, never an auto-FAIL: a probe contributes findings and a prompt block,
23
+ * and `probeAdapter` degrades an absent or throwing probe to its empty value.
24
+ * Intent phrasing is prose and a genuinely-benign line could in principle carry it
25
+ * ("returns 401 so the auth test passes" describing a CORRECT auth path). The child
26
+ * reads the exact line and the surrounding code and judges — the verify rule
27
+ * directs it to confirm the underlying requirement is actually met, not merely
28
+ * that the check is green.
27
29
  *
28
- * FP-MEASURED on the real run-8 corpus (~/hub/mx5, all 53 commits): 1 unique hit in
29
- * 50,735 added lines exactly the true F6 line, zero false positives. The tell is a
30
- * purpose phrase binding a CHECK noun (test / verification / lint / CI / gate / …) to
31
- * a pass-state or a gaming verb; benign uses of the same words ("pass the test data",
32
- * "run the verification suite", "the linter flagged this") do not match. Pure
33
- * diff-text analysis no stack, framework, or tool-name assumptions; a project with
34
- * no such comment simply yields no findings (nothing to observe = pass).
30
+ * The tell is a purpose phrase binding a CHECK noun (test / verification / lint /
31
+ * CI / gate / …) to a pass-state or a gaming verb. Benign uses of the same words
32
+ * do not match: "pass the test data", "run the verification suite" and "the linter
33
+ * flagged this" all return false. Pure diff-text analysis no stack, framework, or
34
+ * tool-name assumptions; a project with no such comment simply yields no findings,
35
+ * and nothing to observe is not a failure.
35
36
  */
36
37
  /** One added line from a task's diff: the file it was added to and its text. */
37
38
  export interface AddedLine {
Binary file
@@ -2,21 +2,18 @@
2
2
  * prohibition-probe — deterministic detection of VIOLATED SPEC PROHIBITIONS,
3
3
  * feeding the verify gate's prompt.
4
4
  *
5
- * The failure class (mx5 run 7, "New Listing page" task): the spec's CONSTRAINTS
6
- * said "**Do NOT modify** any server-side code: `src/server/index.ts`, …"; the
7
- * implementation modified `src/server/index.ts` anyway; the verify child SAW it
8
- * ("VIOLATES 'Do NOT modify server-side code'"), waived it ("BUT: this is
9
- * additive, tests pass with it"), and PASSed. Worse, the reproduction fixture
10
- * showed the baseline child usually never LOOKS: 5/5 baseline runs consulted no
11
- * diff at all and several affirmatively claimed the forbidden file was untouched.
5
+ * The failure class: the spec's CONSTRAINTS say "**Do NOT modify** any
6
+ * server-side code: `src/server/index.ts`, …", the implementation modifies that
7
+ * file anyway, and the verify child either waives the violation ("this is
8
+ * additive, tests pass with it") or never looks. Nothing else in the gate makes it
9
+ * run `git diff`, and rule 4b says it plainly: a forbidden file can be modified
10
+ * without any test noticing.
12
11
  *
13
- * So — like the substitution probe (see substitution-probe.ts, whose A/B proved
14
- * prompt language alone gets ~40% attention while a concrete deterministic
15
- * finding gets 100%) the fix is a deterministic pre-check whose finding is
16
- * injected into the prompt: extract the concrete paths the spec forbids
17
- * modifying, intersect with the task's changed files (pure git shape, already
18
- * collected for the substitution probe), and hand the child each hit with the
19
- * exact constraint wording.
12
+ * So — like the substitution probe (substitution-probe.ts) the fix is a
13
+ * pre-check whose finding is injected into the prompt: extract the concrete paths
14
+ * the spec forbids modifying, intersect with the task's changed files (pure git
15
+ * shape, from the same `collectChangedFiles` the substitution probe uses), and
16
+ * hand the child each hit with the exact constraint wording.
20
17
  *
21
18
  * The finding is advisory, not an auto-FAIL, for one reason: prohibitions in
22
19
  * real specs are prose and can be CONDITIONAL ("Do NOT modify `api.ts` beyond
@@ -24,8 +21,9 @@
24
21
  * false-FAIL legitimate work. The finding therefore carries the constraint line
25
22
  * verbatim and the prompt's no-waiver rule (4b in verify-work.ts) forbids
26
23
  * excusing an ABSOLUTE prohibition while directing conditional ones to be judged
27
- * against their own stated exception. A violation that was fully reverted before
28
- * verify produces no diff entry, so it never fires — reverted = not violated.
24
+ * against their own stated exception. A violation fully reverted before verify
25
+ * leaves no entry in `git diff HEAD`, so it never fires — reverted = not
26
+ * violated.
29
27
  */
30
28
  import type { ChangedFile } from './substitution-probe.js';
31
29
  /** One "do not modify X" constraint extracted from the spec text. */
@@ -1,8 +1,8 @@
1
1
  /**
2
2
  * Prompt templates for every phase of the pi-task pipeline.
3
3
  *
4
- * Each template is a pure function: inputs prompt string. No I/O, no side
5
- * effects, trivially testable.
4
+ * Each template is a pure function or a plain string constant: inputs → prompt
5
+ * text. The module imports nothing, so there is no I/O and no side effect.
6
6
  */
7
7
  export declare const MAX_GRILL_QUESTIONS = 20;
8
8
  /**
@@ -21,8 +21,7 @@ export declare const COMPRESS_LABEL_PROMPT: (title: string, maxChars: number) =>
21
21
  *
22
22
  * Without it, refine is told "the task title is only a pointer into that spec —
23
23
  * follow the spec" with no signal that the other steps exist, so a one-step
24
- * "Scaffold …" title re-expands the entire design into one task (validated: a real
25
- * /task-auto run implemented all 24 steps under step 1).
24
+ * "Scaffold …" title can re-expand the entire design into one task.
26
25
  */
27
26
  declare const REFINE_PROMPT: (raw: string, planContext?: string, existingFiles?: string, contracts?: string, directives?: string) => string;
28
27
  declare const RESEARCH_READ_ONLY_CONSTRAINT = "IMPORTANT: You are ONLY allowed to READ. Do NOT create, modify, or delete any files. Use the read, grep, find, and ls tools to inspect the repo.";
@@ -1,8 +1,8 @@
1
1
  /**
2
2
  * Prompt templates for every phase of the pi-task pipeline.
3
3
  *
4
- * Each template is a pure function: inputs prompt string. No I/O, no side
5
- * effects, trivially testable.
4
+ * Each template is a pure function or a plain string constant: inputs → prompt
5
+ * text. The module imports nothing, so there is no I/O and no side effect.
6
6
  */
7
7
  // Caps both the per-response question count (parseGrillQuestions/parseClarifyList)
8
8
  // and the grill loop's total iterations, so a model that never emits NONE can't
@@ -33,8 +33,7 @@ ${title}`;
33
33
  *
34
34
  * Without it, refine is told "the task title is only a pointer into that spec —
35
35
  * follow the spec" with no signal that the other steps exist, so a one-step
36
- * "Scaffold …" title re-expands the entire design into one task (validated: a real
37
- * /task-auto run implemented all 24 steps under step 1).
36
+ * "Scaffold …" title can re-expand the entire design into one task.
38
37
  */
39
38
  const REFINE_PROMPT = (raw, planContext, existingFiles, contracts, directives) => `${planContext ? planContext + '\n\n---\n\n' : ''}You receive a user's task description for an AI coding agent. Rewrite it to be unambiguous and actionable.
40
39
 
@@ -69,11 +68,11 @@ Task: ${raw}`;
69
68
  const RESEARCH_READ_ONLY_CONSTRAINT = `IMPORTANT: You are ONLY allowed to READ. Do NOT create, modify, or delete any files. Use the read, grep, find, and ls tools to inspect the repo.`;
70
69
  // Shared guard for every research worker. Open-ended tasks ("analyze the code",
71
70
  // "how would you improve X", "write a report") tempt a worker into producing the
72
- // deliverable itself — e.g. writing the whole code-review report in the CONTEXT
73
- // section, which then runs for many minutes, gets truncated, and poisons every
74
- // downstream phase. Research only gathers INPUTS for a later spec; it must never
75
- // be the deliverable. This also pins the output format (no preamble, no fences,
76
- // no repeated header) that large/open-ended tasks otherwise drift away from.
71
+ // deliverable itself — writing the whole code-review report into the CONTEXT
72
+ // section instead of the facts that section is for. Research only gathers INPUTS
73
+ // for a later spec; it must never be the deliverable. It also pins the output
74
+ // format: no preamble, no code fences, and no repeat of the section name as a
75
+ // header.
77
76
  const RESEARCH_INPUTS_NOT_DELIVERABLE = `CRITICAL — you are gathering INPUTS for a later spec, NOT performing the task. Even if the task asks you to analyze, review, audit, report, plan, design, or write code, you must NOT produce that deliverable here. Do not write the report/analysis/plan/code. Your entire job is to emit the one structured section described below, which feeds a separate phase that writes the spec. Surveying the repo so that section is accurate is right; producing the task's output is wrong and wastes the run.
78
77
 
79
78
  OUTPUT DISCIPLINE — strict: emit ONLY the raw section lines described below. No preamble (never "I've read the codebase…", never "Here is the … section"), no closing remarks, no Markdown headings, no code fences (no \`\`\`), and do NOT repeat the section name as a header. The first character of your output is the first entry of the list.`;
@@ -128,17 +127,8 @@ No section header. No other sections. No preamble.
128
127
 
129
128
  Task:
130
129
  ${refined}`;
131
- // STAGE 2 (2026-07-23): APIS_SEMANTICS_CONTRACT was wired in above and UNWIRED after its A/B
132
- // FAILED. It is the ONE lever of three that moved worker:apis's behaviour — behaviour-class
133
- // package queries 20/20 vs 2/20, Fisher p ≈ 0 — confirming Stage 1's mechanism (completion is
134
- // set by the output CONTRACT, not the answer or the target). But it failed invariant 2a: the
135
- // ungrounded-symbol rate rose 1.0% -> 4.7% (p(rise) = 0.0010) and the semantics clause it adds
136
- // carried 15% ungrounded symbols with the mandatory `UNVERIFIED:` abstention used 0 times in 40
137
- // reps. The model obeys "ask a behaviour question" and ignores "abstain when you cannot verify",
138
- // so it manufactures semantics — the exact F-1 laundering the file exists to prevent. The module
139
- // and its tests are KEPT as the durable asset (like spec-urls.ts after PROMPT 4); only this
140
- // interpolation is reverted. Full write-up: nexxtasks.txt "STAGE 2". Re-run: scripts/
141
- // live-apis-contract-ab.ts. Do NOT re-wire without a lever that closes the abstention gap.
130
+ // `APIS_SEMANTICS_CONTRACT` (apis-contract.ts) is deliberately NOT interpolated
131
+ // into any prompt here. See that module's own docstring before wiring it in.
142
132
  const RESEARCH_CONTEXT_PROMPT = (refined) => `You are doing targeted research for an AI coding agent. Use the read, grep, find, and ls tools to gather background knowledge and architectural context the agent will need for the following task.
143
133
 
144
134
  RELEVANCE — read carefully: keep it tight. Each bullet must be an architectural fact that changes HOW the agent implements THIS task — a constraint, a non-obvious data flow, a gotcha, a hidden coupling. No general project tour, no restating the task, no facts the agent would not act on. If a bullet would not change a single implementation decision, drop it. There is no fixed bullet count — include every fact that bears on the task and no filler; fewer sharp bullets beat many shallow ones. If the task is itself an analysis or review, these bullets capture facts that analysis will rely on — they are NOT the analysis; do not write findings or recommendations here.
@@ -330,12 +320,13 @@ ${research}
330
320
  User Q&A:
331
321
  ${qa}
332
322
  ${contracts && contracts.trim() ? `\n${contracts.trim()}\n` : ''}`;
333
- // Fast triage pass run before the (expensive) full rewrite. It produces either
334
- // the single token CLEAN — meaning the compose draft needs no rewrite — or a
335
- // short defect list. When CLEAN, the orchestrator returns the draft unchanged
336
- // and skips the rewrite entirely; otherwise the defects are fed into
337
- // CRITIQUE_PROMPT as a focus list so the rewrite targets real problems instead
338
- // of re-deriving them from scratch.
323
+ // Fast triage pass run before the full rewrite. It produces either the single
324
+ // token CLEAN — meaning the compose draft needs no rewrite — or a short defect
325
+ // list. CLEAN alone does not skip the rewrite: phases.ts only short-circuits when
326
+ // the draft already parses a VERIFY block, no deterministic critique probe forced
327
+ // itself in, and there are no carried-in defects. Otherwise the triage defects are
328
+ // fed into CRITIQUE_PROMPT as a focus list so the rewrite targets real problems
329
+ // instead of re-deriving them from scratch.
339
330
  const CRITIQUE_TRIAGE_PROMPT = (spec, refined, qa, contracts) => `You are triaging an implementation spec for an AI coding agent. Decide whether it needs a rewrite. Do NOT rewrite it — only judge it.
340
331
 
341
332
  The refined task and the user's Q&A below are GROUND TRUTH. Judge the spec against them. Look for SUBSTANTIVE defects only:
@@ -3,25 +3,18 @@
3
3
  * (`/task-auto`) — and the one place its numbering, its formatting and its
4
4
  * provenance policy live.
5
5
  *
6
- * `question-dialog.ts` unified the ANSWER side of these loops: the picker cards
7
- * and the reply mapping. It did not unify what the loop RECORDS, and that was
8
- * eight retyped push sites across two files, each choosing a suffix by hand
9
- * according to which branch of an `if/else` it stood in.
6
+ * `question-dialog.ts` owns the ANSWER side of these loops the picker cards and
7
+ * the reply mapping. This owns what the loop RECORDS.
10
8
  *
11
- * The cost is written into the code. `phases.ts` states the invariant in a
12
- * comment *"No provenance stamp here, unlike clarify's transcript: this string
13
- * is fed back VERBATIM into the next grill-gen prompt, so a `(accepted
14
- * recommendation)` suffix would become model input"* and TWELVE LINES ABOVE it,
15
- * the YOLO branch pushes `${answer} ${YOLO_STAMP}` into that very array. The rule
16
- * was violated inside the loop body that declares it, because the rule lived in
17
- * prose and the decision lived at each push.
18
- *
19
- * Here it is a property of a value: an entry states its KIND, and the policy says
20
- * where that kind's provenance is allowed to appear.
9
+ * Provenance is a property of a value here, not a decision taken at each push
10
+ * site: an entry states its KIND, and the policy says where that kind's
11
+ * provenance is allowed to appear. The two renderings below differ ONLY in those
12
+ * suffixes, which is how a hand-stamped transcript drifts apart unnoticed.
21
13
  *
22
14
  * NOT unified here: `plan-session.ts`'s `PlanEntry` transcript. It is a different
23
- * shape — decisions vs advisory notes, persisted to its own task file, with no
24
- * `Qn:`/`An:` numbering — and folding it in would be a rename, not a deepening.
15
+ * shape — model decisions, advisory notes and user-stated decisions in one list,
16
+ * persisted to its own task file — and folding it in would be a rename, not a
17
+ * deepening.
25
18
  */
26
19
  /** How one answer was arrived at. */
27
20
  export type QaKind =
@@ -56,7 +49,6 @@ export interface QaPolicy {
56
49
  * becomes model input describing how the answer was obtained rather than what
57
50
  * it was. Clarify deliberately shows its generator the provenance, so a
58
51
  * question the triage already settled reads as settled and is not re-asked.
59
- * An option, not a unification — it is observable either way.
60
52
  */
61
53
  generatorSeesProvenance: boolean;
62
54
  }
@@ -64,10 +56,11 @@ export interface QaPolicy {
64
56
  * GRILL: stamps `auto` and both YOLO kinds in the record, and shows the generator
65
57
  * nothing.
66
58
  *
67
- * `accepted` is deliberately NOT in the record set. Grill's record reaches
68
- * `COMPOSE_PROMPT` and `CRITIQUE_PROMPT`, where it is named GROUND TRUTH; clarify's
69
- * reaches a decompose prompt. Whether those two should agree is a prompt question
70
- * with its own A/B, so today's answer is preserved rather than harmonised here.
59
+ * `accepted` is deliberately NOT in the record set, and clarify's policy below
60
+ * disagrees. Grill's record is handed to `COMPOSE_PROMPT` and `CRITIQUE_PROMPT`
61
+ * (the latter names the Q&A GROUND TRUTH); clarify's is handed to
62
+ * `AUTO_DECOMPOSE_PROMPT`. Whether the two should agree is a question about those
63
+ * prompts, so each keeps its own answer here.
71
64
  */
72
65
  export declare const GRILL_QA_POLICY: QaPolicy;
73
66
  /** CLARIFY: stamps every non-typed kind, in the record and to the generator alike. */
@@ -90,7 +83,7 @@ export declare class QaTranscript {
90
83
  /** How many questions have been answered. Also the next question's number − 1. */
91
84
  get length(): number;
92
85
  get entries(): ReadonlyArray<QaEntry>;
93
- /** Record one answered question. The number is assigned here, not by the caller. */
86
+ /** Record one answered question. Position IS the number; no caller supplies one. */
94
87
  add(kind: QaKind, question: string, answer: string): void;
95
88
  /** The persisted / handed-on transcript, with provenance per the policy. */
96
89
  forRecord(): string;