oh-my-opencode 5.0.0-beta.1 → 5.0.0-beta.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (196) hide show
  1. package/.agents/command/publish.md +44 -16
  2. package/.agents/skills/publish/SKILL.md +44 -16
  3. package/.agents/skills/work-with-pr/SKILL.md +37 -23
  4. package/.opencode/command/publish.md +44 -16
  5. package/.opencode/skills/work-with-pr/SKILL.md +37 -23
  6. package/README.md +12 -1
  7. package/dist/agents/atlas/agent.d.ts +0 -1
  8. package/dist/agents/sisyphus/grok-4.d.ts +20 -0
  9. package/dist/agents/sisyphus/index.d.ts +2 -0
  10. package/dist/agents/sisyphus-agent-config.d.ts +6 -0
  11. package/dist/agents/sisyphus-agent-factory.d.ts +1 -1
  12. package/dist/agents/sisyphus-runtime-prompt-reconciler.d.ts +15 -4
  13. package/dist/agents/types.d.ts +2 -2
  14. package/dist/cli/index.js +797 -465
  15. package/dist/cli/run/on-complete-hook.d.ts +2 -0
  16. package/dist/cli-node/index.js +797 -465
  17. package/dist/hooks/atlas/final-wave-approval-gate.test-support.d.ts +50 -0
  18. package/dist/hooks/atlas/system-reminder-templates.d.ts +0 -1
  19. package/dist/index.js +1511 -1140
  20. package/dist/shared/normalize-sdk-response.d.ts +1 -0
  21. package/dist/shared/shell-env.d.ts +1 -1
  22. package/dist/skills/coding-agent-sessions/SKILL.md +3 -2
  23. package/dist/skills/coding-agent-sessions/references/all-platforms.md +1 -1
  24. package/dist/skills/coding-agent-sessions/references/senpi.md +4 -4
  25. package/dist/skills/coding-agent-sessions/scripts/agent_sessions/pi_family.py +1 -1
  26. package/dist/skills/frontend/SKILL.md +10 -7
  27. package/dist/skills/frontend/references/design/_INDEX.md +1 -0
  28. package/dist/skills/frontend/references/design/stylegallery.md +80 -0
  29. package/dist/skills/ultimate-browsing/ATTRIBUTION.md +37 -10
  30. package/dist/skills/ultimate-browsing/engine/AGENTS.md +179 -0
  31. package/dist/skills/ultimate-browsing/engine/templates/package.json +1 -1
  32. package/dist/skills/ultimate-browsing/references/chrome-stealth.md +11 -11
  33. package/dist/skills/ulw-plan/SKILL.md +2 -2
  34. package/dist/skills/ulw-plan/references/full-workflow.md +27 -3
  35. package/dist/skills/ulw-plan/references/intent-clear.md +2 -1
  36. package/dist/skills/ulw-plan/references/intent-unclear.md +3 -3
  37. package/dist/tui.js +176 -14
  38. package/package.json +21 -19
  39. package/packages/lsp-core/src/lsp/client-diagnostics-concurrency.integration.test.ts +44 -0
  40. package/packages/lsp-core/src/lsp/client-diagnostics-freshness.integration.test.ts +0 -28
  41. package/packages/lsp-core/src/lsp/client-wrapper.test.ts +60 -7
  42. package/packages/lsp-core/src/lsp/client-wrapper.ts +69 -16
  43. package/packages/lsp-core/src/lsp/workspace-edit-adversarial.test.ts +20 -1
  44. package/packages/lsp-core/src/tools/diagnostics.ts +3 -3
  45. package/packages/lsp-core/src/tools/navigation.ts +4 -2
  46. package/packages/lsp-core/src/tools/rename.ts +4 -2
  47. package/packages/lsp-core/src/tools/symbols.ts +1 -1
  48. package/packages/lsp-daemon/dist/cli.js +83 -29
  49. package/packages/lsp-daemon/dist/client.js +76 -22
  50. package/packages/lsp-daemon/dist/index.js +81 -27
  51. package/packages/lsp-tools-mcp/dist/cli.js +76 -22
  52. package/packages/lsp-tools-mcp/dist/mcp.js +76 -22
  53. package/packages/lsp-tools-mcp/dist/tools.js +76 -22
  54. package/packages/omo-codex/plugin/.codex-plugin/plugin.json +1 -1
  55. package/packages/omo-codex/plugin/components/bootstrap/hooks/hooks.json +1 -1
  56. package/packages/omo-codex/plugin/components/bootstrap/package.json +1 -1
  57. package/packages/omo-codex/plugin/components/codegraph/dist/cli.js +118 -8
  58. package/packages/omo-codex/plugin/components/codegraph/dist/serve.js +118 -8
  59. package/packages/omo-codex/plugin/components/codegraph/package.json +1 -1
  60. package/packages/omo-codex/plugin/components/comment-checker/hooks/hooks.json +1 -1
  61. package/packages/omo-codex/plugin/components/comment-checker/package.json +1 -1
  62. package/packages/omo-codex/plugin/components/git-bash/hooks/hooks.json +2 -2
  63. package/packages/omo-codex/plugin/components/git-bash/package.json +1 -1
  64. package/packages/omo-codex/plugin/components/lazycodex-executor-verify/hooks/hooks.json +1 -1
  65. package/packages/omo-codex/plugin/components/lazycodex-executor-verify/package.json +1 -1
  66. package/packages/omo-codex/plugin/components/lazycodex-executor-verify/test/codex-hook.test.ts +3 -17
  67. package/packages/omo-codex/plugin/components/lsp/dist/.omo-runtime-manifest.json +3 -3
  68. package/packages/omo-codex/plugin/components/lsp/dist/cli.js +83 -29
  69. package/packages/omo-codex/plugin/components/lsp/hooks/hooks.json +2 -2
  70. package/packages/omo-codex/plugin/components/lsp/package.json +1 -1
  71. package/packages/omo-codex/plugin/components/rules/hooks/hooks.json +4 -4
  72. package/packages/omo-codex/plugin/components/rules/package.json +1 -1
  73. package/packages/omo-codex/plugin/components/rules/test/bundled-rules-priority.test.ts +11 -16
  74. package/packages/omo-codex/plugin/components/rules/test/bundled-rules.test.ts +16 -23
  75. package/packages/omo-codex/plugin/components/rules/test/codex-hook-post-compact-budget.test.ts +9 -7
  76. package/packages/omo-codex/plugin/components/rules/test/codex-hook-post-compact-context.test.ts +0 -6
  77. package/packages/omo-codex/plugin/components/rules/test/codex-hook-post-compact-dedup.test.ts +6 -4
  78. package/packages/omo-codex/plugin/components/rules/test/codex-hook-post-compact-directive.test.ts +12 -9
  79. package/packages/omo-codex/plugin/components/rules/test/codex-hook.test.ts +28 -37
  80. package/packages/omo-codex/plugin/components/rules/test/formatter.test.ts +37 -69
  81. package/packages/omo-codex/plugin/components/rules/test/hook-output.test.ts +2 -3
  82. package/packages/omo-codex/plugin/components/rules/test/windows-git-bash-bundled-rule.test.ts +1 -15
  83. package/packages/omo-codex/plugin/components/start-work-continuation/AGENTS.md +3 -2
  84. package/packages/omo-codex/plugin/components/start-work-continuation/hooks/hooks.json +2 -2
  85. package/packages/omo-codex/plugin/components/start-work-continuation/package.json +1 -1
  86. package/packages/omo-codex/plugin/components/start-work-continuation/test/cli.test.ts +0 -3
  87. package/packages/omo-codex/plugin/components/start-work-continuation/test/codex-hook.test.ts +2 -16
  88. package/packages/omo-codex/plugin/components/teammode/hooks/hooks.json +1 -1
  89. package/packages/omo-codex/plugin/components/teammode/package.json +1 -1
  90. package/packages/omo-codex/plugin/components/teammode/test/thread-title-hook.test.ts +3 -9
  91. package/packages/omo-codex/plugin/components/telemetry/dist/cli.js +24 -12
  92. package/packages/omo-codex/plugin/components/telemetry/dist/posthog.js +24 -12
  93. package/packages/omo-codex/plugin/components/telemetry/hooks/hooks.json +1 -1
  94. package/packages/omo-codex/plugin/components/telemetry/package.json +1 -1
  95. package/packages/omo-codex/plugin/components/ultrawork/directive.md +6 -0
  96. package/packages/omo-codex/plugin/components/ultrawork/hooks/hooks.json +1 -1
  97. package/packages/omo-codex/plugin/components/ultrawork/package.json +1 -1
  98. package/packages/omo-codex/plugin/components/ultrawork/skills/ultrawork/SKILL.md +6 -0
  99. package/packages/omo-codex/plugin/components/ultrawork/skills/ulw-plan/SKILL.md +2 -2
  100. package/packages/omo-codex/plugin/components/ultrawork/skills/ulw-plan/references/full-workflow.md +27 -3
  101. package/packages/omo-codex/plugin/components/ultrawork/skills/ulw-plan/references/intent-clear.md +2 -1
  102. package/packages/omo-codex/plugin/components/ultrawork/skills/ulw-plan/references/intent-unclear.md +3 -3
  103. package/packages/omo-codex/plugin/components/ultrawork/test/codex-hook.test.ts +0 -136
  104. package/packages/omo-codex/plugin/components/ultrawork/test/skill-pointer.test.ts +0 -2
  105. package/packages/omo-codex/plugin/components/ulw-loop/directive.md +6 -0
  106. package/packages/omo-codex/plugin/components/ulw-loop/hooks/hooks.json +4 -4
  107. package/packages/omo-codex/plugin/components/ulw-loop/package.json +1 -1
  108. package/packages/omo-codex/plugin/components/ulw-loop/skills/ulw-loop/SKILL.md +3 -2
  109. package/packages/omo-codex/plugin/components/ulw-loop/skills/ulw-loop/references/define-goal.md +108 -0
  110. package/packages/omo-codex/plugin/components/ulw-loop/skills/ulw-loop/references/full-workflow.md +1 -0
  111. package/packages/omo-codex/plugin/components/ulw-loop/test/checkpoint-continuation.test.ts +0 -1
  112. package/packages/omo-codex/plugin/components/ulw-loop/test/codex-goal-instruction.test.ts +2 -2
  113. package/packages/omo-codex/plugin/components/ulw-loop/test/codex-hook.test.ts +0 -3
  114. package/packages/omo-codex/plugin/components/ulw-loop/test/package-smoke.test.ts +2 -35
  115. package/packages/omo-codex/plugin/components/ulw-loop/test/ultrawork-directive.test.ts +4 -5
  116. package/packages/omo-codex/plugin/hooks/post-compact-resetting-git-bash-mcp-reminder.json +1 -1
  117. package/packages/omo-codex/plugin/hooks/post-compact-resetting-lsp-diagnostics-cache.json +1 -1
  118. package/packages/omo-codex/plugin/hooks/post-compact-resetting-project-rule-cache.json +1 -1
  119. package/packages/omo-codex/plugin/hooks/post-tool-use-checking-codegraph-init-guidance.json +1 -1
  120. package/packages/omo-codex/plugin/hooks/post-tool-use-checking-comments.json +1 -1
  121. package/packages/omo-codex/plugin/hooks/post-tool-use-checking-lsp-diagnostics.json +1 -1
  122. package/packages/omo-codex/plugin/hooks/post-tool-use-checking-thread-title-hygiene.json +1 -1
  123. package/packages/omo-codex/plugin/hooks/post-tool-use-matching-project-rules.json +1 -1
  124. package/packages/omo-codex/plugin/hooks/pre-tool-use-enforcing-unlimited-goal-budget.json +1 -1
  125. package/packages/omo-codex/plugin/hooks/pre-tool-use-guarding-ulw-loop-spawns.json +1 -1
  126. package/packages/omo-codex/plugin/hooks/pre-tool-use-recommending-git-bash-mcp.json +1 -1
  127. package/packages/omo-codex/plugin/hooks/session-start-checking-auto-update.json +1 -1
  128. package/packages/omo-codex/plugin/hooks/session-start-checking-bootstrap-provisioning.json +1 -1
  129. package/packages/omo-codex/plugin/hooks/session-start-checking-codegraph-bootstrap.json +1 -1
  130. package/packages/omo-codex/plugin/hooks/session-start-loading-project-rules.json +1 -1
  131. package/packages/omo-codex/plugin/hooks/session-start-recording-session-telemetry.json +1 -1
  132. package/packages/omo-codex/plugin/hooks/stop-checking-start-work-continuation.json +1 -1
  133. package/packages/omo-codex/plugin/hooks/stop-checking-ulw-loop-resume.json +1 -1
  134. package/packages/omo-codex/plugin/hooks/subagent-stop-checking-start-work-continuation.json +1 -1
  135. package/packages/omo-codex/plugin/hooks/subagent-stop-verifying-lazycodex-executor-evidence.json +1 -1
  136. package/packages/omo-codex/plugin/hooks/user-prompt-submit-checking-ultrawork-trigger.json +1 -1
  137. package/packages/omo-codex/plugin/hooks/user-prompt-submit-checking-ulw-loop-steering.json +1 -1
  138. package/packages/omo-codex/plugin/hooks/user-prompt-submit-loading-project-rules.json +1 -1
  139. package/packages/omo-codex/plugin/package-lock.json +20 -20
  140. package/packages/omo-codex/plugin/package.json +1 -1
  141. package/packages/omo-codex/plugin/skills/coding-agent-sessions/SKILL.md +3 -2
  142. package/packages/omo-codex/plugin/skills/coding-agent-sessions/references/all-platforms.md +1 -1
  143. package/packages/omo-codex/plugin/skills/coding-agent-sessions/references/senpi.md +4 -4
  144. package/packages/omo-codex/plugin/skills/coding-agent-sessions/scripts/agent_sessions/pi_family.py +1 -1
  145. package/packages/omo-codex/plugin/skills/frontend/SKILL.md +10 -7
  146. package/packages/omo-codex/plugin/skills/frontend/references/design/_INDEX.md +1 -0
  147. package/packages/omo-codex/plugin/skills/frontend/references/design/stylegallery.md +80 -0
  148. package/packages/omo-codex/plugin/skills/ultimate-browsing/ATTRIBUTION.md +37 -10
  149. package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/AGENTS.md +179 -0
  150. package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/templates/package.json +1 -1
  151. package/packages/omo-codex/plugin/skills/ultimate-browsing/references/chrome-stealth.md +11 -11
  152. package/packages/omo-codex/plugin/skills/ultrawork/SKILL.md +6 -0
  153. package/packages/omo-codex/plugin/skills/ulw-loop/SKILL.md +3 -2
  154. package/packages/omo-codex/plugin/skills/ulw-loop/references/define-goal.md +108 -0
  155. package/packages/omo-codex/plugin/skills/ulw-loop/references/full-workflow.md +1 -0
  156. package/packages/omo-codex/plugin/skills/ulw-plan/SKILL.md +2 -2
  157. package/packages/omo-codex/plugin/skills/ulw-plan/references/full-workflow.md +27 -3
  158. package/packages/omo-codex/plugin/skills/ulw-plan/references/intent-clear.md +2 -1
  159. package/packages/omo-codex/plugin/skills/ulw-plan/references/intent-unclear.md +3 -3
  160. package/packages/omo-codex/plugin/test/aggregate-agents.test.mjs +19 -173
  161. package/packages/omo-codex/plugin/test/aggregate-hooks.test.mjs +4 -24
  162. package/packages/omo-codex/plugin/test/aggregate-plugin-fixture.mjs +175 -13
  163. package/packages/omo-codex/plugin/test/aggregate.test.mjs +78 -2
  164. package/packages/omo-codex/plugin/test/auto-update-release-notes.test.mjs +19 -33
  165. package/packages/omo-codex/plugin/test/lcx-contribute-bug-fix-template.test.mjs +21 -27
  166. package/packages/omo-codex/plugin/test/scaffold-plan.test.mjs +0 -36
  167. package/packages/omo-codex/plugin/test/sync-skills-codex-compatibility.test.mjs +101 -0
  168. package/packages/omo-codex/plugin/test/sync-skills.test.mjs +1 -119
  169. package/packages/omo-codex/plugin/test/teammode-archive-ambiguity.test.mjs +0 -40
  170. package/packages/omo-codex/plugin/test/teammode-communication.test.mjs +6 -62
  171. package/packages/omo-codex/plugin/test/teammode-thread-links.test.mjs +3 -36
  172. package/packages/omo-codex/plugin/test/teammode-transport.test.mjs +0 -44
  173. package/packages/omo-codex/plugin/test/teammode-worktree.test.mjs +2 -6
  174. package/packages/omo-codex/plugin/test/ultrawork-skill-pointer.test.mjs +0 -3
  175. package/packages/omo-codex/plugin/test/ulw-plan-review-state-contract.test.mjs +0 -3
  176. package/packages/omo-codex/scripts/install-dist/install-local.mjs +57 -19
  177. package/packages/omo-codex/scripts/install-lazycodex-version-stamp.test.mjs +7 -2
  178. package/packages/shared-skills/skills/coding-agent-sessions/SKILL.md +3 -2
  179. package/packages/shared-skills/skills/coding-agent-sessions/references/all-platforms.md +1 -1
  180. package/packages/shared-skills/skills/coding-agent-sessions/references/senpi.md +4 -4
  181. package/packages/shared-skills/skills/coding-agent-sessions/scripts/agent_sessions/pi_family.py +1 -1
  182. package/packages/shared-skills/skills/frontend/SKILL.md +10 -7
  183. package/packages/shared-skills/skills/frontend/references/design/_INDEX.md +1 -0
  184. package/packages/shared-skills/skills/frontend/references/design/stylegallery.md +80 -0
  185. package/packages/shared-skills/skills/ultimate-browsing/ATTRIBUTION.md +37 -10
  186. package/packages/shared-skills/skills/ultimate-browsing/engine/AGENTS.md +179 -0
  187. package/packages/shared-skills/skills/ultimate-browsing/engine/templates/package.json +1 -1
  188. package/packages/shared-skills/skills/ultimate-browsing/references/chrome-stealth.md +11 -11
  189. package/packages/shared-skills/skills/ulw-plan/SKILL.md +2 -2
  190. package/packages/shared-skills/skills/ulw-plan/references/full-workflow.md +27 -3
  191. package/packages/shared-skills/skills/ulw-plan/references/intent-clear.md +2 -1
  192. package/packages/shared-skills/skills/ulw-plan/references/intent-unclear.md +3 -3
  193. package/dist/tools/call-omo-agent/background-agent-executor.d.ts +0 -5
  194. package/packages/omo-codex/plugin/test/aggregate-skills.test.mjs +0 -92
  195. package/packages/omo-codex/plugin/test/sync-skills-orchestration.test.mjs +0 -314
  196. package/packages/omo-codex/plugin/test/ulw-plan-scope-contract.test.mjs +0 -24
@@ -136,7 +136,7 @@ No Metis, no plan file, no execution until the user approves. The UNCLEAR path a
136
136
 
137
137
  ## Phase 3 - Generate the plan (only after approval)
138
138
  1. Rerun `node "<skill-root>/scripts/scaffold-plan.mjs" <slug> [--clear|--unclear]` without `--draft-only`. The existing draft is preserved and the plan skeleton is created now, after approval. A plain rerun is a safe no-op; never hand-build the skeleton.
139
- 2. **Metis gap analysis (mandatory):** spawn a metis reviewer for contradictions, missing constraints, scope-creep, unvalidated assumptions, and missing acceptance criteria; fold findings in silently.
139
+ 2. **Metis gap analysis (mandatory):** spawn a metis reviewer for contradictions, missing constraints — including unstated extrinsic ones: budget/spend, mandated stack, expected scale, target audience / compliance — scope-creep, unvalidated assumptions, and missing acceptance criteria; fold findings in silently; require each constraint gap to return as a proposed default plus reversibility, or a single owner-question when defaulting is unsafe.
140
140
  3. APPEND todo batches into the `## Todos` region with edit/apply_patch - never rewrite the script-emitted headers; 50+ todos is fine; one request -> one plan.
141
141
  4. Fill `## TL;DR (For humans)` LAST, after the detailed plan, so it summarizes the real plan, not an intention.
142
142
  5. Self-review: every todo has references + agent-executable acceptance criteria + happy+failure QA scenarios; no business-logic assumption without evidence; zero criteria need a human. HR6 backstop - confirm the plan's FIRST `## ` heading is `## TL;DR (For humans)` and that every header below it appears in the template order; if you ever hand-built or reordered the file, the human summary must still lead.
@@ -180,7 +180,7 @@ Every "present the plan summary/brief" above delivers THIS structure, in the use
180
180
  6. **Execution handoff** - the plan runs in a worker session via `$start-work <plan-name>`; introduce the options: `--worktree <absolute-path>` (task-owned worktree; required for PR/branch work), `--make-pr` (deliver as a PR; auto-creates a task-owned worktree), `--ship` (implies `--make-pr`, keeps working until the PR is reviewed and MERGED).
181
181
 
182
182
  ### High-accuracy review (dual review)
183
- The high-accuracy review is DUAL and both passes must return OKAY before handoff: (1) the native `momus` reviewer subagent, and (2) an independent Codex CLI review on gpt-5.6-sol at xhigh reasoning, run in a disposable isolated workspace and `CODEX_HOME` with the harness's normal approval and sandbox policy. Do not add flags that disable approvals or sandboxing. Momus runs at High and may take substantially longer than other agents. One round = exactly ONE `momus` + ONE independent review, dispatched together against the COMPLETE plan file (todos + TL;DR filled) at the draft's exact recorded `plan_path`. Keep Momus in flight and wait for its terminal result: elapsed time alone never justifies cancelling, duplicating, replacing, or treating it as failed. After both verdicts return, fix every cited issue and resubmit both fresh until each approves. CLEAR: runs when the user opts in or `review_required: true`. UNCLEAR: runs automatically unless Classify=Trivial.
183
+ The high-accuracy review is DUAL and both passes must return OKAY before handoff: (1) the native `momus` reviewer subagent, and (2) an independent Codex CLI review on gpt-5.6-sol at xhigh reasoning, run in a disposable isolated workspace and `CODEX_HOME` with the harness's normal approval and sandbox policy. Do not add flags that disable approvals or sandboxing. Momus runs at High and may take substantially longer than other agents. One round = exactly ONE `momus` + ONE independent review, dispatched together against the COMPLETE plan file (todos + TL;DR filled) at the draft's exact recorded `plan_path`. Keep Momus in flight and wait for its terminal result: elapsed time alone never justifies cancelling, duplicating, replacing, or treating it as failed. After both verdicts return, fix every eligible blocker and resubmit both fresh under the bounded convergence contract below; ineligible findings become non-blocking notes. CLEAR: runs when the user opts in or `review_required: true`. UNCLEAR: runs automatically unless Classify=Trivial.
184
184
 
185
185
  Every reviewer prompt must carry this intake contract with all angle-bracket values replaced by literals from the current round before dispatch. Never pass `draft.plan_path`, `draft.plan_sha256`, field names, or another symbolic reference to an isolated reviewer. For the independent Codex lane, materialize the complete plan at that same literal workspace-relative path inside the disposable review workspace, verify the copied file's SHA-256, then dispatch with that disposable workspace's literal canonical root. Its first action is to read the exact recorded path; retrieval drift stops that lane before review:
186
186
 
@@ -209,7 +209,31 @@ Every reviewer prompt must carry this intake contract with all angle-bracket val
209
209
 
210
210
  The first action must open the literal workspace root as a directory descriptor, then traverse `.omo`, `plans`, and the final target with descriptor-relative no-follow opens, `fstat` each ancestor as a directory and the final descriptor as a regular file, and hash all bytes read from that same final descriptor. If the platform cannot guarantee this chain, or any path/runtime/launch/receipt/digest check drifts, return `INCONCLUSIVE` before reviewing. Echo the literal workspace, runtime home, target, digest, round, and launch ID; the parent separately matches the completion envelope to the persisted session/process receipt. Never search or use another artifact.
211
211
 
212
- The draft must record the native Momus session/result, the independent Codex CLI review command/result, and the fix/retry summary. Immediately before handoff, repeat the same live canonical-path and SHA-256 validation and require it to match the approved round digest; drift invalidates both approvals and starts a fresh round. Do not say "high-accuracy review completed" unless both receipts exist, both final verdicts are unconditional approval, and the final live-plan validation passes.
212
+ ### Bounded convergence (the review must terminate)
213
+ Review rounds are capped at 5 (unlimited only on explicit user request), and an approval whose only remaining items are notes counts as approval. A finding may BLOCK only when it names at least one `blocker_eligibility` category below with its concrete evidence; every other finding - speculative durability, replay/crash-recovery, schema, CLI-parsing, state-machine, or hardening concerns the accepted scope never required - is recorded as a non-blocking note and becomes implementation/test work, never plan expansion. After round 1 the blocker ledger FREEZES: later rounds verify accepted ledger blockers, regressions introduced by fixes, and new findings that pass eligibility - they never rediscover the plan from scratch. Fixes apply the smallest edit that resolves the cited blocker; neither reviews nor fixes grow the plan's scope. Every reviewer prompt carries this convergence contract alongside the intake contract. On cap exhaustion without approval: STOP, report outstanding blockers, ask the user - continue / accept / adjust.
214
+
215
+ <!-- ulw-plan-review-convergence-contract -->
216
+ ```json
217
+ {
218
+ "max_rounds": 5,
219
+ "max_rounds_override": "explicit_user_request_only",
220
+ "on_cap_reached": "stop_report_outstanding_blockers_ask_user",
221
+ "blocker_eligibility": [
222
+ "explicit_requirement_or_accepted_decision",
223
+ "existing_failing_regression",
224
+ "reproducible_broken_flow",
225
+ "concrete_security_data_loss_or_compatibility_risk",
226
+ "external_api_provider_or_release_contract_conflict"
227
+ ],
228
+ "ineligible_finding_disposition": "non_blocking_note",
229
+ "approval_with_notes_counts_as_approval": true,
230
+ "ledger_freeze_after_round": 1,
231
+ "closure_round_scope": ["accepted_ledger_blockers", "regressions_introduced_by_fixes", "new_findings_passing_blocker_eligibility"],
232
+ "fix_edit_policy": "smallest_edit_no_scope_expansion"
233
+ }
234
+ ```
235
+
236
+ The draft must record the native Momus session/result, the independent Codex CLI review command/result, and the fix/retry summary, plus the convergence ledger (accepted blockers, non-blocking notes, round count). Immediately before handoff, repeat the same live canonical-path and SHA-256 validation and require it to match the approved round digest; drift invalidates both approvals and starts a fresh round. Do not say "high-accuracy review completed" unless both receipts exist, both final verdicts are unconditional approval, and the final live-plan validation passes.
213
237
 
214
238
  ## Delegation discipline (Codex-native)
215
239
  Every spawn starts with `TASK:`, then DELIVERABLE / SCOPE / VERIFY inside `message`; state the role inside `message` (agent_type is a routing hint, not a guaranteed TOML selection); use `fork_context: false` unless full history is truly required:
@@ -26,7 +26,7 @@ ASK WITH WHY: name what you explored, why it did not resolve, and which part of
26
26
 
27
27
  FOGGIEST-GAP targeting (ordinal, NO numbers): each turn aim at the single open gap whose resolution most unblocks the plan, and say why in one sentence; rotate across equally-foggy components. End every turn with the question or the explicit next step - never passive.
28
28
 
29
- CLEARANCE CHECK after each turn: objective defined? scope IN/OUT explicit? approach decided? test strategy confirmed? no blocking ambiguity left? Any NO is your next question; all YES -> present the approval brief and stop.
29
+ CLEARANCE CHECK after each turn: objective defined? scope IN/OUT explicit? approach decided? test strategy confirmed? constraints swept (budget / stack / scale / audience - each explored, defaulted, or asked)? no blocking ambiguity left? Any NO is your next question; all YES -> present the approval brief and stop.
30
30
  </interview>
31
31
 
32
32
  <approval_and_deliver>
@@ -40,5 +40,6 @@ Request: "add a 5/min-per-IP rate-limit to `/login`".
40
40
  3. Two surviving forks, each asked WITH WHY:
41
41
  - Storage backend (explored: repo already uses Redis; default = Redis; options Redis / in-memory / per-node) - why: persistence across nodes forks the design.
42
42
  - Over-limit response (default = 429 + Retry-After; options 429 / 423 / silent drop) - why: client contract forks on it.
43
+ - Swept axes: no budget/audience fork (internal service); scale bound = existing Redis capacity (defaulted, reversible).
43
44
  4. Approval brief -> explicit okay -> scaffold -> append todos -> if `review_required`, run dual review and deliver receipts; otherwise deliver with the optional review question.
44
45
  </worked_example>
@@ -20,13 +20,13 @@ TOPOLOGY LOCK still applies: enumerate the 1-6 independently-succeed/fail compon
20
20
  </research_protocol>
21
21
 
22
22
  <default_selection>
23
- For each open decision, adopt the defensible best-practice default (industry standard or repo convention), RECORD it in the draft's Open-assumptions ledger with rationale and reversibility, and proceed. NO numeric scoring - the ledger IS the audit trail. The ONLY default escalated to a single focused question is one that is irreversible, destructive, or safety-critical and research cannot settle.
23
+ For each open decision - including the extrinsic axes the sweep names (budget, mandated stack, expected scale, target audience / compliance) - adopt the defensible best-practice default (industry standard or repo convention), RECORD it in the draft's Open-assumptions ledger with rationale and reversibility, and proceed. NO numeric scoring - the ledger IS the audit trail. The ONLY default escalated to a single focused question is one that is irreversible, destructive, or safety-critical, or commits real spend the user never authorized, and research cannot settle.
24
24
 
25
25
  Fold a contrarian self-grill into the Metis spawn: challenge the single highest-leverage adopted assumption - is this constraint real or habitual; does any adopted default add complexity the request never asked for? - and return concrete reframes. The grill targets incidental complexity (unneeded abstraction, speculative capacity), NEVER the feature set: reducing, phasing, or deferring part of the request is not a reframe. Fold a reframe into the plan only as a recommended default plus rationale, never as a forced change.
26
26
  </default_selection>
27
27
 
28
28
  <high_accuracy_auto>
29
- Because the human did not steer, adversarial review SUBSTITUTES for the interview you skipped - this is what catches a bad default. Metis runs during plan generation as always; after Metis findings are folded and the plan file is complete, run the dual high-accuracy review defined in `full-workflow.md` AUTOMATICALLY - no "do you want a review?" question - and resubmit fresh until BOTH passes APPROVE, fixing every cited issue.
29
+ Because the human did not steer, adversarial review SUBSTITUTES for the interview you skipped - this is what catches a bad default. Metis runs during plan generation as always; after Metis findings are folded and the plan file is complete, run the dual high-accuracy review defined in `full-workflow.md` AUTOMATICALLY - no "do you want a review?" question - and drive it to convergence under the bounded convergence contract in `full-workflow.md`: fix every eligible blocker and resubmit fresh, record ineligible findings as non-blocking notes, and on cap exhaustion stop and ask the user.
30
30
 
31
31
  TRIVIAL-TIER GUARD: if Classify sized the work Trivial, the auto-Momus loop is SUPPRESSED (Metis still runs once) - a vague-but-tiny request ("clean this up") must not trigger the full adversarial loop. UNCLEAR raises the research-plus-default posture; it does not override the Trivial cost guard for Momus.
32
32
  </high_accuracy_auto>
@@ -40,5 +40,5 @@ Request: "make auth better".
40
40
  1. Research waves -> current auth at `src/auth/*` and evidence for the requested improvement; best-practice baselines via librarian.
41
41
  2. Topology lock as an ANNOUNCEMENT, not a question: components refine the evidenced auth intent in full, such as session hardening, brute-force protection, and password policy when the repository supports them. MFA is an adjacent capability and stays in Scope OUT unless the user asks for it or evidence establishes it as part of the requested outcome.
42
42
  3. Adopted-defaults table (assumption | default | rationale | reversible?): bcrypt rounds 8 -> 12 (reversible), add 5/min-per-IP login limit (reversible), rotate session id on privilege change (reversible).
43
- 4. Metis folded -> auto dual review (fix cited gaps until both approve) -> brief LEADING with the approach and the defaults, surfaced in the human TL;DR for veto.
43
+ 4. Metis folded -> auto dual review (fix eligible gaps under the bounded convergence contract) -> brief LEADING with the approach and the defaults, surfaced in the human TL;DR for veto.
44
44
  </worked_example>
@@ -15,103 +15,25 @@ const agentSchemaKeys = new Set([
15
15
  "developer_instructions",
16
16
  ]);
17
17
 
18
+ // `includes` entries below are machine-consumed tokens only: EVIDENCE_RECORDED is the sentinel the
19
+ // lazycodex-executor-verify SubagentStop hook greps from worker messages; recommendation, blockers,
20
+ // codeQualityStatus, not_applicable, surfaceEvidence, and adversarialCases are fields the ulw-loop
21
+ // quality gate parses out of reviewer/QA reports; <attemptDir>/<goalId>-code-review.md and
22
+ // -manual-qa.md are the artifact names the ulw-loop spawn guard requires before a gate-reviewer
23
+ // spawn; currentAttemptDir is a field of `ulw-loop status --json`.
18
24
  const lazycodexAgentInvariants = new Map([
19
- [
20
- "explorer.toml",
21
- {
22
- model: "gpt-5.6-luna",
23
- effort: "low",
24
- includes: [/Read-only/, /working tree/, /rg/],
25
- },
26
- ],
27
- [
28
- "librarian.toml",
29
- {
30
- model: "gpt-5.6-luna",
31
- effort: "low",
32
- includes: [/Read-only/, /SHA-pinned GitHub permalink/, /external/],
33
- },
34
- ],
35
- [
36
- "metis.toml",
37
- {
38
- model: "gpt-5.6-sol",
39
- effort: "high",
40
- includes: [/pre-planning analyst/i, /contradictions/, /Read-only/],
41
- },
42
- ],
43
- [
44
- "momus.toml",
45
- {
46
- model: "gpt-5.6-terra",
47
- effort: "high",
48
- includes: [/plan reviewer/i, /OKAY, ITERATE, or REJECT/, /Read-only/],
49
- },
50
- ],
51
- [
52
- "plan.toml",
53
- {
54
- model: "gpt-5.6-sol",
55
- effort: "high",
56
- includes: [/strategic planning consultant/i, /\.omo\/plans\/<slug>\.md/, /never implements/i],
57
- },
58
- ],
59
- [
60
- "lazycodex-worker-low.toml",
61
- {
62
- model: "gpt-5.6-luna",
63
- effort: "high",
64
- includes: [/EVIDENCE_RECORDED: <path>/, /low-difficulty/i, /smallest correct change/i],
65
- },
66
- ],
67
- [
68
- "lazycodex-worker-medium.toml",
69
- {
70
- model: "gpt-5.6-terra",
71
- effort: "high",
72
- includes: [/EVIDENCE_RECORDED: <path>/, /medium-difficulty/i, /smallest correct change/i],
73
- },
74
- ],
75
- [
76
- "lazycodex-worker-high.toml",
77
- {
78
- model: "gpt-5.6-sol",
79
- effort: "medium",
80
- includes: [/EVIDENCE_RECORDED: <path>/, /high-difficulty/i, /smallest correct change/i],
81
- },
82
- ],
83
- [
84
- "lazycodex-clone-fidelity-reviewer.toml",
85
- {
86
- model: "gpt-5.6-terra",
87
- effort: "high",
88
- includes: [/recommendation/, /blockers/, /\.omo\/evidence\/<goal>-clone-fidelity\.md/],
89
- },
90
- ],
91
- [
92
- "lazycodex-code-reviewer.toml",
93
- {
94
- model: "gpt-5.6-terra",
95
- effort: "medium",
96
- includes: [/codeQualityStatus/, /recommendation/, /<attemptDir>\/<goalId>-code-review\.md/, /currentAttemptDir/],
97
- },
98
- ],
99
- [
100
- "lazycodex-qa-executor.toml",
101
- {
102
- model: "gpt-5.6-luna",
103
- effort: "high",
104
- includes: [/not_applicable/, /surfaceEvidence/, /adversarialCases/, /<attemptDir>\/<goalId>-manual-qa\.md/],
105
- },
106
- ],
107
- [
108
- "lazycodex-gate-reviewer.toml",
109
- {
110
- model: "gpt-5.6-sol",
111
- effort: "low",
112
- includes: [/APPROVE\/REJECT/, /blockers/, /<attemptDir>\/<goalId>-gate-review\.md/, /currentAttemptDir/],
113
- },
114
- ],
25
+ ["explorer.toml", { model: "gpt-5.6-luna", effort: "low" }],
26
+ ["librarian.toml", { model: "gpt-5.6-luna", effort: "low" }],
27
+ ["metis.toml", { model: "gpt-5.6-sol", effort: "high" }],
28
+ ["momus.toml", { model: "gpt-5.6-terra", effort: "high" }],
29
+ ["plan.toml", { model: "gpt-5.6-sol", effort: "high" }],
30
+ ["lazycodex-worker-low.toml", { model: "gpt-5.6-luna", effort: "high", includes: [/EVIDENCE_RECORDED: <path>/] }],
31
+ ["lazycodex-worker-medium.toml", { model: "gpt-5.6-terra", effort: "high", includes: [/EVIDENCE_RECORDED: <path>/] }],
32
+ ["lazycodex-worker-high.toml", { model: "gpt-5.6-sol", effort: "medium", includes: [/EVIDENCE_RECORDED: <path>/] }],
33
+ ["lazycodex-clone-fidelity-reviewer.toml", { model: "gpt-5.6-terra", effort: "high", includes: [/recommendation/, /blockers/] }],
34
+ ["lazycodex-code-reviewer.toml", { model: "gpt-5.6-terra", effort: "medium", includes: [/codeQualityStatus/, /recommendation/, /<attemptDir>\/<goalId>-code-review\.md/, /currentAttemptDir/] }],
35
+ ["lazycodex-qa-executor.toml", { model: "gpt-5.6-luna", effort: "high", includes: [/not_applicable/, /surfaceEvidence/, /adversarialCases/, /<attemptDir>\/<goalId>-manual-qa\.md/] }],
36
+ ["lazycodex-gate-reviewer.toml", { model: "gpt-5.6-sol", effort: "low", includes: [/blockers/, /currentAttemptDir/] }],
115
37
  ]);
116
38
 
117
39
  const externalSourceTokenPattern = new RegExp(
@@ -156,7 +78,6 @@ test("#given bundled Codex agents #when components/ultrawork/agents directory is
156
78
  }
157
79
  }
158
80
  });
159
-
160
81
  test("#given bundled agent TOMLs #when nickname_candidates are inspected #then they use only the codex-accepted charset", async () => {
161
82
  // given: codex_app_server ignores a role whose nickname has characters outside
162
83
  // ASCII letters, digits, spaces, hyphens, underscores (observed live in task-15 QA)
@@ -174,16 +95,6 @@ test("#given bundled agent TOMLs #when nickname_candidates are inspected #then t
174
95
  }
175
96
  });
176
97
 
177
- test("#given planner agent prompt #when inspected #then generated artifacts stay under .omo", async () => {
178
- const prompt = await readFile(join(root, "components", "ultrawork", "agents", "plan.toml"), "utf8");
179
-
180
- assert.match(prompt, /\.omo\/plans\/<slug>\.md/);
181
- assert.match(prompt, /<attemptDir>\/task-<N>-<slug>\.<ext>/);
182
- assert.match(prompt, /\.omo\/evidence\/ulw\/<session>\/<goalId>\/a<attempt>/);
183
- assert.doesNotMatch(prompt, /(?<!\.omo\/)plans\/<slug>\.md/);
184
- assert.doesNotMatch(prompt, /(?<!\.omo\/)evidence\/task-/);
185
- });
186
-
187
98
  test("#given lazycodex agent prompts #when inspected #then each role pins model effort and evidence discipline", async () => {
188
99
  const agentsDir = join(root, "components", "ultrawork", "agents");
189
100
 
@@ -197,73 +108,8 @@ test("#given lazycodex agent prompts #when inspected #then each role pins model
197
108
  assert.doesNotMatch(prompt, /^blocking\s*=/m);
198
109
  assert.doesNotMatch(prompt, externalSourceTokenPattern);
199
110
 
200
- for (const pattern of invariant.includes) {
111
+ for (const pattern of invariant.includes ?? []) {
201
112
  assert.match(prompt, pattern, `${fileName} must include ${pattern}`);
202
113
  }
203
114
  }
204
115
  });
205
-
206
- test("#given LazyCodex reviewer prompts #when inspected #then anti-slop review coverage is required", async () => {
207
- const agentsDir = join(root, "components", "ultrawork", "agents");
208
- const codeReviewer = await readFile(join(agentsDir, "lazycodex-code-reviewer.toml"), "utf8");
209
- const gateReviewer = await readFile(join(agentsDir, "lazycodex-gate-reviewer.toml"), "utf8");
210
-
211
- assert.match(codeReviewer, /remove-ai-slops/);
212
- assert.match(codeReviewer, /programming/);
213
- assert.match(codeReviewer, /load or consult/);
214
- assert.match(codeReviewer, /documented criteria/);
215
- assert.match(codeReviewer, /violates either skill perspective/);
216
- assert.match(codeReviewer, /overfit\/slop review pass/);
217
- assert.match(codeReviewer, /deletion-only tests/);
218
- assert.match(codeReviewer, /tests that merely verify a requested removal/);
219
- assert.match(codeReviewer, /tautological tests/);
220
- assert.match(codeReviewer, /mirror implementation constants/);
221
- assert.match(codeReviewer, /unnecessary production data extraction, parsing, or normalization/);
222
- assert.match(codeReviewer, /false confidence/);
223
-
224
- assert.match(gateReviewer, /remove-ai-slops/);
225
- assert.match(gateReviewer, /programming/);
226
- assert.match(gateReviewer, /load or consult/);
227
- assert.match(gateReviewer, /documented criteria/);
228
- assert.match(gateReviewer, /Run the `remove-ai-slops`/);
229
- assert.match(gateReviewer, /Apply the `programming`/);
230
- assert.match(gateReviewer, /overfit\/slop pass yourself/);
231
- assert.match(gateReviewer, /tests that merely verify a requested removal/);
232
- assert.match(gateReviewer, /deletion-only/);
233
- assert.match(gateReviewer, /tautological/);
234
- assert.match(gateReviewer, /implementation-mirroring tests/);
235
- assert.match(gateReviewer, /unnecessary production extraction, parsing, or normalization/);
236
-
237
- const directPassIndex = gateReviewer.indexOf("overfit/slop pass yourself");
238
- const reportCoverageIndex = gateReviewer.indexOf("Then confirm the code review report");
239
- assert.notEqual(directPassIndex, -1);
240
- assert.notEqual(reportCoverageIndex, -1);
241
- assert.ok(
242
- directPassIndex < reportCoverageIndex,
243
- "gate reviewer must perform the overfit/slop pass directly before checking report coverage",
244
- );
245
- });
246
-
247
- test("#given done-gate reviewer prompts #when inspected #then burden of proof is approve-unless-cited and reject priors are gone", async () => {
248
- const agentsDir = join(root, "components", "ultrawork", "agents");
249
- const gateReviewer = await readFile(join(agentsDir, "lazycodex-gate-reviewer.toml"), "utf8");
250
- const qaExecutor = await readFile(join(agentsDir, "lazycodex-qa-executor.toml"), "utf8");
251
- const codeReviewer = await readFile(join(agentsDir, "lazycodex-code-reviewer.toml"), "utf8");
252
-
253
- assert.match(gateReviewer, /APPROVE unless you can cite/);
254
- assert.match(gateReviewer, /violatedCriterion/);
255
- assert.match(gateReviewer, /evidencePointer/);
256
- assert.match(gateReviewer, /top blockers inline/);
257
- assert.match(gateReviewer, /is a NOTE, not a blocker/);
258
- assert.match(gateReviewer, /You do NOT check/);
259
- assert.doesNotMatch(gateReviewer, /Assume the work has already failed/);
260
- assert.doesNotMatch(gateReviewer, /Return exactly one recommendation: APPROVE\/REJECT\./);
261
-
262
- assert.match(qaExecutor, /one-line reason/);
263
- assert.match(qaExecutor, /rejecting a legitimately untriggered class is itself an error/);
264
- assert.doesNotMatch(qaExecutor, /Trust nothing\./);
265
-
266
- assert.match(codeReviewer, /MEDIUM by default/);
267
- assert.doesNotMatch(codeReviewer, /Treat useless tests or needless production complexity as CRITICAL\/HIGH/);
268
- });
269
-
@@ -84,7 +84,7 @@ test("#given aggregate SubagentStop hooks #when inspected #then start-work and L
84
84
  assert.equal(verifierGroups.length, 1);
85
85
  assert.equal(verifierGroups[0]?.groupIndex, 0);
86
86
  assert.equal(verifierGroups[0]?.handler.timeout, 10);
87
- assert.match(verifierGroups[0]?.handler.statusMessage ?? "", /^\(OmO [^)]+\) Verifying LazyCodex Executor Evidence$/);
87
+ assert.match(verifierGroups[0]?.handler.statusMessage ?? "", /\S/);
88
88
  });
89
89
 
90
90
  test("#given aggregate PostCompact hooks #when hooks are inspected #then LSP diagnostics cache reset is registered", async () => {
@@ -100,7 +100,7 @@ test("#given aggregate PostCompact hooks #when hooks are inspected #then LSP dia
100
100
 
101
101
  // then
102
102
  assert.equal(lspPostCompactHooks.length, 1);
103
- assert.match(lspPostCompactHooks[0]?.handler.statusMessage ?? "", /^\(OmO [^)]+\) Resetting LSP Diagnostics Cache$/);
103
+ assert.match(lspPostCompactHooks[0]?.handler.statusMessage ?? "", /\S/);
104
104
  });
105
105
 
106
106
  test("#given aggregate hook commands #when inspected #then every command exposes a Codex status message", async () => {
@@ -142,23 +142,6 @@ test("#given component hook commands #when inspected #then standalone packages e
142
142
  assert.deepEqual(missingStatusMessages, []);
143
143
  });
144
144
 
145
- test("#given hook status messages #when inspected #then labels describe OMO responsibilities instead of the hook runner", async () => {
146
- // given
147
- const componentHooks = await readComponentHookManifests();
148
-
149
- // when
150
- const commandHooks = [
151
- ...(await readAggregateCommandHooks()),
152
- ...componentHooks.flatMap(({ source, hooks }) => collectCommandHooks(hooks, source)),
153
- ];
154
- const genericStatusMessages = commandHooks
155
- .filter(({ handler }) => typeof handler.statusMessage !== "string" || /\bhook\b/i.test(handler.statusMessage))
156
- .map(hookLocation);
157
-
158
- // then
159
- assert.deepEqual(genericStatusMessages, []);
160
- });
161
-
162
145
  test("#given aggregate OMO plugin is enabled #when hooks are inspected #then shell guidance and ulw-loop guard are registered", async () => {
163
146
  // given
164
147
  const manifests = await readAggregateHookManifests();
@@ -169,9 +152,7 @@ test("#given aggregate OMO plugin is enabled #when hooks are inspected #then she
169
152
 
170
153
  // then
171
154
  assert.match(text, /components\/git-bash\/dist\/cli\.js/);
172
- assert.match(text, /Recommending Git Bash MCP/);
173
155
  assert.match(text, /hook post-compact/);
174
- assert.match(text, /Resetting Git Bash MCP Reminder/);
175
156
  assert.match(text, /components\/ulw-loop\/dist\/cli\.js/);
176
157
  assert.match(text, /hook pre-tool-use/);
177
158
  assert.deepEqual(preToolUseGroups.map((group) => group.matcher), [
@@ -221,7 +202,6 @@ test("#given aggregate SessionStart hooks #when inspected #then LazyCodex auto-u
221
202
  // then
222
203
  assert.equal(autoUpdateGroup?.matcher, "^startup$");
223
204
  assert.match(text, /scripts\/auto-update\.mjs/);
224
- assert.match(text, /Checking Auto Update/);
225
205
  assert(sessionStartCommands.some((command) => command.includes("scripts/auto-update.mjs")));
226
206
  });
227
207
 
@@ -265,7 +245,7 @@ test("#given aggregate PostToolUse hooks #when inspected #then CodeGraph init gu
265
245
  // then
266
246
  assert.equal(codegraphPostToolUseHooks.length, 1);
267
247
  assert.equal(codegraphPostToolUseHooks[0]?.matcher, "^(codegraph[._].*|mcp__codegraph__.*)$");
268
- assert.match(codegraphPostToolUseHooks[0]?.handler.statusMessage ?? "", /^\(OmO [^)]+\) Checking CodeGraph Init Guidance$/);
248
+ assert.match(codegraphPostToolUseHooks[0]?.handler.statusMessage ?? "", /\S/);
269
249
  });
270
250
 
271
251
  test("#given aggregate PostToolUse hooks #when inspected #then thread title hygiene is registered for created Codex threads", async () => {
@@ -282,7 +262,7 @@ test("#given aggregate PostToolUse hooks #when inspected #then thread title hygi
282
262
  // then
283
263
  assert.equal(threadTitleHooks.length, 1);
284
264
  assert.equal(threadTitleHooks[0]?.matcher, "^(create_thread|codex_app\\.create_thread)$");
285
- assert.match(threadTitleHooks[0]?.handler.statusMessage ?? "", /^\(OmO [^)]+\) Checking Thread Title Hygiene$/);
265
+ assert.match(threadTitleHooks[0]?.handler.statusMessage ?? "", /\S/);
286
266
  });
287
267
 
288
268
  test("#given aggregate plugin packaging #when inspected #then hooks and compatibility sentinels stay Python-free", async () => {
@@ -1,10 +1,8 @@
1
1
  import { readdir, readFile, stat } from "node:fs/promises";
2
- import { dirname, join } from "node:path";
2
+ import { basename, dirname, join, sep } from "node:path";
3
3
  import { fileURLToPath } from "node:url";
4
-
5
4
  export const root = dirname(dirname(fileURLToPath(import.meta.url)));
6
5
  export const repoRoot = join(root, "..", "..", "..");
7
-
8
6
  export async function readJson(relativePath) {
9
7
  return JSON.parse(await readFile(join(root, relativePath), "utf8"));
10
8
  }
@@ -79,18 +77,182 @@ export function hookLocation({ source, eventName, groupIndex, handlerIndex, hand
79
77
  return `${source}:${eventName}:${groupIndex}:${handlerIndex}:${handler.command}`;
80
78
  }
81
79
 
82
- export function findInvalidSpawnAgentRoleParameters(content) {
83
- return [...content.matchAll(/spawn_agent\([^)\n]*(?:agent_type|model|reasoning_effort)\s*=/g)].map((match) => match[0]);
80
+ const SPAWN_AGENT_START = /(?:(?<receiver>\b[A-Za-z_]\w*)\s*\.\s*)?(?<callee>\bspawn_agent)\s*\(/g;
81
+
82
+ async function collectFiles(directory, predicate) {
83
+ let entries;
84
+ try {
85
+ entries = await readdir(directory, { withFileTypes: true });
86
+ } catch (error) {
87
+ if (error instanceof Error && "code" in error && error.code === "ENOENT") return [];
88
+ throw error;
89
+ }
90
+
91
+ const files = [];
92
+ for (const entry of entries) {
93
+ const path = join(directory, entry.name);
94
+ if (entry.isDirectory()) files.push(...(await collectFiles(path, predicate)));
95
+ else if (entry.isFile() && predicate(path)) files.push(path);
96
+ }
97
+ return files;
84
98
  }
85
99
 
86
- export function findSpawnAgentCallsWithoutForkContextFalse(content) {
87
- const missingForkContext = [];
88
- const regex = /spawn_agent\(([^)]*)\)/g;
89
- for (const match of content.matchAll(regex)) {
90
- const call = match[0];
91
- if (!/"fork_context"\s*:\s*false|fork_context:\s*false|fork_context=false|"fork_turns"\s*:\s*"none"|fork_turns:\s*"none"|fork_turns="none"/.test(call)) {
92
- missingForkContext.push(call);
100
+ export async function listShippedSpawnPromptFiles() {
101
+ const skillFiles = [
102
+ ...(await collectFiles(join(root, "skills"), (path) => basename(path) === "SKILL.md")),
103
+ ...(await collectFiles(
104
+ join(root, "components"),
105
+ (path) => path.includes(`${sep}skills${sep}`) && basename(path) === "SKILL.md",
106
+ )),
107
+ ];
108
+ const hephaestusRules = await collectFiles(
109
+ join(root, "components", "rules", "bundled-rules", "hephaestus"),
110
+ (path) => path.endsWith(".md"),
111
+ );
112
+ return [...skillFiles, ...hephaestusRules].sort();
113
+ }
114
+
115
+ export function findSpawnAgentCalls(content) {
116
+ const calls = [];
117
+ for (const match of content.matchAll(SPAWN_AGENT_START)) {
118
+ const openingParenthesis = match.index + match[0].lastIndexOf("(");
119
+ let depth = 1;
120
+ let quote = null;
121
+ let escaped = false;
122
+
123
+ for (let index = openingParenthesis + 1; index < content.length; index += 1) {
124
+ const character = content[index];
125
+ if (quote !== null) {
126
+ if (escaped) escaped = false;
127
+ else if (character === "\\") escaped = true;
128
+ else if (character === quote) quote = null;
129
+ continue;
130
+ }
131
+ if (character === '"' || character === "'" || character === "`") {
132
+ quote = character;
133
+ continue;
134
+ }
135
+ if (character === "(") depth += 1;
136
+ else if (character === ")") depth -= 1;
137
+ if (depth !== 0) continue;
138
+
139
+ const { receiver = null, callee } = match.groups;
140
+ calls.push({ call: content.slice(match.index, index + 1), index: match.index, receiver, callee });
141
+ break;
93
142
  }
94
143
  }
95
- return missingForkContext;
144
+ return calls;
96
145
  }
146
+
147
+ function tokensForSpawnCall(call) {
148
+ const tokens = [];
149
+ let braceDepth = 0;
150
+ let bracketDepth = 0;
151
+ let parenthesisDepth = 0;
152
+
153
+ for (let index = call.indexOf("(") + 1; index < call.length - 1; ) {
154
+ const character = call[index];
155
+ if (/\s/.test(character)) {
156
+ index += 1;
157
+ continue;
158
+ }
159
+ if (character === '"' || character === "'" || character === "`") {
160
+ const start = index;
161
+ const quote = character;
162
+ let value = "";
163
+ index += 1;
164
+ while (index < call.length - 1) {
165
+ const current = call[index];
166
+ if (current === "\\" && index + 1 < call.length - 1) {
167
+ value += call[index + 1];
168
+ index += 2;
169
+ continue;
170
+ }
171
+ index += 1;
172
+ if (current === quote) break;
173
+ value += current;
174
+ }
175
+ tokens.push({ type: "string", value, start, end: index, braceDepth, bracketDepth, parenthesisDepth });
176
+ continue;
177
+ }
178
+ if (/[A-Za-z_]/.test(character)) {
179
+ const start = index;
180
+ while (/[A-Za-z0-9_]/.test(call[index] ?? "")) index += 1;
181
+ tokens.push({ type: "identifier", value: call.slice(start, index), start, end: index,
182
+ braceDepth, bracketDepth, parenthesisDepth });
183
+ continue;
184
+ }
185
+
186
+ if (character === "{") braceDepth += 1;
187
+ else if (character === "}") braceDepth -= 1;
188
+ else if (character === "[") bracketDepth += 1;
189
+ else if (character === "]") bracketDepth -= 1;
190
+ else if (character === "(") parenthesisDepth += 1;
191
+ else if (character === ")") parenthesisDepth -= 1;
192
+ else if (character === ":" || character === "=") {
193
+ tokens.push({ type: "separator", value: character, start: index, end: index + 1,
194
+ braceDepth, bracketDepth, parenthesisDepth });
195
+ }
196
+ index += 1;
197
+ }
198
+ return tokens;
199
+ }
200
+
201
+ function spawnParameters(call) {
202
+ const tokens = tokensForSpawnCall(call);
203
+ const parameters = [];
204
+ for (let index = 0; index < tokens.length - 2; index += 1) {
205
+ const key = tokens[index];
206
+ const separator = tokens[index + 1];
207
+ const value = tokens[index + 2];
208
+ const isArgumentScope = key.parenthesisDepth === 0 && key.bracketDepth === 0 && key.braceDepth <= 1;
209
+ const isSameScope =
210
+ separator.braceDepth === key.braceDepth && separator.bracketDepth === key.bracketDepth &&
211
+ separator.parenthesisDepth === key.parenthesisDepth && value.braceDepth === key.braceDepth &&
212
+ value.bracketDepth === key.bracketDepth &&
213
+ value.parenthesisDepth === key.parenthesisDepth;
214
+ const isAdjacent = /^\s*$/.test(call.slice(key.end, separator.start)) &&
215
+ /^\s*$/.test(call.slice(separator.end, value.start));
216
+ if (!isArgumentScope || !isSameScope || !isAdjacent || separator.type !== "separator") continue;
217
+ parameters.push({ name: key.value, value, direct: key.braceDepth === 0 });
218
+ }
219
+ return parameters;
220
+ }
221
+
222
+ export function hasSpawnIsolationArgument(call) {
223
+ return spawnParameters(call).some(
224
+ ({ name, value }) => (name === "fork_context" && value.type === "identifier" && value.value === "false") ||
225
+ (name === "fork_turns" && value.type === "string" && value.value === "none"),
226
+ );
227
+ }
228
+
229
+ export function findSpawnAgentCallsWithoutIsolation(content) {
230
+ return findSpawnAgentCalls(content).filter(({ call }) => !hasSpawnIsolationArgument(call));
231
+ }
232
+
233
+ export function findSpawnAgentCallsWithUnsupportedParameters(content) {
234
+ return findSpawnAgentCalls(content).flatMap((entry) => {
235
+ const allowsObjectAgentType = entry.receiver === "multi_agent_v1";
236
+ const parameters = spawnParameters(entry.call)
237
+ .filter(({ name, direct }) => name === "model" || name === "reasoning_effort" ||
238
+ (name === "agent_type" && (direct || !allowsObjectAgentType)))
239
+ .map(({ name }) => name);
240
+ return parameters.length === 0 ? [] : [{ ...entry, parameters }];
241
+ });
242
+ }
243
+
244
+ async function findShippedSpawnViolations(findViolations) {
245
+ const violations = [];
246
+ for (const path of await listShippedSpawnPromptFiles()) {
247
+ const content = await readFile(path, "utf8");
248
+ for (const violation of findViolations(content)) {
249
+ const line = content.slice(0, violation.index).split("\n").length;
250
+ violations.push({ path, line, ...violation });
251
+ }
252
+ }
253
+ return violations;
254
+ }
255
+
256
+ export const findShippedSpawnIsolationViolations = () => findShippedSpawnViolations(findSpawnAgentCallsWithoutIsolation);
257
+ export const findShippedUnsupportedSpawnParameterViolations = () =>
258
+ findShippedSpawnViolations(findSpawnAgentCallsWithUnsupportedParameters);