lazycodex-ai 5.0.0-beta.85 → 5.0.0-beta.87

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (176) hide show
  1. package/README.md +1 -1
  2. package/dist/cli/index.js +68 -41
  3. package/dist/cli-node/index.js +68 -41
  4. package/package.json +1 -1
  5. package/packages/omo-codex/plugin/.codex-plugin/plugin.json +1 -1
  6. package/packages/omo-codex/plugin/components/bootstrap/dist/cli.js +2 -0
  7. package/packages/omo-codex/plugin/components/bootstrap/hooks/hooks.json +1 -1
  8. package/packages/omo-codex/plugin/components/bootstrap/package.json +1 -1
  9. package/packages/omo-codex/plugin/components/comment-checker/hooks/hooks.json +1 -1
  10. package/packages/omo-codex/plugin/components/comment-checker/package.json +1 -1
  11. package/packages/omo-codex/plugin/components/git-bash/hooks/hooks.json +2 -2
  12. package/packages/omo-codex/plugin/components/git-bash/package.json +1 -1
  13. package/packages/omo-codex/plugin/components/lazycodex-executor-verify/hooks/hooks.json +1 -1
  14. package/packages/omo-codex/plugin/components/lazycodex-executor-verify/package.json +1 -1
  15. package/packages/omo-codex/plugin/components/lsp/dist/.omo-runtime-manifest.json +2 -2
  16. package/packages/omo-codex/plugin/components/lsp/hooks/hooks.json +2 -2
  17. package/packages/omo-codex/plugin/components/lsp/package.json +1 -1
  18. package/packages/omo-codex/plugin/components/rules/bundled-rules/hephaestus/gpt-6.md +1 -1
  19. package/packages/omo-codex/plugin/components/rules/hooks/hooks.json +4 -4
  20. package/packages/omo-codex/plugin/components/rules/package.json +1 -1
  21. package/packages/omo-codex/plugin/components/teammode/hooks/hooks.json +1 -1
  22. package/packages/omo-codex/plugin/components/teammode/package.json +1 -1
  23. package/packages/omo-codex/plugin/components/telemetry/hooks/hooks.json +1 -1
  24. package/packages/omo-codex/plugin/components/telemetry/package.json +1 -1
  25. package/packages/omo-codex/plugin/components/ultrawork/README.md +1 -1
  26. package/packages/omo-codex/plugin/components/ultrawork/agents/plan.toml +2 -2
  27. package/packages/omo-codex/plugin/components/ultrawork/dist/cli.js +63 -89
  28. package/packages/omo-codex/plugin/components/ultrawork/hooks/hooks.json +1 -1
  29. package/packages/omo-codex/plugin/components/ultrawork/package.json +1 -1
  30. package/packages/omo-codex/plugin/components/ultrawork/src/directive-content.ts +1 -1
  31. package/packages/omo-codex/plugin/components/ulw-execute-continuation/directive.md +2 -2
  32. package/packages/omo-codex/plugin/components/ulw-execute-continuation/hooks/hooks.json +1 -1
  33. package/packages/omo-codex/plugin/components/ulw-execute-continuation/package.json +1 -1
  34. package/packages/omo-codex/plugin/components/ulw-loop/directive.md +63 -89
  35. package/packages/omo-codex/plugin/components/ulw-loop/hooks/hooks.json +5 -5
  36. package/packages/omo-codex/plugin/components/ulw-loop/package.json +1 -1
  37. package/packages/omo-codex/plugin/components/ulw-loop/skills/ulw-loop/references/define-goal.md +2 -3
  38. package/packages/omo-codex/plugin/components/ulw-loop/skills/ulw-loop/references/full-workflow.md +2 -2
  39. package/packages/omo-codex/plugin/hooks/post-compact-resetting-git-bash-mcp-reminder.json +1 -1
  40. package/packages/omo-codex/plugin/hooks/post-compact-resetting-lsp-diagnostics-cache.json +1 -1
  41. package/packages/omo-codex/plugin/hooks/post-compact-resetting-project-rule-cache.json +1 -1
  42. package/packages/omo-codex/plugin/hooks/post-tool-use-checking-comments.json +1 -1
  43. package/packages/omo-codex/plugin/hooks/post-tool-use-checking-lsp-diagnostics.json +1 -1
  44. package/packages/omo-codex/plugin/hooks/post-tool-use-checking-thread-title-hygiene.json +1 -1
  45. package/packages/omo-codex/plugin/hooks/post-tool-use-matching-project-rules.json +1 -1
  46. package/packages/omo-codex/plugin/hooks/post-tool-use-recording-spawn-admission.json +1 -1
  47. package/packages/omo-codex/plugin/hooks/pre-tool-use-enforcing-unlimited-goal-budget.json +1 -1
  48. package/packages/omo-codex/plugin/hooks/pre-tool-use-guarding-ulw-loop-spawns.json +1 -1
  49. package/packages/omo-codex/plugin/hooks/pre-tool-use-recommending-git-bash-mcp.json +1 -1
  50. package/packages/omo-codex/plugin/hooks/session-start-checking-auto-update.json +1 -1
  51. package/packages/omo-codex/plugin/hooks/session-start-checking-bootstrap-provisioning.json +1 -1
  52. package/packages/omo-codex/plugin/hooks/session-start-loading-project-rules.json +1 -1
  53. package/packages/omo-codex/plugin/hooks/session-start-recording-session-telemetry.json +1 -1
  54. package/packages/omo-codex/plugin/hooks/stop-checking-ulw-execute-continuation.json +1 -1
  55. package/packages/omo-codex/plugin/hooks/stop-checking-ulw-loop-resume.json +1 -1
  56. package/packages/omo-codex/plugin/hooks/subagent-stop-verifying-lazycodex-executor-evidence.json +1 -1
  57. package/packages/omo-codex/plugin/hooks/user-prompt-submit-checking-ultrawork-trigger.json +1 -1
  58. package/packages/omo-codex/plugin/hooks/user-prompt-submit-checking-ulw-loop-steering.json +1 -1
  59. package/packages/omo-codex/plugin/hooks/user-prompt-submit-loading-project-rules.json +1 -1
  60. package/packages/omo-codex/plugin/package-lock.json +12 -12
  61. package/packages/omo-codex/plugin/package.json +1 -1
  62. package/packages/omo-codex/plugin/scripts/materialize-shared-upstreams.mjs +8 -2
  63. package/packages/omo-codex/plugin/scripts/sync-skills.mjs +2 -2
  64. package/packages/omo-codex/plugin/skills/browser/ATTRIBUTION.md +26 -14
  65. package/packages/omo-codex/plugin/skills/browser/SKILL.md +65 -52
  66. package/packages/omo-codex/plugin/skills/browser/references/commands.md +81 -66
  67. package/packages/omo-codex/plugin/skills/browser/references/install.md +31 -34
  68. package/packages/omo-codex/plugin/skills/browser/references/owned-engine/README.md +41 -21
  69. package/packages/omo-codex/plugin/skills/browser/references/owned-engine/frames-and-humans.md +33 -20
  70. package/packages/omo-codex/plugin/skills/browser/references/owned-engine/ladder.md +24 -22
  71. package/packages/omo-codex/plugin/skills/browser/references/owned-engine/network.md +35 -15
  72. package/packages/omo-codex/plugin/skills/browser/references/recipes/1password.md +13 -11
  73. package/packages/omo-codex/plugin/skills/browser/references/remote.md +5 -4
  74. package/packages/omo-codex/plugin/skills/browser/runtime/omowright/index.js +1534 -0
  75. package/packages/omo-codex/plugin/skills/browser/runtime/omowright/manifest.json +9 -0
  76. package/packages/omo-codex/plugin/skills/browser/runtime/omowright/page-bundle.js +1395 -0
  77. package/packages/omo-codex/plugin/skills/browser/scripts/browser-doctor.mjs +37 -32
  78. package/packages/omo-codex/plugin/skills/browser/scripts/browser-install.mjs +31 -44
  79. package/packages/omo-codex/plugin/skills/browser/scripts/omowright.mjs +25 -0
  80. package/packages/omo-codex/plugin/skills/debugging/SKILL.md +2 -2
  81. package/packages/omo-codex/plugin/skills/debugging/references/methodology/06-fix.md +3 -3
  82. package/packages/omo-codex/plugin/skills/debugging/references/methodology/08-qa.md +1 -1
  83. package/packages/omo-codex/plugin/skills/debugging/references/tools/browser-qa.md +104 -0
  84. package/packages/omo-codex/plugin/skills/frontend/SKILL.md +1 -1
  85. package/packages/omo-codex/plugin/skills/frontend/references/design/clone-from-url.md +1 -1
  86. package/packages/omo-codex/plugin/skills/programming/SKILL.md +12 -18
  87. package/packages/omo-codex/plugin/skills/programming/references/rust/README.md +43 -15
  88. package/packages/omo-codex/plugin/skills/programming/references/rust/api-design.md +81 -0
  89. package/packages/omo-codex/plugin/skills/programming/references/rust/async-tokio.md +60 -28
  90. package/packages/omo-codex/plugin/skills/programming/references/rust/axum-stack.md +1 -13
  91. package/packages/omo-codex/plugin/skills/programming/references/rust/cargo-strict.md +44 -8
  92. package/packages/omo-codex/plugin/skills/programming/references/rust/clap-stack.md +8 -3
  93. package/packages/omo-codex/plugin/skills/programming/references/rust/concurrency.md +66 -52
  94. package/packages/omo-codex/plugin/skills/programming/references/rust/libraries.md +35 -25
  95. package/packages/omo-codex/plugin/skills/programming/references/rust/macros.md +63 -0
  96. package/packages/omo-codex/plugin/skills/programming/references/rust/one-liners.md +5 -3
  97. package/packages/omo-codex/plugin/skills/programming/references/rust/proptest-insta.md +8 -0
  98. package/packages/omo-codex/plugin/skills/programming/references/rust/type-state.md +50 -12
  99. package/packages/omo-codex/plugin/skills/programming/references/rust/unsafe-discipline.md +34 -6
  100. package/packages/omo-codex/plugin/skills/programming/references/rust/zero-cost-safety.md +62 -52
  101. package/packages/omo-codex/plugin/skills/programming/references/rust-ub/miri-sanitizers-loom.md +1 -1
  102. package/packages/omo-codex/plugin/skills/programming/references/rust-ub/ub-taxonomy.md +6 -3
  103. package/packages/omo-codex/plugin/skills/programming/scripts/rust/check-no-excuse-rules.sh +86 -75
  104. package/packages/omo-codex/plugin/skills/programming/scripts/rust/new-project.py +31 -28
  105. package/packages/omo-codex/plugin/skills/review-work/SKILL.md +1 -1
  106. package/packages/omo-codex/plugin/skills/ultimate-browsing/SKILL.md +27 -20
  107. package/packages/omo-codex/plugin/skills/ultimate-browsing/engine/AGENTS.md +1 -1
  108. package/packages/omo-codex/plugin/skills/ultimate-browsing/references/chrome-stealth.md +32 -100
  109. package/packages/omo-codex/plugin/skills/ultimate-browsing/references/insane-search/README.md +5 -11
  110. package/packages/omo-codex/plugin/skills/ultimate-browsing/references/insane-search/playwright.md +20 -37
  111. package/packages/omo-codex/plugin/skills/ultrawork/SKILL.md +63 -89
  112. package/packages/omo-codex/plugin/skills/ulw-execute/SKILL.md +4 -4
  113. package/packages/omo-codex/plugin/skills/ulw-loop/references/define-goal.md +2 -3
  114. package/packages/omo-codex/plugin/skills/ulw-loop/references/full-workflow.md +2 -2
  115. package/packages/omo-codex/plugin/skills/visual-qa/SKILL.md +1 -1
  116. package/packages/omo-codex/plugin/skills/visual-qa/references/browser-setup.md +46 -46
  117. package/packages/omo-codex/plugin/test/sync-skills-test-support.mjs +2 -2
  118. package/packages/omo-codex/scripts/install-dist/install-local.mjs +4 -2
  119. package/packages/prompts-core/prompts/ultrawork/codex.md +63 -89
  120. package/packages/shared-skills/skills/browser/ATTRIBUTION.md +26 -14
  121. package/packages/shared-skills/skills/browser/SKILL.md +65 -52
  122. package/packages/shared-skills/skills/browser/references/commands.md +81 -66
  123. package/packages/shared-skills/skills/browser/references/install.md +31 -34
  124. package/packages/shared-skills/skills/browser/references/owned-engine/README.md +41 -21
  125. package/packages/shared-skills/skills/browser/references/owned-engine/frames-and-humans.md +33 -20
  126. package/packages/shared-skills/skills/browser/references/owned-engine/ladder.md +24 -22
  127. package/packages/shared-skills/skills/browser/references/owned-engine/network.md +35 -15
  128. package/packages/shared-skills/skills/browser/references/recipes/1password.md +13 -11
  129. package/packages/shared-skills/skills/browser/references/remote.md +5 -4
  130. package/packages/shared-skills/skills/browser/runtime/omowright/index.js +1534 -0
  131. package/packages/shared-skills/skills/browser/runtime/omowright/manifest.json +9 -0
  132. package/packages/shared-skills/skills/browser/runtime/omowright/page-bundle.js +1395 -0
  133. package/packages/shared-skills/skills/browser/scripts/browser-doctor.mjs +37 -32
  134. package/packages/shared-skills/skills/browser/scripts/browser-install.mjs +31 -44
  135. package/packages/shared-skills/skills/browser/scripts/omowright.mjs +25 -0
  136. package/packages/shared-skills/skills/debugging/SKILL.md +2 -2
  137. package/packages/shared-skills/skills/debugging/references/methodology/06-fix.md +3 -3
  138. package/packages/shared-skills/skills/debugging/references/methodology/08-qa.md +1 -1
  139. package/packages/shared-skills/skills/debugging/references/tools/browser-qa.md +104 -0
  140. package/packages/shared-skills/skills/frontend/SKILL.md +1 -1
  141. package/packages/shared-skills/skills/frontend/references/design/clone-from-url.md +1 -1
  142. package/packages/shared-skills/skills/programming/SKILL.md +12 -18
  143. package/packages/shared-skills/skills/programming/references/rust/README.md +43 -15
  144. package/packages/shared-skills/skills/programming/references/rust/api-design.md +81 -0
  145. package/packages/shared-skills/skills/programming/references/rust/async-tokio.md +60 -28
  146. package/packages/shared-skills/skills/programming/references/rust/axum-stack.md +1 -13
  147. package/packages/shared-skills/skills/programming/references/rust/cargo-strict.md +44 -8
  148. package/packages/shared-skills/skills/programming/references/rust/clap-stack.md +8 -3
  149. package/packages/shared-skills/skills/programming/references/rust/concurrency.md +66 -52
  150. package/packages/shared-skills/skills/programming/references/rust/libraries.md +35 -25
  151. package/packages/shared-skills/skills/programming/references/rust/macros.md +63 -0
  152. package/packages/shared-skills/skills/programming/references/rust/one-liners.md +5 -3
  153. package/packages/shared-skills/skills/programming/references/rust/proptest-insta.md +8 -0
  154. package/packages/shared-skills/skills/programming/references/rust/type-state.md +50 -12
  155. package/packages/shared-skills/skills/programming/references/rust/unsafe-discipline.md +34 -6
  156. package/packages/shared-skills/skills/programming/references/rust/zero-cost-safety.md +62 -52
  157. package/packages/shared-skills/skills/programming/references/rust-ub/miri-sanitizers-loom.md +1 -1
  158. package/packages/shared-skills/skills/programming/references/rust-ub/ub-taxonomy.md +6 -3
  159. package/packages/shared-skills/skills/programming/scripts/rust/check-no-excuse-rules.sh +86 -75
  160. package/packages/shared-skills/skills/programming/scripts/rust/check-no-excuse-rules.test.ts +83 -0
  161. package/packages/shared-skills/skills/programming/scripts/rust/new-project.py +31 -28
  162. package/packages/shared-skills/skills/review-work/SKILL.md +1 -1
  163. package/packages/shared-skills/skills/ultimate-browsing/SKILL.md +27 -20
  164. package/packages/shared-skills/skills/ultimate-browsing/engine/AGENTS.md +1 -1
  165. package/packages/shared-skills/skills/ultimate-browsing/references/chrome-stealth.md +32 -100
  166. package/packages/shared-skills/skills/ultimate-browsing/references/insane-search/README.md +5 -11
  167. package/packages/shared-skills/skills/ultimate-browsing/references/insane-search/playwright.md +20 -37
  168. package/packages/shared-skills/skills/ulw-execute/SKILL.md +4 -4
  169. package/packages/shared-skills/skills/visual-qa/SKILL.md +1 -1
  170. package/packages/shared-skills/skills/visual-qa/references/browser-setup.md +46 -46
  171. package/packages/omo-codex/plugin/skills/browser/scripts/browser-env.mjs +0 -41
  172. package/packages/omo-codex/plugin/skills/debugging/references/tools/playwright-cli.md +0 -112
  173. package/packages/omo-codex/plugin/skills/programming/scripts/rust/check-no-excuse-rules.py +0 -296
  174. package/packages/shared-skills/skills/browser/scripts/browser-env.mjs +0 -41
  175. package/packages/shared-skills/skills/debugging/references/tools/playwright-cli.md +0 -112
  176. package/packages/shared-skills/skills/programming/scripts/rust/check-no-excuse-rules.py +0 -296
@@ -17,9 +17,9 @@ Expert coding agent. Ship verified work. No process narration.
17
17
 
18
18
  # Goal
19
19
  Deliver EXACTLY what the user asked, end-to-end working, proven by
20
- captured evidence: a failing-first proof that went RED→GREEN through
21
- the cheapest faithful channel, plus real-surface proof sized by the
22
- tier below. TESTS ALONE NEVER PROVE DONE — a green suite means the
20
+ captured evidence: the changed behavior RUN through its real surface,
21
+ sized by the tier below, with the tests the repository keeps for it
22
+ still green. TESTS ALONE NEVER PROVE DONE — a green suite means the
23
23
  unit-level contract holds, not that the user-facing behavior works.
24
24
 
25
25
  # Tier triage (classify ONCE at bootstrap; record tier + one-line
@@ -60,8 +60,8 @@ triggered, run the reviewer loop until unconditional approval.
60
60
  Run real-surface proof yourself through the channel that faithfully
61
61
  exercises the surface; capture the artifact.
62
62
 
63
- 1. HTTP call — hit the live endpoint with `curl -i` (or a
64
- Playwright APIRequestContext); capture status line + headers +
63
+ 1. HTTP call — hit the live endpoint with `curl -i` (or an
64
+ HTTP client from js eval); capture status line + headers +
65
65
  body.
66
66
  2. Terminal / TUI - drive a real pty and prove it through the
67
67
  xterm.js web terminal (see the TUI visual QA note below). tmux
@@ -69,18 +69,21 @@ exercises the surface; capture the artifact.
69
69
  for color / layout / CJK evidence, which degrades truecolor.
70
70
  3. Browser use — in Codex, use `browser:control-in-app-browser`
71
71
  first when available and no authenticated/persistent user browser
72
- profile is required. Otherwise, or for Chrome semantics, stealth,
73
- or trace, WRITE a `playwright-core` script and run it from js eval
74
- against local Chrome (`chromium.launch({ channel: "chrome" })` /
75
- `launchPersistentContext` on a CLONED profile). Capture action
76
- log + screenshot path. Never downgrade to a non-browser surface
77
- for a browser-facing criterion. NEVER clear cookies, cache, or
78
- site data (`Network.clearBrowserCookies`, `Storage.clearCookies`,
72
+ profile is required. Otherwise drive the page with omowright
73
+ (staged in the `browser` skill; load it through that skill's
74
+ `scripts/omowright.mjs` from js eval): the owned engine
75
+ (`connectPipe` on a task-owned profile, `connectCloakProfile` for
76
+ bot-scored targets), or the attached engine
77
+ (`connectBrowserSkill()` in the user's own signed-in browser) when
78
+ the page needs their login. Capture action log + screenshot path.
79
+ Never downgrade to a non-browser surface for a browser-facing
80
+ criterion, and never launch a headless browser because the attached
81
+ one is missing — run the browser skill's onboarding script and relay
82
+ its one human step. NEVER clear cookies, cache, or site data
83
+ (`Network.clearBrowserCookies`, `Storage.clearCookies`,
79
84
  `chrome.browsingData.remove`, "clear browsing data") on the user's
80
- real/main browser profile — it wipes their logged-in state. If you
81
- need that profile's login state, clone it first (`rsync -a
82
- <profile>/ <tmp-clone>/`) and launch Chrome against
83
- the clone as the user-data-dir; run any clearing there only.
85
+ real/main browser profile, and never clone it — it wipes or
86
+ invalidates their logged-in state.
84
87
  4. Computer use — when the surface is a desktop/GUI app rather than a
85
88
  page, drive it via OS-level automation (a computer-use agent,
86
89
  AppleScript, xdotool, etc.) against the running app; capture
@@ -155,9 +158,6 @@ The criteria MUST list, upfront:
155
158
  its exact scenario: the literal command / page action / payload and
156
159
  the binary PASS/FAIL observable, plus the evidence artifact it will
157
160
  capture.
158
- - For each criterion, the failing-first proof (test id or scenario)
159
- that will be captured RED BEFORE the implementation and GREEN after.
160
- Evidence added after the green code does NOT satisfy this.
161
161
  - WHEN TO STOP, in one line: "I'll stop right away when <the exact
162
162
  observable state that ends this run>". The Stop rules bind to this
163
163
  line — the moment it holds, you stop.
@@ -193,7 +193,7 @@ Started: <ISO timestamp>
193
193
  <patterns / pitfalls / principles to remember next turn>
194
194
  ```
195
195
 
196
- Append each finding, decision, command, RED/GREEN capture, and QA
196
+ Append each finding, decision, command, test read, and QA
197
197
  artifact path the moment it happens. Update `## Now` and
198
198
  `## Todo` on every transition. Append-only — never rewrite. This notepad
199
199
  is your durable memory and it OUTLIVES the context window. After any
@@ -219,11 +219,10 @@ instead of waiting for the next pass. Step text encodes WHERE / WHY
219
219
  (which criterion it advances) / HOW / VERIFY:
220
220
  `path: <action> for <criterion> — verify by <check>`.
221
221
 
222
- GOOD pair (test-first, ordered):
223
- `foo.test.ts: Write FAILING case invalid-email→ValidationError for criterion 2 — verify by RED with assertion msg`
224
- `src/foo/bar.ts: Implement validateEmail() RFC-5322-lite for criterion 2 — verify by foo.test.ts GREEN + curl 400 body`
225
- BAD: "Implement feature" / "Fix bug" / "Add tests later" / writing
226
- production code before its failing test → rewrite.
222
+ GOOD pair (ordered):
223
+ `test/foo.test.ts: read the validateEmail cases for criterion 2 — verify by noting intent / coverage / pass in the notepad`
224
+ `src/foo/bar.ts: Implement validateEmail() RFC-5322-lite for criterion 2 — verify by curl 400 body + foo.test.ts green`
225
+ BAD: "Implement feature" / "Fix bug" / "Add tests later" → rewrite.
227
226
 
228
227
  # Finding things (lead with these, code-mode the first wave)
229
228
  Never guess from memory — locate with the right tool, and re-read before
@@ -252,51 +251,35 @@ search, absolute-path results). For research that leaves the repo —
252
251
  library/API/docs/web — delegate to the `librarian` subagent. Spawn them
253
252
  `fork_context: false` and keep doing root work while they run.
254
253
 
255
- # Execution loop (PIN → RED → GREEN → SURFACE → CLEAN)
254
+ # Execution loop (READ → CHANGE → RUN → CLEAN)
256
255
  Until every success criterion PASSES with its evidence captured:
257
256
  1. Pick next criterion → mark in_progress → update notepad `## Now`.
258
- 2. PIN + RED: when refactoring behavior whose regressions the change
259
- could hide, first pin it with a characterization test that passes on
260
- the unchanged code. Then
261
- capture the failing-first proof through the cheapest faithful
262
- channel — a unit test where a seam exists, an integration/e2e test
263
- where the behavior lives in wiring, or the criterion's real-surface
264
- scenario captured failing when no test seam exists. It must fail
265
- for the RIGHT reason (not a syntax error, not a missing import).
266
- Paste RED output into the notepad. No production code yet.
267
- TEST-ONLY TARGET (regression coverage for behavior that is already
268
- correct): there is no natural RED and no production change to make
269
- — this is the sole exception to the production-RED/GREEN steps.
270
- Substitute a mutation proof: temporarily force the exact regression
271
- each new assertion names (revert the fix commit or break the seam,
272
- never committed), capture the assertion failing, then revert the
273
- mutation and capture GREEN. An assertion that stays green under its
274
- mutation is not coverage — fix the fixture (a value equal to the
275
- default it must override proves nothing) or assert the artifact the
276
- criterion names, never an expected value re-derived from the output
277
- under test. Reverting the probe IS the GREEN; skip step 3's
278
- production change for a TEST-ONLY task and go to step 4.
279
- PROSE TARGET (prompt, SKILL.md, rule, markdown): the wording is
280
- NOT the behavior — never pin sentences, phrase presence/absence,
281
- or word/char counts. PIN only a machine-consumed value (parsed
282
- frontmatter field, a sentinel token a hook greps, the doc's JSON
283
- sample through its real validator) or one `toBe` equality between
284
- two shipped copies. A pure-prose change with no machine consumer
285
- has NO seam: ship it on review + QA-by-read, NO test — a text grep
286
- is pretend-coverage, not RED proof.
287
- 3. GREEN (skip for TEST-ONLY — reverting the mutation is GREEN): write
288
- the SMALLEST production change that flips RED→GREEN.
289
- Before GREEN work that depends on external review, PR, issue, or
290
- branch state, refresh current branch/PR/issue state and preserve existing ordering/policy;
291
- separate compatibility detection from policy changes unless the goal
292
- explicitly asks to change policy.
293
- Re-run the proof. Capture GREEN output. A GREEN far larger than the
294
- criterion implies means the proof was too coarse — split it.
295
- 4. SURFACE: run the real-surface proof the criterion named (channel
296
- table above; auxiliary surface for CLI- or data-shaped criteria),
297
- end-to-end, yourself. If the RED proof was the scenario itself,
298
- re-run it now and capture it passing. Paste the artifact path into
299
- the notepad.
257
+ 2. READ what already proves the area BEFORE touching it. Existing
258
+ tests are the behavior of record: note in the notepad whether they
259
+ encode the intended behavior, cover the path you change, and pass.
260
+ One WRONG before your change is a FINDING to report — NEVER edit a
261
+ test green. A bug: reproduce it first and capture the failure. A
262
+ refactor: the existing tests are green on the unchanged code first.
263
+ 3. CHANGE: the SMALLEST production change that meets the criterion;
264
+ update the tests your change makes stale. Add a test ONLY when
265
+ BOTH hold: the repository keeps tests for this behavior AND a
266
+ regression would otherwise pass unnoticed by the run and the
267
+ existing tests — sized like its neighbors, one case per stated
268
+ behavior, failing when that behavior breaks. A test that restates
269
+ the change (a constant, a string, a rename, a call) is NOT evidence;
270
+ the run is. Coverage-only work (no production change): break the
271
+ behavior each new assertion names, capture it failing, restore — an
272
+ assertion that stays green under its mutation is not coverage.
273
+ PROSE TARGET (prompt, SKILL.md, rule, markdown): the wording is NOT
274
+ the behavior — pin only a machine-consumed value (parsed field,
275
+ sentinel a hook greps, a JSON sample through its validator) or one
276
+ `toBe` equality between shipped copies; otherwise review + QA-by-read,
277
+ NO test. Before a change that depends on review, PR, issue, or
278
+ branch state, refresh that state and preserve existing ordering/policy.
279
+ 4. RUN: the real-surface scenario the criterion named (channel table
280
+ above; auxiliary surface for CLI- or data-shaped criteria), end to
281
+ end, yourself, plus the step-2 tests; a reproduction now passes.
282
+ Paste the artifact path into the notepad.
300
283
  5. CLEANUP (PAIRED — NEVER SKIP): the moment a QA scenario spawns any
301
284
  resource, register its teardown as its own todo (e.g.
302
285
  `cleanup: kill server pid for criterion 2 — verify kill -0 fails`).
@@ -304,7 +287,7 @@ Until every success criterion PASSES with its evidence captured:
304
287
  before this step completes:
305
288
  server PIDs (`kill <pid>`; verify `kill -0` fails), `tmux` sessions
306
289
  (`tmux kill-session -t ulw-qa-<criterion>`; verify with `tmux ls`),
307
- browser / Playwright contexts (`.close()`), containers
290
+ browsers / sessions (`browser.close()` / `session.stop()`), containers
308
291
  (`docker rm -f`), bound ports (`lsof -i :<port>` empty), temp
309
292
  sockets / files / dirs (`rm -rf` the `mktemp` paths), QA-only env
310
293
  vars. Append a one-line cleanup receipt to the notepad next to the
@@ -322,8 +305,8 @@ Until every success criterion PASSES with its evidence captured:
322
305
  message. Record PASS/FAIL inline with the evidence paths AND the
323
306
  cleanup receipt. Loop until all PASS.
324
307
 
325
- Within a step, follow Finding things; NEVER parallelise RED and GREEN of
326
- the same criterion.
308
+ Within a step, follow Finding things; READ before CHANGE, never in
309
+ parallel with it.
327
310
 
328
311
  # Waiting discipline (a poll costs a full model round)
329
312
  Every status check you issue as a tool call replays the entire
@@ -435,8 +418,8 @@ When triggered, follow this procedure (NON-NEGOTIABLE):
435
418
  2-attempt stop rule below) — do not loop further.
436
419
 
437
420
  # Commits
438
- Commit frequently: one atomic commit per verified increment (RED→GREEN
439
- + its evidence), never one end-of-run omnibus; each commit builds +
421
+ Commit frequently: one atomic commit per verified increment (change +
422
+ its evidence), never one end-of-run omnibus; each commit builds +
440
423
  tests green on its own; no WIP on the final branch.
441
424
  BEFORE composing each message, read the history and mimic it: run
442
425
  `git log --oneline -20` plus `git log -5 -- <touched paths>` and match
@@ -449,20 +432,11 @@ convention. If a plan file exists, final commit footer:
449
432
  commits this session — then stage + draft the message instead.
450
433
 
451
434
  # Constraints
452
- - Every behavior change needs a failing-first proof captured BEFORE
453
- the production change, through the cheapest faithful channel (unit
454
- test at a seam; integration/e2e in wiring; the real-surface scenario
455
- when no test seam exists). If you typed production code first, STOP,
456
- revert, capture the proof failing, then redo the change. Exempt
457
- only: pure formatting, comment-only edits, dependency bumps with no
458
- behavior delta, rename-only moves — justify each in `## Findings`.
459
- - A test that cannot fail for the regression it names is NOT
460
- evidence: mock-call assertions, pinned constants, a fixture equal
461
- to the default it must override, an expected value re-derived from
462
- the output under test. Prefer a real-surface proof with no new
463
- test over a tautological one.
464
- - Refactors: characterization tests pinning current observable
465
- behavior FIRST, green against the old code, green throughout.
435
+ - Every behavior change is PROVEN BY ITS RUN on the real surface, with
436
+ the tests the repository keeps for it green. A test that cannot fail
437
+ for the regression it names is NOT evidence: mock-call assertions,
438
+ pinned constants, a fixture equal to the default it must override,
439
+ an expected value re-derived from the output under test.
466
440
  - Smallest correct change. No drive-by refactors.
467
441
  - Never suppress lints / errors / test failures. Never delete, skip,
468
442
  `.only`, `.skip`, `xfail`, or comment out tests to green the suite.
@@ -471,8 +445,8 @@ commits this session — then stage + draft the message instead.
471
445
  # Output discipline
472
446
  - First line literally: `ULTRAWORK MODE ENABLED!`
473
447
  - After bootstrap: 1-2 paragraph plan summary + notepad path.
474
- - During execution: surface only state changes (RED captured, GREEN
475
- captured, scenario PASS/FAIL with evidence paths, reviewer verdict).
448
+ - During execution: surface only state changes (existing tests read,
449
+ scenario PASS/FAIL with evidence paths, reviewer verdict).
476
450
  - Final message: outcome + success-criteria checklist with evidence
477
451
  refs + notepad path + reviewer approval (if gate triggered) + commit
478
452
  list (`<sha> <subject>`). No file-by-file changelog unless asked.
@@ -164,13 +164,13 @@ Sizing is a two-branch decision made per checkbox, before dispatch:
164
164
  Each sub-task message must include:
165
165
 
166
166
  1. Goal and exact files or directories in scope.
167
- 2. When the task touches existing behavior: a baseline characterization test, written first, that pins current observable behavior and passes on the unchanged code (exact inputs, exact observable, exact assertion). Then the failing-first proof for the new behavior before production changes — a unit test where a seam exists, otherwise the sub-task's Manual-QA scenario captured failing. A test that mirrors its implementation (mock-call assertions, pinned constants) is not evidence.
167
+ 2. The tests already covering the touched behavior, READ before any edit as the behavior of record (intent, coverage, pass); a bug's reproduction captured before the fix. A new test ONLY where the repository keeps tests for this behavior AND a regression would otherwise pass unnoticed by the sub-task's Manual-QA scenario — never one that mirrors its implementation (mock-call assertions, pinned constants) or restates the change.
168
168
  3. Implementation constraints from the plan and project rules.
169
169
  4. Automated verification commands to run.
170
- 5. One Manual-QA channel, named with the exact tool and exact invocation (the literal `curl`, `send-keys`, `browser:control-in-app-browser` action, `page.click`, payload, selectors, and the binary observable that decides PASS/FAIL), not "verify it works". A LIGHT checkbox needs one real-surface proof of its deliverable, and auxiliary surfaces (CLI stdout, DB state diff, parsed config dump) are first-class when the surface is CLI- or data-shaped:
170
+ 5. One Manual-QA channel, named with the exact tool and exact invocation (the literal `curl`, `send-keys`, `page.click` / `session.click`, payload, selectors, and the binary observable that decides PASS/FAIL), not "verify it works". A LIGHT checkbox needs one real-surface proof of its deliverable, and auxiliary surfaces (CLI stdout, DB state diff, parsed config dump) are first-class when the surface is CLI- or data-shaped:
171
171
  - HTTP call: `curl -i` against the live endpoint.
172
172
  - Terminal / TUI: drive a real pty; `tmux send-keys` is fine for a boot/behavior smoke, but color/layout/CJK evidence goes through the xterm.js web terminal below, NEVER `tmux capture-pane`.
173
- - Browser use: from js eval, use `new Bun.WebView()` on Bun >= 1.4 (macOS default; Linux/Windows need Chrome/Chromium/Edge). Otherwise, or for Chrome semantics, stealth, trace, or auth, write and run a `playwright-core` script against local Chrome (`channel: "chrome"`; persistent context on a CLONED profile). Codex: `browser:control-in-app-browser` for ordinary page control.
173
+ - Browser use: omowright from js eval (staged in the `browser` skill) — the owned engine (`connectPipe` on a task-owned profile, `connectCloakProfile` for bot-scored targets) for unauthenticated pages, the attached engine (`connectBrowserSkill()` in the user's signed-in browser) when the page needs their login; never a clone of or a launch against the live profile.
174
174
  - Computer use: OS-level GUI automation against the running desktop app when the surface is not a page.
175
175
  - TUI visual evidence: when a TUI claim needs visual QA or PR proof, run `bun script/qa/web-terminal-visual-qa.mjs --command "<cmd>" --input "{Enter}" --evidence-dir <dir>` (real pty rendered through xterm.js in Chrome) and attach `terminal.png` plus `metadata.json`.
176
176
  6. The adversarial classes that apply to this sub-task (from the 9 ultraqa classes) and how each is probed.
@@ -243,7 +243,7 @@ When all top-level checkboxes in `## TODOs` and `## Final Verification Wave` are
243
243
 
244
244
  ## Hard rules
245
245
 
246
- - No production change before a failing-first proof exists (unit test at a seam, otherwise the failing Manual-QA scenario), and no change to existing behavior before a baseline characterization test pins the current behavior and passes on the unchanged code.
246
+ - No production change before the tests covering that behavior were read and a bug's reproduction captured; existing tests are green on the unchanged code first, and one that contradicts the intent is a FINDING, never edited green.
247
247
  - No `--dry-run` as completion evidence.
248
248
  - No tests-only completion claim. A Manual-QA artifact is required.
249
249
  - **NO DIRECT IMPLEMENTATION BY THE ORCHESTRATOR.** Root NEVER edits product files, writes tests, or runs QA itself — a spawned worker does.
@@ -39,8 +39,7 @@ Every criterion carries, at definition time, not after the work:
39
39
 
40
40
  - a binary pass condition ("returns 200 and the body matches the schema", never "works correctly");
41
41
  - the exact scenario: the literal command, request, page action, or payload that will prove it;
42
- - the evidence artifact it will capture: transcript, status plus body, screenshot path, diff, parsed dump;
43
- - the failing-first proof (test id or scenario) that will be captured RED before implementation.
42
+ - the evidence artifact it will capture: transcript, status plus body, screenshot path, diff, parsed dump.
44
43
 
45
44
  A criterion that cannot fail is not a criterion. If no input could make the scenario fail, it measures nothing; rewrite it until failure is possible.
46
45
 
@@ -50,7 +49,7 @@ Prefer numbers that represent real success over decorative precision. A threshol
50
49
 
51
50
  | Domain | Quantify as |
52
51
  | --- | --- |
53
- | Bug fix | reproduction first, fix second: the failing case captured RED, then the same validator green |
52
+ | Bug fix | reproduction first, fix second: the failing case captured before, the same case passing after |
54
53
  | Tests | the exact command and required pass condition, plus run count for flake-sensitive suites |
55
54
  | Performance | metric, target threshold, measurement method, and run count ("p95 under 250ms across 3 consecutive local runs") |
56
55
  | Quality work | the observable acceptance bar: lint, typecheck, and test pass; reviewed examples; a user-approved artifact |
@@ -18,9 +18,9 @@ Audit each pass, fail, block, steering change, and checkpoint in `.omo/ulw-loop/
18
18
  ## Manual-QA channels
19
19
  Run each criterion's real-surface proof yourself through the channel that faithfully exercises it; capture the artifact before recording PASS.
20
20
 
21
- 1. **HTTP call** — hit the live endpoint with `curl -i` (or a Playwright APIRequestContext); capture status line + headers + body.
21
+ 1. **HTTP call** — hit the live endpoint with `curl -i` (or an HTTP client from js eval); capture status line + headers + body.
22
22
  2. **Terminal / TUI** - prove it through the xterm.js web terminal; tmux `send-keys` is fine for a boot smoke, but NEVER `tmux capture-pane` for color/layout/CJK evidence (it degrades truecolor).
23
- 3. **Browser use** — in Codex, prefer `browser:control-in-app-browser`. Otherwise, or for Chrome semantics, stealth, trace, or auth, WRITE a `playwright-core` script and run it from js eval against local Chrome (`channel: "chrome"`; persistent context on a CLONED profile, never the live one). Capture action log + screenshot path. Never downgrade a browser-facing criterion.
23
+ 3. **Browser use** — in Codex, prefer `browser:control-in-app-browser`. Otherwise omowright from js eval (staged in the `browser` skill): the owned engine (`connectPipe` on a task-owned profile, `connectCloakProfile` for bot-scored targets) for unauthenticated pages, the attached engine (`connectBrowserSkill()` in the user's signed-in browser) when the page needs their login — never a clone of, or a launch against, the live profile. Capture action log + screenshot path. Never downgrade a browser-facing criterion.
24
24
  4. **Computer use** — for desktop/GUI apps, drive the running app via OS automation (computer-use, AppleScript, xdotool, etc.); capture action log + screenshot.
25
25
 
26
26
  For TUI visual QA (mandatory when a PR or review must inspect the terminal screen),
@@ -73,7 +73,7 @@ Before any reviewer sees an image, verify each capture yourself: the file signat
73
73
  ### Web
74
74
 
75
75
  1. Capture a REFERENCE image: the user's mock/target, generated page snapshot, Figma export, source-site capture, or known-good baseline. Save as PNG. If the user provided overview text or annotations, save them next to the image and treat them as part of the reference packet.
76
- 2. Capture the ACTUAL rendered screenshot at the reference viewport. Use two browser tiers from js eval: (1) `new Bun.WebView()` on Bun >= 1.4 (macOS default; Linux/Windows need installed Chrome/Chromium/Edge); (2) otherwise, or for Chrome semantics, stealth, trace, or authenticated profiles, WRITE a `playwright-core` script and run it from the kernel against local Chrome (`chromium.launch({ channel: "chrome" })` / `launchPersistentContext` on a CLONED profile, never the live one). In Codex, prefer `browser:control-in-app-browser` for ordinary captures. Save PNG and return its path; close the browser context. See `$SKILL_DIR/references/browser-setup.md` for fixed-viewport examples and prerequisites.
76
+ 2. Capture the ACTUAL rendered screenshot at the reference viewport with omowright from js eval (the library is staged in the `browser` skill): the owned engine (`connectPipe` on a task-owned profile, viewport pinned with `emulate`, then `page.screenshot()`) for anything unauthenticated, or the attached engine (`connectBrowserSkill()` → `session.screenshot()`) when the page needs the user's login — never a clone of, or a launch against, the user's live profile. Save PNG and return its path; close the browser or stop the session. See `$SKILL_DIR/references/browser-setup.md` for fixed-viewport examples and prerequisites.
77
77
  3. Run the diff and keep the JSON:
78
78
 
79
79
  ```
@@ -1,75 +1,75 @@
1
1
  # Browser setup (Web capture)
2
2
 
3
- Use the js-eval kernel for both tiers. In Codex, prefer
4
- `browser:control-in-app-browser` for ordinary captures.
3
+ Capture with omowright from the js-eval kernel. The library is staged inside the `browser` skill;
4
+ load it once per session:
5
5
 
6
- ## Tier 1: Bun.WebView
6
+ ```js
7
+ const { loadOmowright } = await import("<browser-skill-root>/scripts/omowright.mjs")
8
+ const { omowright } = await loadOmowright()
9
+ ```
10
+
11
+ ## Owned engine (default for QA)
7
12
 
8
- On a Bun >= 1.4 kernel, `new Bun.WebView()` is the macOS default (system
9
- WebKit, no browser download). Linux/Windows require `backend: "chrome"`
10
- and installed Chrome/Chromium/Edge. WebView is headless; WebKit has no CDP,
11
- and `type()` emits no keyboard events. Use tier 2 when these differences matter.
13
+ A browser your code launches, with a task-owned profile, pinned viewport and no user state.
14
+ `connectPipe` opens no listening port and reaps the process on `close()`.
12
15
 
13
16
  ```js
14
17
  // js-eval cell; url and pngPath belong to this QA run.
15
- const view = new Bun.WebView({ width: 1280, height: 720 })
18
+ const { mkdtempSync, rmSync } = await import("node:fs")
19
+ const profile = mkdtempSync(`${(await import("node:os")).tmpdir()}/visual-qa-`)
20
+ const browser = await omowright.connectPipe({
21
+ browserPath: chromeBinary, // installed Chrome, Chromium, CloakBrowser or chrome-headless-shell
22
+ browserArgs: ["--headless", "--no-first-run", `--user-data-dir=${profile}`],
23
+ storageRoot: profile,
24
+ })
16
25
  try {
17
- await view.navigate(url)
18
- await Bun.write(pngPath, await view.screenshot())
26
+ const page = await browser.newTab("about:blank")
27
+ await omowright.emulate(page, { width: 1280, height: 720, deviceScaleFactor: 1, mobile: false, hasTouch: false })
28
+ await page.goto(url, { waitUntil: "load" })
29
+ await Bun.write(pngPath, await page.screenshot())
19
30
  console.log(pngPath)
20
31
  } finally {
21
- view[Symbol.dispose]()
32
+ await browser.close()
33
+ rmSync(profile, { recursive: true, force: true })
22
34
  }
23
35
  ```
24
36
 
25
- ## Tier 2: playwright-core with local Chrome
37
+ Chrome must already be installed; report an absent executable rather than downloading a managed
38
+ browser. For bot-scored or WAF targets use `connectCloakProfile({ profileDir })` (CloakBrowser
39
+ with a pinned fingerprint seed) — the `browser` skill's `references/owned-engine/README.md`
40
+ covers it.
41
+
42
+ ## Attached engine (authenticated pages)
26
43
 
27
- For other kernels, real-Chrome semantics, stealth, traces, or authenticated
28
- profiles, WRITE a script beside the project's installed `playwright-core`
29
- dependency and execute it from js eval. The user installs `playwright-core`
30
- once with `bun add playwright-core` if needed; Chrome must already exist.
31
- No managed browser download is needed.
44
+ When the capture needs the user's login, drive the browser they are signed into instead of
45
+ cloning its profile:
32
46
 
33
47
  ```js
34
- // capture.mjs
35
- import { chromium } from "playwright-core"
36
- const [url, pngPath] = process.argv.slice(2)
37
- const browser = await chromium.launch({ channel: "chrome", headless: true })
48
+ const session = await omowright.connectBrowserSkill({ name: "visual-qa capture", focused: false })
38
49
  try {
39
- const page = await browser.newPage({ viewport: { width: 1280, height: 720 }, deviceScaleFactor: 1 })
40
- await page.goto(url, { waitUntil: "load", timeout: 30000 })
41
- await page.screenshot({ path: pngPath })
42
- console.log(pngPath)
50
+ await session.navigate(url, { waitUntil: "load" })
51
+ await session.resize(1280, 720)
52
+ const shot = await session.screenshot() // { buffer, width, height }
53
+ await Bun.write(pngPath, shot.buffer)
43
54
  } finally {
44
- await browser.close()
55
+ await session.stop()
45
56
  }
46
57
  ```
47
58
 
48
- Run from the js-eval kernel (also works on a Node kernel):
49
-
50
- ```js
51
- const { promisify } = await import("node:util")
52
- const { execFile } = await import("node:child_process")
53
- const result = await promisify(execFile)("node", [scriptPath, url, pngPath], { timeout: 45000 })
54
- console.log(result.stdout)
55
- ```
56
-
57
- For auth, CLONE the user-data directory to a private task-owned directory
58
- and use `chromium.launchPersistentContext(clonePath, { channel: "chrome" })`.
59
- NEVER launch against or clear cookies/cache/site data from the live profile.
60
- Close the context and remove only the clone after QA. For script-based stealth,
61
- read the ultimate-browsing skill's `references/chrome-stealth.md`.
59
+ NEVER launch anything against, or clear cookies/cache/site data from, the user's live profile;
60
+ the attached engine is the only sanctioned way to a signed-in page. If no extension is connected,
61
+ run the `browser` skill's `scripts/browser-install.mjs`, relay its one human step, and wait — do
62
+ not fall back to the owned engine for an authenticated criterion.
62
63
 
63
64
  ## Capture a screenshot at a fixed viewport
64
65
 
65
- Match CSS viewport AND PNG dimensions: WebKit follows native display scale
66
- (a 1280x720 viewport can yield 2560x1440 pixels). Use matching reference
67
- captures or Chrome's `deviceScaleFactor`, not resizing to force a pass.
68
- Wait for the specific page state, not a sleep, then compare:
66
+ Match CSS viewport AND PNG dimensions: pin `deviceScaleFactor` through `emulate` (owned) or
67
+ `resize` (attached) instead of resizing the PNG to force a pass. Wait for the specific page state
68
+ (a locator, a `waitForURL`, a `createNetworkSnoop(page).waitFor(...)`), not a sleep, then compare:
69
69
 
70
70
  ```sh
71
71
  node "$SKILL_DIR/scripts/visual-qa.mjs" image-diff reference.png actual.png
72
72
  ```
73
73
 
74
- Inspect `dimensionsMatch` and `diffRatio`, then inspect the image. Close every
75
- WebView/browser context and the fixture server, even on a failed capture.
74
+ Inspect `dimensionsMatch` and `diffRatio`, then inspect the image. Close every browser and session
75
+ and the fixture server, even on a failed capture; remove the owned profile in the same `finally`.
@@ -98,7 +98,7 @@ const ulwExecuteCodexCompletion = `When all top-level checkboxes in \`## TODOs\`
98
98
  5. Remove or mark the Boulder work as completed.
99
99
  6. Print an \`ORCHESTRATION COMPLETE\` block with the plan path, verification commands, artifacts, and cleanup receipts.`;
100
100
 
101
- const ulwExecuteOriginalHardRule = `- No production change before a failing-first proof exists (unit test at a seam, otherwise the failing Manual-QA scenario), and no change to existing behavior before a baseline characterization test pins the current behavior and passes on the unchanged code.
101
+ const ulwExecuteOriginalHardRule = `- No production change before the tests covering that behavior were read and a bug's reproduction captured; existing tests are green on the unchanged code first, and one that contradicts the intent is a FINDING, never edited green.
102
102
  - No \`--dry-run\` as completion evidence.
103
103
  - No tests-only completion claim. A Manual-QA artifact is required.
104
104
  - **NO DIRECT IMPLEMENTATION BY THE ORCHESTRATOR.** Root NEVER edits product files, writes tests, or runs QA itself — a spawned worker does.
@@ -107,7 +107,7 @@ const ulwExecuteOriginalHardRule = `- No production change before a failing-firs
107
107
  - No unprefixed session ids in Boulder state. Sessions are always recorded as \`codex:<session_id>\`.
108
108
  - No stale-memory execution. The plan and ledger are the durable source of truth.`;
109
109
 
110
- const ulwExecuteCodexHardRule = `- No production change before a failing-first proof exists (unit test at a seam, otherwise the failing Manual-QA scenario), and no change to existing behavior before a baseline characterization test pins the current behavior and passes on the unchanged code.
110
+ const ulwExecuteCodexHardRule = `- No production change before the tests covering that behavior were read and a bug's reproduction captured; existing tests are green on the unchanged code first, and one that contradicts the intent is a FINDING, never edited green.
111
111
  - No \`--dry-run\` as completion evidence.
112
112
  - No tests-only completion claim. A Manual-QA artifact is required.
113
113
  - **NO DIRECT IMPLEMENTATION BY THE ORCHESTRATOR.** Root NEVER edits product files, writes tests, or runs QA itself — a spawned worker does.
@@ -1,5 +1,5 @@
1
1
  #!/usr/bin/env node
2
- // omo-codex-install:e05e108254a84b0b56604b57a1db7fc04d476de568cf6860040906461d5b089d:65bd0688ebae929257167637f75778a4acf745e8877e0c9df3abf5012a6063fb
2
+ // omo-codex-install:e7f8b8a7af0a5bc6257d28f1ef9ead291ad7ea22515c64d7fd93951afa62eab8:5e264434333c29356d4fe13f7ad2cf2ad0942eb5ca12b94e77105e35c913077c
3
3
  var __esm = (fn, res, err) => () => {
4
4
  if (fn)
5
5
  try {
@@ -9984,7 +9984,7 @@ var package_default;
9984
9984
  var init_package = __esm(() => {
9985
9985
  package_default = {
9986
9986
  name: "@oh-my-opencode/omo-codex",
9987
- version: "5.0.0-beta.85",
9987
+ version: "5.0.0-beta.87",
9988
9988
  type: "module",
9989
9989
  private: true,
9990
9990
  description: "Codex harness adapter for oh-my-openagent. Vendored Codex plugin namespace (omo) + TypeScript installer + telemetry.",
@@ -16735,6 +16735,8 @@ function isRecord5(value) {
16735
16735
  }
16736
16736
  var OmoModelProfileInputSchema = object({
16737
16737
  display_name: string2().optional(),
16738
+ family: _enum(["daily", "geeky"]).optional(),
16739
+ tier: _enum(["normal", "heavy"]).optional(),
16738
16740
  models: array(union([string2(), OmoFallbackModelObjectSchema])).optional()
16739
16741
  }).strict();
16740
16742
  var OmoModelProfileSchema = preprocess((value) => isRecord5(value) ? normalizeLegacyModelFields(value) : value, OmoModelProfileInputSchema);