muse-crew 0.14.8 → 0.14.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/decisions/workflow-core.md +14 -7
- package/docs/ooda-report.md +21 -4
- package/lib/AGENTS.md +2 -2
- package/lib/read-ooda-verdict.js +61 -86
- package/lib/see-act.js +616 -6
- package/lib/test-worktree-backend.sh +42 -0
- package/lib/worktree-lifecycle.sh +64 -4
- package/package.json +1 -1
- package/workflows/bugfix.js +66 -17
- package/workflows/chore.js +63 -15
- package/workflows/docs.js +11 -13
- package/workflows/standard.js +66 -17
|
@@ -52,17 +52,24 @@ Applies to: standard, bugfix, chore.
|
|
|
52
52
|
<a id="closeout-envelope"></a>
|
|
53
53
|
## Closeout transport envelope
|
|
54
54
|
|
|
55
|
-
Invariant: the work
|
|
55
|
+
Invariant: the schema-less work-agent courier returns prose; the workflow
|
|
56
|
+
consumes the return as a plain string. Nothing parses a JSON envelope —
|
|
57
|
+
the prompt names none (blocker 43, 2026-09-22: the old "Return your work as
|
|
58
|
+
JSON in exactly this shape: {...}" invitation drew brace-shaped prose into
|
|
59
|
+
a channel where the runtime's JSON-candidate scan threw, and three
|
|
60
|
+
consecutive Integrate failures parked room #28's J1 at the retry cap; the
|
|
61
|
+
second REVIEW pass then reworded the retry trailer's "discarded" why off
|
|
62
|
+
"machine-read as JSON" to the true event — {...}-shaped fragments the scan
|
|
63
|
+
could not parse — since the old framing was an implicit JSON invitation).
|
|
64
|
+
The verdict is extracted mechanically by extractVerdict — never by an agent.
|
|
56
65
|
|
|
57
66
|
Applies to: standard, bugfix, chore.
|
|
58
67
|
|
|
59
68
|
```
|
|
60
|
-
// Work agent returns the
|
|
61
|
-
//
|
|
62
|
-
//
|
|
63
|
-
// agent
|
|
64
|
-
// string. The verdict is still extracted deterministically from the report
|
|
65
|
-
// text by extractVerdict below — never by an agent.
|
|
69
|
+
// Work agent returns prose on the schema-less courier; the workflow
|
|
70
|
+
// consumes the return as a plain string. The verdict is still extracted
|
|
71
|
+
// deterministically from the report text by extractVerdict — never by
|
|
72
|
+
// an agent.
|
|
66
73
|
```
|
|
67
74
|
|
|
68
75
|
<a id="extract-verdict-contract"></a>
|
package/docs/ooda-report.md
CHANGED
|
@@ -120,6 +120,22 @@ into the phase dir automatically:
|
|
|
120
120
|
No frame can be lost: Reproduce, QA, and exploratory hunting all funnel through
|
|
121
121
|
the same wrapper.
|
|
122
122
|
|
|
123
|
+
## Session protocol (blocker 41, 2026-09-22)
|
|
124
|
+
|
|
125
|
+
The one-shot driver launches a fresh browser per invocation, so it cannot
|
|
126
|
+
drive multi-step in-page flows (open a deck, then click Study — the second
|
|
127
|
+
step needs the first step's living page state). The session protocol holds
|
|
128
|
+
ONE Chromium + page alive across invocations:
|
|
129
|
+
|
|
130
|
+
- `session-start --name <name> --url <url> [--session-dir <dir>] [--log <ooda-log.jsonl>] [--viewport desktop|mobile] [--timeout-ms <n>] [--idle-ttl-ms <n>]` — launches Chromium once (same loopback-proxy stack), navigates, spawns a detached daemon holding the browser. Prints one JSON line `{ok:true, session, session_dir, control_port}`. The daemon exits after `--idle-ttl-ms` (default 10 min) with no act.
|
|
131
|
+
- `session-act --name <name> --attempt <id> [--session-dir <dir>] [--timeout-ms <n>] <aria|shot|click|scroll|type> [args]` — performs ONE action on the living page, appends one JSON line to the session's ooda-log (the `--log` path from session-start — point it at the phase dir's `ooda-log.jsonl` so the guard reads it) in the exact append-ooda-step schema `{step, attempt, action, args, exit, screenshot, observation}`, prints the action's JSON result, exits with the action's exit code (0/2/3, same meanings as one-shot).
|
|
132
|
+
- `session-end --name <name> [--session-dir <dir>]` — closes the browser, prints `{ok:true, session, log, steps}`. The session dir (and its ooda-log) is kept as evidence.
|
|
133
|
+
|
|
134
|
+
The control server binds 127.0.0.1 only. Session state lives in
|
|
135
|
+
`<session-dir>/<name>/session.json` (atomic writes). A corrupt ooda-log
|
|
136
|
+
fails the act loudly BEFORE it performs — lost evidence fails loud, never
|
|
137
|
+
silently. The one-shot invocation shape is untouched by the protocol.
|
|
138
|
+
|
|
123
139
|
## Workflow integration
|
|
124
140
|
|
|
125
141
|
- **Standard QA** (artifact): archive to `task-evidence/<task>/postchange/`,
|
|
@@ -151,10 +167,11 @@ never touches the clock or randomness:
|
|
|
151
167
|
visual_loop_unavailable}` — the record exists, parses, carries a `verdict`
|
|
152
168
|
field, agrees with the prose expectation, and a FAIL carries a non-empty
|
|
153
169
|
reason. `visual_loop_unavailable` reports whether the experiential browser
|
|
154
|
-
loop could not run
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
170
|
+
loop could not run, read from the ooda-log only (never from verdict prose,
|
|
171
|
+
blocker 44 2026-09-22): a see-act browser step exited 3 with `NOT
|
|
172
|
+
POSSIBLE` (unresolvable tooling), and/or zero completed browser steps in
|
|
173
|
+
the log — the loop left no evidence it ran. `missing_evidence` stays the
|
|
174
|
+
designed home for reach-gap documentation, but prose never trips the guard.
|
|
158
175
|
- Exit 2, `{ok:false, code}` — `missing` (no verdict.json), `corrupt`
|
|
159
176
|
(unparseable or no verdict field), `contradiction` (record disagrees with
|
|
160
177
|
the prose line), `no_reason` (FAIL with no machine-readable reason).
|
package/lib/AGENTS.md
CHANGED
|
@@ -13,7 +13,7 @@ Shipped library: ESM JavaScript CLIs and import-safe modules, shell scripts for
|
|
|
13
13
|
- `schema.sql` — the crew-owned state schema: projects, tasks, poll_state, config, agent_sessions, events. Vocabularies enforced by CHECK constraints; `rejected` is a valid event type (the 2026-09-11 crash was a stored session whose event was rejected). Column names match the historical dashboard tables for a verbatim migration.
|
|
14
14
|
- `crew-release.sh` — immutable release manager: deploy, rollback, prune
|
|
15
15
|
- `merge-lock.sh` — serialized merge lock for concurrent agents: time-based holder lease (bug 2fc8f52f — an unexpired lease is held regardless of process liveness; only an expired lease may be broken). Requires both `CREW_REPO` and `CREW_HOME` (fail closed: BLOCKED, exit 2 when either is unset). Lock file is key=value: task_id, opaque holder identity (never a PID), acquired_at epoch, lease_seconds (default 600, override via MERGE_LOCK_LEASE_SECONDS). acquire/refresh/release/status/force-release; holder-only refresh and release; every op appends to $CREW_HOME/.merge-lock.log. Stale-lease breaks serialize on a sidecar `$LOCK_FILE.flock` with an in-critical-section lease re-read (R-B1, 2026-09-21); refresh rewrites via temp-file + atomic rename so readers never see a torn file, and takes the same sidecar flock with an identity-only re-check inside the critical section — a reclaim always changes `task_id`, so identity alone closes the clobber (no expiry check on refresh: a long build that outran the lease legitimately revives its lock); release takes the same flock with the identity re-check inside (review pass 2, 2026-09-21) so a release can never `rm` a reclaimer's fresh lock.
|
|
16
|
-
- `worktree-lifecycle.sh` — the worktree lifecycle seam: prepare/cleanup/inspect/integrate/verify-merge/status/post-deploy/terminal-cleanup/refresh-lock/lock-status over git worktrees (the integration target is the repo's current checkout — a branch name, or HEAD when detached — resolved by `integration_target()`; nothing in the lifecycle ever checks out a branch. `integrate` merges into the target, reconciles `origin/<target>` under the merge lock after the task merge — canary `a6d8b0c8`, 2026-09-11 — then pushes inline via `do_push` (no agent round-trip; 2026-09-19 REVIEW collapsed the old STEP-2 agent run, which added only a lock refresh). `do_push`: PUSHED (the refspec is always explicit — a branch pushes as `<branch>`, a detached HEAD as `HEAD:<destination>` where the destination is the remote's default branch from `resolve_push_destination()`; 2026-09-19 detached-HEAD audit REDO) / NO_REMOTE_PUSH (no origin — fail-soft, 2026-09-19 REVIEW: the old unconditional push exited 128 and mislabeled it); an unresolvable detached destination fails closed (never a silent skip, never a refusal of a knowable push); on push failure one fetch+merge retry while the lock is held, then fail closed without asserting a cause (never force-push). PUSH_SKIPPED survives only for the nothing-merged R5 path (no merge record + no lock held). Merged-but-unpushed retry recovery (blocker 35, 2026-09-21): identity-based, not ancestry-based — `current_branch_merge()` finds the newest merge on live HEAD's first-parent chain whose branch-side parent (^2) is the branch's CURRENT tip (the recorded merge, a do_push reconcile merge above it, or the manual R5 merge); that merge takes the lock and pushes instead of reporting MERGED_EMPTY. A record with no such merge means the branch moved since the merge — STALE_MERGE fails closed (never pushed, never MERGED_EMPTY). Branch gone (merge-lease reclaim): the recorded merge on the line is still the deliverable (no rework could have moved it). `push-target` proves the same identity before the ERROR-after-MERGED retry push. `CREW_STAGED_BASE` guard (ferried from get-provenance): the line must descend from the staged publish base — STAGED_BASE_MISMATCH fails closed before the merge (room #21 J3 shape). `verify-merge` always re-resolves the target from the current checkout and checks the recorded commit against the LIVE tip — never the record's word for the tip (2026-09-19 REVIEW: the old read-back compared the record to itself for target=HEAD). A detached HEAD skips the inline reconcile loudly (NO_REMOTE_RECONCILE) — the inline reconcile needs a static branch target, so on detached the destination is resolved explicitly in do_push, which reconciles lazily on push failure (2026-09-19 detached-HEAD audit REDO; the old "local by design" rationale is rejected); `push-target` is the thin R5 wrapper over `do_push`: refreshes the lock, distinguishes nothing-merged (no record + no lock → loud skip) from lock-lost (fail closed). (`.worktrees/<id>`, branch `task/<id>`). Branch resolution order: the crew registry (`<repo>/.worktrees/.registry/<id>`), then the canonical `task/<full-id>` ref, then a `task/<id-prefix>` ref matched by strict prefix enumeration (task 4e1a1bba — a Build agent may create the branch with raw git from an abbreviated id, bypassing prepare; prefix names are tolerated, never created; ambiguous prefixes fail closed). `resolve-branch` prints the resolved branch for agent-side one-liners. Nothing reconstructs the branch name — every command resolves it. Prepare fails closed on a dirty integration target; cleanup is forgiving. `terminal-cleanup` is the run's last act at every park/fail boundary (called from `parkTask` in standard/bugfix/chore): releases the merge lock unconditionally and reclaims the worktree+branch only when the task branch is fully merged into the integration target — unmerged work is preserved for the human by design, and a dirty worktree is reported, never force-removed. `post-deploy` reports worktree removal honestly (a lying "removed" echo hid real leftovers — canary run 9, 2026-09-12). Requires both `CREW_REPO` and `CREW_HOME` (fail closed: BLOCKED, exit 2 when either is unset); `CREW_REPO` is exported so internal merge-lock.sh calls inherit the repo being worked on; `LIB_DIR` defaults to `$CREW_HOME/lib` (`CREW_LIB` override). `lock-status` reports the merge-lock state explicitly (`UNLOCKED`, or key=value: locked=true, task_id, holder, acquired_at, lease_seconds, age_seconds, remaining_seconds — always exit 0) so Publish can distinguish an empty-diff Integrate (no lock taken) from a refresh failure. post-deploy's `.worktrees/` guard is add-then-reset, never an all-negative pathspec: `git add -A -- ':!.worktrees/'` still exits 1 on git 2.43.0 against the crew's own gitignored non-empty `.worktrees/` (room #15 J3), killing post-deploy under `set -e` — a plain `add -A` never errors on ignored paths and the follow-up `git reset -q -- .worktrees/` keeps stray lock files out of the integration target even on pre-gitignore-entry repos.
|
|
16
|
+
- `worktree-lifecycle.sh` — the worktree lifecycle seam: prepare/cleanup/inspect/integrate/verify-merge/status/post-deploy/terminal-cleanup/refresh-lock/lock-status over git worktrees (the integration target is the repo's current checkout — a branch name, or HEAD when detached — resolved by `integration_target()`; nothing in the lifecycle ever checks out a branch. `integrate` merges into the target, reconciles `origin/<target>` under the merge lock after the task merge — canary `a6d8b0c8`, 2026-09-11 — then pushes inline via `do_push` (no agent round-trip; 2026-09-19 REVIEW collapsed the old STEP-2 agent run, which added only a lock refresh). `do_push`: PUSHED (the refspec is always explicit — a branch pushes as `<branch>`, a detached HEAD as `HEAD:<destination>` where the destination is the remote's default branch from `resolve_push_destination()`; 2026-09-19 detached-HEAD audit REDO) / NO_REMOTE_PUSH (no origin — fail-soft, 2026-09-19 REVIEW: the old unconditional push exited 128 and mislabeled it); an unresolvable detached destination fails closed (never a silent skip, never a refusal of a knowable push); on push failure one fetch+merge retry while the lock is held, then fail closed without asserting a cause (never force-push). PUSH_SKIPPED survives only for the nothing-merged R5 path (no merge record + no lock held). Merged-but-unpushed retry recovery (blocker 35, 2026-09-21): identity-based, not ancestry-based — `current_branch_merge()` finds the newest merge on live HEAD's first-parent chain whose branch-side parent (^2) is the branch's CURRENT tip (the recorded merge, a do_push reconcile merge above it, or the manual R5 merge); that merge takes the lock and pushes instead of reporting MERGED_EMPTY. A record with no such merge means the branch moved since the merge — STALE_MERGE fails closed (never pushed, never MERGED_EMPTY). Branch gone (merge-lease reclaim): the recorded merge on the line is still the deliverable (no rework could have moved it). `push-target` proves the same identity before the ERROR-after-MERGED retry push. `CREW_STAGED_BASE` guard (ferried from get-provenance): the line must descend from the staged publish base — STAGED_BASE_MISMATCH fails closed before the merge (room #21 J3 shape). `verify-merge` always re-resolves the target from the current checkout and checks the recorded commit against the LIVE tip — never the record's word for the tip (2026-09-19 REVIEW: the old read-back compared the record to itself for target=HEAD). A detached HEAD skips the inline reconcile loudly (NO_REMOTE_RECONCILE) — the inline reconcile needs a static branch target, so on detached the destination is resolved explicitly in do_push, which reconciles lazily on push failure (2026-09-19 detached-HEAD audit REDO; the old "local by design" rationale is rejected); `push-target` is the thin R5 wrapper over `do_push`: refreshes the lock, distinguishes nothing-merged (no record + no lock → loud skip) from lock-lost (fail closed). (`.worktrees/<id>`, branch `task/<id>`). Branch resolution order: the crew registry (`<repo>/.worktrees/.registry/<id>`), then the canonical `task/<full-id>` ref, then a `task/<id-prefix>` ref matched by strict prefix enumeration (task 4e1a1bba — a Build agent may create the branch with raw git from an abbreviated id, bypassing prepare; prefix names are tolerated, never created; ambiguous prefixes fail closed; then — blocker 45, 2026-09-22 — the existing worktree's actual checked-out branch, the mechanical truth of last resort for an agent-improvised branch outside the task/ namespace). `resolve-branch` prints the resolved branch for agent-side one-liners. Nothing reconstructs the branch name — every command resolves it. Prepare fails closed on a dirty integration target; cleanup is forgiving. Prepare's reuse contract (blocker 45): an existing worktree prints REUSED only when checked out on the canonical task/<id> branch — a worktree on another branch is repaired (the registry is re-registered under the actual branch, never deleting the agent's commits) and prepare prints REPAIRED instead, so the repair is attributable in the record. `terminal-cleanup` is the run's last act at every park/fail boundary (called from `parkTask` in standard/bugfix/chore): releases the merge lock unconditionally and reclaims the worktree+branch only when the task branch is fully merged into the integration target — unmerged work is preserved for the human by design, and a dirty worktree is reported, never force-removed. `post-deploy` reports worktree removal honestly (a lying "removed" echo hid real leftovers — canary run 9, 2026-09-12). Requires both `CREW_REPO` and `CREW_HOME` (fail closed: BLOCKED, exit 2 when either is unset); `CREW_REPO` is exported so internal merge-lock.sh calls inherit the repo being worked on; `LIB_DIR` defaults to `$CREW_HOME/lib` (`CREW_LIB` override). `lock-status` reports the merge-lock state explicitly (`UNLOCKED`, or key=value: locked=true, task_id, holder, acquired_at, lease_seconds, age_seconds, remaining_seconds — always exit 0) so Publish can distinguish an empty-diff Integrate (no lock taken) from a refresh failure. post-deploy's `.worktrees/` guard is add-then-reset, never an all-negative pathspec: `git add -A -- ':!.worktrees/'` still exits 1 on git 2.43.0 against the crew's own gitignored non-empty `.worktrees/` (room #15 J3), killing post-deploy under `set -e` — a plain `add -A` never errors on ignored paths and the follow-up `git reset -q -- .worktrees/` keeps stray lock files out of the integration target even on pre-gitignore-entry repos.
|
|
17
17
|
- `test-detached-integrate.sh` — behavioral regression tests for the detached-HEAD Integrate fix (room #21 J3, 2026-09-19; extended 2026-09-19 REVIEW): fixture A stages a detached HEAD (reviewed SHA + crew-init commit) and proves integrate leaves HEAD detached, keeps the reviewed SHA and crew-init as ancestors, lands the task merge, never moves main, records `integration_target=HEAD`, verify-merge VERIFIEDs, and push-target reports NO_REMOTE_PUSH (fixture A has no origin); fixture B proves the branch path still merges, reconciles, and pushes inline to a local origin; fixture C proves verify-merge checks the record against the LIVE tip (a detached line reset past the merge does NOT verify; the branch-gone path verifies only when the recorded commit is an ancestor of the live tip); fixture D proves the no-remote branch path fails soft with NO_REMOTE_PUSH (exit 0); fixture E proves the CREW_STAGED_BASE guard (non-ancestor base → STAGED_BASE_MISMATCH before any merge; true base → STAGED_BASE_OK); fixture F proves merged-but-unpushed retry recovery (re-running integrate after a crashed push takes the lock and pushes instead of parking on MERGED_EMPTY); fixture F3 proves the blocker-35 stale-record case (branch moved backward after the recorded merge → STALE_MERGE fails closed: never pushed, never MERGED_EMPTY); fixture H proves the detached-HEAD audit REDO contract (2026-09-19) on a detached checkout with a local origin: push-destination fails closed when origin/HEAD is unset (genuinely unknowable destination), otherwise resolves the remote default branch; integrate pushes inline as `HEAD:main` (`PUSHED: origin/main (refspec HEAD:main)`), the remote ref advances to the merge commit, local main stays at the reviewed base, HEAD stays detached, verify-merge VERIFIEDs, and push-target re-pushes idempotently; fixture I proves the push-target identity contract on a branch with a local origin: I1 stale record → STALE_MERGE and the remote ref does not move, I2 branch restored to the merged tip → the legitimate retry still PUSHEDs, I3 the manual R5 shape (stale record plus a newer manual merge of the current tip) → PUSHEDs. Scratch dirs under /tmp only; CREW_HOME lives outside the scratch repos (as in real rooms) so lock files never trip the integrate preflight.
|
|
18
18
|
- `test-worktree-backend.sh` — regression tests for the lifecycle script (validate, prepare/reuse, inspect, status, cleanup, idempotent cleanup, dirty-main preflight, integrate remote-reconcile with fixture sensitivity + no-remote fail-soft) on scratch repos
|
|
19
19
|
- `test-version-write.sh` — regression tests for the escape-preserving step-8 version write in publish-npm.sh (fixture: current package.json with the \\u2014 escape; extracts the shipped block by anchor)
|
|
@@ -25,7 +25,7 @@ Shipped library: ESM JavaScript CLIs and import-safe modules, shell scripts for
|
|
|
25
25
|
- `compose-evidence-caption.js` — deterministic QA-evidence caption composer (2026-09-14): given `--crew-home`, `--task-id`, and the resolved `--audit-dir` name, extracts the task title/description, the QA verdict, the `publish: verified` commit (or the task's project-scoped provenance record when it belongs to this task), and the merge commit's subject line from the project repo — and prints the out-of-context transition Eric asked for (problem → solution → evidence handoff) to stdout. The caller appends the desktop/mobile attachment lines after this caption. Verdict source (2026-09-16): Hazel's `task-evidence/<taskId>/postchange/verdict.json` is authoritative, then the last line of the append-only `verdicts.jsonl` ledger, then the newest legacy `visual_verdict:` note (the parent visual-verdict protocol was retired 2026-09-15). Honest labels: "provenance not yet stamped" when no commit is known, "no QA verdict recorded" when none exists, and the visual-protocol-unavailable note only when no verdict evidence exists at all, the newest `baseline:` note is `baseline: none`, AND captures exist (a `baseline: none` note means no baseline captures to compare against — it never overrules a real QA verdict; the note's premise is the post-deploy capture, so without captures it would fabricate evidence again). Capture guard (2026-09-17, room #14): the composer resolves the audit dir itself (`<home>/workspace/ts-spaces/<deploy_slug>/audits/<auditDir>/`, the same root crew-api.js and readback-disk.js resolve) and verifies `screenshot.png` + `screenshot-mobile.png` are files before claiming "desktop + mobile captures below" — a bare `--audit-dir` name is never trusted. Missing deploy_slug, missing dir, or missing PNGs gets the honest no-captures opener and Evidence line (terminal surfaces named explicitly: "terminal surface — QA judges the CLI, not pixels"); the caller only appends attachment lines when its own freshness check passes. "Latest" is by event timestamp, sorted explicitly. Exits non-zero on missing input.
|
|
26
26
|
- `edit-image.py` — deterministic pixel-level evidence editing (2026-09-15, Pillow): `crop --region x,y,w,h` (pixel-exact, rejects out-of-bounds), `zoom --factor <f> [--center x,y]` (nearest-neighbor, returns to source frame size), `label --text <caption>` (caption bar), `nup --cols <n> --in <a.png> --in <b.png> [--labels "a|b"]` (side-by-side grid). Byte-deterministic PNG output; no network, no time, no randomness. Derivatives supplement raw frames — the source frame stays archived.
|
|
27
27
|
- `render-html.js` — deterministic HTML evidence composition (2026-09-15, headless Chromium): renders a local HTML layout to PNG (fixed viewport width, device scale 1, full-page screenshot). Hermetic: remote HTTP(S) assets are blocked and fail the render loudly; relative image paths resolve against the HTML file; local/system fonts only. Exit 3 reports NOT POSSIBLE when Chromium is unavailable. The composition layer above edit-image.py's pixel layer.
|
|
28
|
-
- `see-act.js` — single-step browser driver for experiential QA (2026-09-14): one browser action per invocation (`aria`, `shot`, `click`, `scroll`, `type`), one JSON line on stdout, exit 0/2/3 (3 = NOT POSSIBLE: missing playwright-core or Chromium). Spawns its own loopback forward proxy on an ephemeral port (Chromium blocks direct loopback). The QA work agent closes the OODA loop: run a step, read the screenshot/aria, decide the next. Determinism: no wall-clock reads, no randomness. `SEE_ACT_ARCHIVE_DIR=<phase-dir>` (2026-09-14): every screenshot is archived automatically as `001-shot-desktop.png`, `002-click-mobile.png`, ... (counter in `<dir>/.seq`); `--out` becomes optional; JSON carries `screenshot` + `archived`; explicit `--out` + archive copies into the archive; unusable dir is NOT POSSIBLE exit 3.
|
|
28
|
+
- `see-act.js` — single-step browser driver for experiential QA (2026-09-14): one browser action per invocation (`aria`, `shot`, `click`, `scroll`, `type`), one JSON line on stdout, exit 0/2/3 (3 = NOT POSSIBLE: missing playwright-core or Chromium). Spawns its own loopback forward proxy on an ephemeral port (Chromium blocks direct loopback). The QA work agent closes the OODA loop: run a step, read the screenshot/aria, decide the next. Determinism: no wall-clock reads, no randomness. `SEE_ACT_ARCHIVE_DIR=<phase-dir>` (2026-09-14): every screenshot is archived automatically as `001-shot-desktop.png`, `002-click-mobile.png`, ... (counter in `<dir>/.seq`); `--out` becomes optional; JSON carries `screenshot` + `archived`; explicit `--out` + archive copies into the archive; unusable dir is NOT POSSIBLE exit 3. Session protocol (2026-09-22, blocker 41): `session-start --name --url` holds ONE Chromium + page alive in a detached daemon (loopback control server, 127.0.0.1 only, idle-TTL exit); `session-act --name --attempt <action>` performs one action on the living page and appends one ooda-log line in the exact append-ooda-step schema; `session-end --name` closes the browser and prints the summary. The one-shot code path is frozen — the daemon duplicates the action implementations deliberately.
|
|
29
29
|
- `append-ooda-step.js` — deterministic writer for the OODA report log (2026-09-14, attempt identity 2026-09-15): `node append-ooda-step.js --log <path> --attempt <id> --step <n> --action <a> --exit <code> [--args <json>] [--screenshot <path>] [--observation <text>]` appends one JSON line to `<phase-dir>/ooda-log.jsonl` (`{step, attempt, action, args, exit, screenshot|null, observation}`). `--attempt` is required; steps must be strictly monotonic within an attempt (a rerun is a new attempt at step 1 — attempts accumulate, never overwrite). Actions: browser (`aria|shot|click|scroll|type`) and image (`crop|zoom|label|nup|compose` — the last five log evidence derivatives made with `lib/edit-image.py` / `lib/render-html.js`; frame-producing actions require `--screenshot`). Corrupt logs or sequence gaps fail loudly (exit 2). No wall-clock reads, no randomness.
|
|
30
30
|
- `write-ooda-verdict.js` — deterministic writer for the OODA terminal verdict (2026-09-14, append-only ledger 2026-09-15): `node write-ooda-verdict.js --dir <phase-dir> --attempt <id> --verdict <PASS|FAIL|NOT_POSSIBLE> --summary <text> --expected <text> --actual <text> --missing <json-array> [--reason <text>]` writes `<phase-dir>/verdict.json` (the latest verdict) and appends one JSON line to `<phase-dir>/verdicts.jsonl` — the append-only ledger: every attempt's verdict is preserved with a mechanical `seq`, never overwritten; corrupt or non-contiguous ledgers fail loudly. `--reason` is REQUIRED and must be non-empty for `FAIL` and `NOT_POSSIBLE` — a reason-less negative verdict fails with exit 2 before anything is written (2026-09-15). Exit 2 on bad input. See `docs/ooda-report.md`.
|
|
31
31
|
- `read-ooda-verdict.js` — deterministic cross-checker for the OODA terminal verdict (2026-09-15): `node read-ooda-verdict.js --dir <phase-dir> --expect <PASS|FAIL>` reads `<dir>/verdict.json`, prints one JSON line to stdout, exits 0 with `{ok:true, verdict, reason, summary, expected, actual, attempt}` when the record agrees with the prose expectation and a FAIL carries a non-empty reason, or exits 2 with `{ok:false, code}` — `missing|corrupt|contradiction|no_reason`. No wall-clock reads, no randomness. The bugfix QA closeout runs it against the prose `VERDICT:` line before any rework routing: a failed cross-check records the phase as failed for retry, never routes to rework.
|
package/lib/read-ooda-verdict.js
CHANGED
|
@@ -18,28 +18,32 @@
|
|
|
18
18
|
// parses, has a verdict field, the verdict equals --expect, and a FAIL
|
|
19
19
|
// carries a non-empty reason.
|
|
20
20
|
//
|
|
21
|
-
// visual_loop_unavailable (2026-09-16
|
|
22
|
-
// experiential evidence must never be terminal —
|
|
23
|
-
// instead of stamping done (the clean-room rubber
|
|
24
|
-
// never ran because playwright-core was
|
|
25
|
-
// layout, yet the task stamped done on a
|
|
26
|
-
// mechanical
|
|
27
|
-
//
|
|
28
|
-
//
|
|
29
|
-
// observation
|
|
30
|
-
//
|
|
31
|
-
//
|
|
32
|
-
//
|
|
21
|
+
// visual_loop_unavailable (2026-09-16, reworked 2026-09-22 blocker 44): a
|
|
22
|
+
// PASS verdict with missing experiential evidence must never be terminal —
|
|
23
|
+
// the QA closeout parks instead of stamping done (the clean-room rubber
|
|
24
|
+
// stamp: the see-act loop never ran because playwright-core was
|
|
25
|
+
// unresolvable from the release layout, yet the task stamped done on a
|
|
26
|
+
// mechanical-only PASS). "The loop could not run" is a machine claim, never
|
|
27
|
+
// a prose match — two ooda-log witnesses, no judgment about verdict prose:
|
|
28
|
+
// (1) a browser-action step (aria|shot|click|scroll|type) exited 3 with
|
|
29
|
+
// NOT POSSIBLE in its machine-written observation (the see-act
|
|
30
|
+
// contract for "tooling unresolvable");
|
|
31
|
+
// (2) the ooda-log contains zero COMPLETED (exit 0) browser-action steps
|
|
32
|
+
// — the loop left no evidence it ran. A missing or unreadable log is
|
|
33
|
+
// the zero case: the loop-didn't-run case fails closed.
|
|
34
|
+
// The old Signal 2 (regex-matching verdict.json's missing_evidence prose)
|
|
35
|
+
// is deleted: room #29 J4 parked on hazel's honest "not possible" reach-gap
|
|
36
|
+
// prose attached to a PASS whose loop had run 9 exit-0 steps. missing_evidence
|
|
37
|
+
// stays the designed home for reach-gap documentation, but prose never
|
|
38
|
+
// trips the guard — the room's verdict rules (not the dispatcher park) own
|
|
39
|
+
// reach-limited PASS verdicts.
|
|
33
40
|
//
|
|
34
|
-
// terminal_loop_unavailable (2026-09-17): the
|
|
35
|
-
//
|
|
36
|
-
//
|
|
37
|
-
//
|
|
38
|
-
//
|
|
39
|
-
//
|
|
40
|
-
// (2) verdict.json's missing_evidence names CLI tool-unavailability
|
|
41
|
-
// (command not found, not recognized, no runtime, not installed,
|
|
42
|
-
// not possible, unavailable, could not run/execute/launch).
|
|
41
|
+
// terminal_loop_unavailable (2026-09-17, same blocker-44 rework): the
|
|
42
|
+
// terminal-surface counterpart. Same fib shape, same fix: an exit-3 NOT
|
|
43
|
+
// POSSIBLE terminal step, or zero completed (exit 0) terminal steps. A
|
|
44
|
+
// terminal-surface run with no browser steps reports visual_loop_unavailable
|
|
45
|
+
// = true-by-absence — honest, and harmless: the closeout is surface-aware
|
|
46
|
+
// and reads only the flag matching the task's surface.
|
|
43
47
|
//
|
|
44
48
|
// Exit 2 with {ok:false, code, error} when:
|
|
45
49
|
// missing — verdict.json is absent (the agent never wrote one)
|
|
@@ -126,96 +130,67 @@ function main() {
|
|
|
126
130
|
}) + "\n");
|
|
127
131
|
}
|
|
128
132
|
|
|
129
|
-
//
|
|
130
|
-
//
|
|
131
|
-
// exited 3 with NOT POSSIBLE in its observation (the driver's contract for
|
|
132
|
-
// unresolvable tooling). Signal 2 — the verdict's own missing_evidence
|
|
133
|
-
// names tool-unavailability. Both are string matches on machine-written
|
|
134
|
-
// records, never a judgment about report prose.
|
|
133
|
+
// The guard reads the ooda-log only. Prose is never consulted: a guard
|
|
134
|
+
// that string-matches an agent's honesty notes parks the honest.
|
|
135
135
|
const BROWSER_ACTIONS = { aria: true, shot: true, click: true, scroll: true, type: true };
|
|
136
|
-
const
|
|
136
|
+
const TERMINAL_ACTIONS = { terminal: true };
|
|
137
|
+
const NOT_POSSIBLE = /not possible/i;
|
|
137
138
|
|
|
138
|
-
|
|
139
|
+
// Parse the ooda-log's machine-written lines; corrupt lines are skipped,
|
|
140
|
+
// a missing log is the empty set (the zero-completions case fails closed).
|
|
141
|
+
function oodaSteps(dir) {
|
|
139
142
|
let text;
|
|
140
143
|
try {
|
|
141
144
|
text = readFileSync(join(dir, "ooda-log.jsonl"), "utf8");
|
|
142
145
|
} catch (e) {
|
|
143
|
-
return
|
|
146
|
+
return [];
|
|
144
147
|
}
|
|
148
|
+
const out = [];
|
|
145
149
|
const lines = text.split("\n");
|
|
146
150
|
for (let k = 0; k < lines.length; k++) {
|
|
147
151
|
const line = lines[k].trim();
|
|
148
152
|
if (!line) continue;
|
|
149
|
-
let step;
|
|
150
153
|
try {
|
|
151
|
-
step = JSON.parse(line);
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
}
|
|
155
|
-
if (!step || !BROWSER_ACTIONS[step.action]) continue;
|
|
156
|
-
if (step.exit === 3 && /not possible/i.test(String(step.observation || ""))) return true;
|
|
154
|
+
const step = JSON.parse(line);
|
|
155
|
+
if (step && typeof step === "object") out.push(step);
|
|
156
|
+
} catch (e) { /* skip corrupt lines */ }
|
|
157
157
|
}
|
|
158
|
-
return
|
|
158
|
+
return out;
|
|
159
159
|
}
|
|
160
160
|
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
for (let k = 0; k <
|
|
166
|
-
|
|
161
|
+
// Signal 1: an instrument step exited 3 with NOT POSSIBLE in its
|
|
162
|
+
// machine-written observation — the driver's contract for "the tooling
|
|
163
|
+
// itself could not run".
|
|
164
|
+
function stepExit3NotPossible(steps, actions) {
|
|
165
|
+
for (let k = 0; k < steps.length; k++) {
|
|
166
|
+
const step = steps[k];
|
|
167
|
+
if (!actions[step.action]) continue;
|
|
168
|
+
if (step.exit === 3 && NOT_POSSIBLE.test(String(step.observation || ""))) return true;
|
|
167
169
|
}
|
|
168
170
|
return false;
|
|
169
171
|
}
|
|
170
172
|
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
// not run"). Signal 2 — the verdict's own missing_evidence names CLI
|
|
179
|
-
// tool-unavailability. Same string-match shape as the visual version, never
|
|
180
|
-
// a judgment about report prose.
|
|
181
|
-
const TERMINAL_ACTIONS = { terminal: true };
|
|
182
|
-
const TERMINAL_TOOL_UNAVAILABLE = /(command not found|not recognized|no runtime|not possible|not installed|unavailable|could not (run|execute|launch)|no cli)/i;
|
|
183
|
-
|
|
184
|
-
function oodaLogTerminalUnavailable(dir) {
|
|
185
|
-
let text;
|
|
186
|
-
try {
|
|
187
|
-
text = readFileSync(join(dir, "ooda-log.jsonl"), "utf8");
|
|
188
|
-
} catch (e) {
|
|
189
|
-
return false;
|
|
190
|
-
}
|
|
191
|
-
const lines = text.split("\n");
|
|
192
|
-
for (let k = 0; k < lines.length; k++) {
|
|
193
|
-
const line = lines[k].trim();
|
|
194
|
-
if (!line) continue;
|
|
195
|
-
let step;
|
|
196
|
-
try {
|
|
197
|
-
step = JSON.parse(line);
|
|
198
|
-
} catch (e) {
|
|
199
|
-
continue;
|
|
200
|
-
}
|
|
201
|
-
if (!step || !TERMINAL_ACTIONS[step.action]) continue;
|
|
202
|
-
if (step.exit === 3 && /not possible/i.test(String(step.observation || ""))) return true;
|
|
173
|
+
// Signal 2 (blocker 44): the complementary machine claim — the loop ran iff
|
|
174
|
+
// it left at least one completed (exit 0) instrument step in the session.
|
|
175
|
+
// Zero completions means the loop left no evidence it ran: fail closed.
|
|
176
|
+
function noCompletedSteps(steps, actions) {
|
|
177
|
+
for (let k = 0; k < steps.length; k++) {
|
|
178
|
+
const step = steps[k];
|
|
179
|
+
if (actions[step.action] && step.exit === 0) return false;
|
|
203
180
|
}
|
|
204
|
-
return
|
|
181
|
+
return true;
|
|
205
182
|
}
|
|
206
183
|
|
|
207
|
-
function
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
if (TERMINAL_TOOL_UNAVAILABLE.test(String(me[k]))) return true;
|
|
213
|
-
}
|
|
214
|
-
return false;
|
|
184
|
+
function visualLoopUnavailable(dir, record) {
|
|
185
|
+
// record (verdict.json) is deliberately not consulted: the guard's only
|
|
186
|
+
// witnesses are the ooda-log's machine-written lines.
|
|
187
|
+
const steps = oodaSteps(dir);
|
|
188
|
+
return stepExit3NotPossible(steps, BROWSER_ACTIONS) || noCompletedSteps(steps, BROWSER_ACTIONS);
|
|
215
189
|
}
|
|
216
190
|
|
|
217
191
|
function terminalLoopUnavailable(dir, record) {
|
|
218
|
-
|
|
192
|
+
const steps = oodaSteps(dir);
|
|
193
|
+
return stepExit3NotPossible(steps, TERMINAL_ACTIONS) || noCompletedSteps(steps, TERMINAL_ACTIONS);
|
|
219
194
|
}
|
|
220
195
|
|
|
221
196
|
main();
|