muse-crew 0.17.2 → 0.17.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. package/API.md +11 -0
  2. package/docs/decisions/composition-machinery.md +180 -0
  3. package/docs/decisions/publish-path.md +6 -0
  4. package/docs/decisions/workflow-core.md +5 -4
  5. package/lib/AGENTS.md +2 -1
  6. package/lib/bugfix/phases/build.js +168 -0
  7. package/lib/bugfix/phases/capture.js +170 -0
  8. package/lib/bugfix/phases/integrate.js +165 -0
  9. package/lib/bugfix/phases/map.js +129 -0
  10. package/lib/bugfix/phases/publish.js +592 -0
  11. package/lib/bugfix/phases/qa.js +356 -0
  12. package/lib/bugfix/phases/reproduce.js +254 -0
  13. package/lib/bugfix/phases/review.js +292 -0
  14. package/lib/bugfix/phases/triage.js +69 -0
  15. package/lib/chore/CONTRACT.md +181 -0
  16. package/lib/chore/DISPOSITION.md +98 -0
  17. package/lib/chore/extract.js +204 -0
  18. package/lib/chore/phase-lib.js +885 -0
  19. package/lib/chore/phases/build.js +110 -0
  20. package/lib/chore/phases/capture.js +82 -0
  21. package/lib/chore/phases/integrate.js +109 -0
  22. package/lib/chore/phases/map.js +81 -0
  23. package/lib/chore/phases/publish.js +543 -0
  24. package/lib/chore/phases/review.js +255 -0
  25. package/lib/chore/phases/triage.js +57 -0
  26. package/lib/chore/prompts/evidence-gatherer.js +41 -0
  27. package/lib/chore/prompts/evidence-gatherer.schema.json +1 -0
  28. package/lib/chore/prompts/tool-check.js +15 -0
  29. package/lib/chore/prompts/trailers.js +56 -0
  30. package/lib/chore/prompts/verdict-reask.js +28 -0
  31. package/lib/chore/prompts/verdict-reask.schema.json +1 -0
  32. package/lib/chore/prompts/work-agent.js +52 -0
  33. package/lib/chore/prompts/work-agent.schema.json +1 -0
  34. package/lib/chore/spawn-keys.js +44 -0
  35. package/lib/chore/spawn-vocab.js +87 -0
  36. package/lib/chore-run.js +538 -0
  37. package/lib/chore-tick.js +289 -0
  38. package/lib/crew-api.js +273 -0
  39. package/lib/crew-dispatch-worker.js +27 -7
  40. package/lib/crew-release.sh +7 -2
  41. package/lib/extract.js +252 -0
  42. package/lib/prompts/tool-check.js +18 -0
  43. package/lib/prompts/trailers.js +59 -0
  44. package/lib/prompts/verdict-reask.js +31 -0
  45. package/lib/prompts/verdict-reask.schema.json +1 -0
  46. package/lib/prompts/work-agent.js +56 -0
  47. package/lib/prompts/work-agent.schema.json +1 -0
  48. package/lib/reap-spawns.js +407 -0
  49. package/lib/schema.sql +12 -1
  50. package/lib/spawn-keys.js +47 -0
  51. package/lib/spawn-step.js +572 -0
  52. package/lib/standard/phases/build.js +120 -0
  53. package/lib/standard/phases/capture.js +163 -0
  54. package/lib/standard/phases/integrate.js +172 -0
  55. package/lib/standard/phases/map.js +119 -0
  56. package/lib/standard/phases/publish.js +565 -0
  57. package/lib/standard/phases/qa.js +399 -0
  58. package/lib/standard/phases/review.js +281 -0
  59. package/lib/standard/phases/triage.js +64 -0
  60. package/lib/test-detached-integrate.sh +47 -0
  61. package/lib/workflow-driver.js +605 -0
  62. package/lib/workflow-lib.js +1012 -0
  63. package/lib/workflow-spec.js +187 -0
  64. package/lib/worktree-lifecycle.sh +55 -3
  65. package/package.json +1 -1
  66. package/seed/cron-body-template.md +61 -9
  67. package/workflows/bugfix.js +17 -17
  68. package/workflows/chore.js +16 -16
  69. package/workflows/docs.js +14 -11
  70. package/workflows/standard.js +16 -16
package/API.md CHANGED
@@ -147,6 +147,17 @@ Records a worker-layer execution in the `worker_runs` ledger (Piece 1, 2026-09-2
147
147
 
148
148
  Bad transitions (start with an `id`, finish without one, finish with `running`, missing `phase`) throw usage errors (exit 2). Returns `{ "ok": true, "id" }` on start and `{ "ok": true, "finished": true }` on finish. Orphan note: a dispatcher killed between start and finish leaves a `running` row — readers treat a `running` row with no live dispatcher as stale, never as success.
149
149
 
150
+ ### `get-worker-run`
151
+
152
+ Reads back the newest orchestration-telemetry row for a task and phase (Piece 3, 2026-09-26). The chore driver has no run-state file, so a relaunched invocation re-discovers its own `worker_runs` row through this read and finishes it instead of minting a duplicate.
153
+
154
+ | Field | Type | Required | Notes |
155
+ |-------|------|----------|-------|
156
+ | `task_id` | uuid | yes | |
157
+ | `phase` | string | yes | e.g. `chore` |
158
+
159
+ Returns `{ "ok": true, "row": <worker_runs row> }` for the newest row with matching `task_id`+`phase` and NULL `kind` (spawn-ledger rows carry a `kind` and are never returned); `{ "ok": false, "error": "not_found" }` (exit 3) when no such row exists. Missing args are a usage error (exit 2).
160
+
150
161
  ### `record-version-ack`
151
162
 
152
163
  Stamps a version acknowledgement (0.14.6). The version must name an issued attempt for the task/commit.
@@ -0,0 +1,180 @@
1
+ # Composition Machinery — Phase A Design Note
2
+
3
+ **Status:** Proposal — awaiting Eric's ruling.
4
+ **Date:** 2026-09-26
5
+ **In service of:** Park 14 (a real Standard task is the reason this exists).
6
+
7
+ ## 1. What this is
8
+
9
+ One generic driver replaces the per-workflow drivers (`chore-run.js`, and the
10
+ future `standard-run.js` / `bugfix-run.js` that we are deliberately NOT
11
+ writing). Workflows become declarative specs: phase sequences, transition
12
+ rules, guards, typed handoffs. The driver interprets the spec. Phases remain
13
+ code modules with declared I/O contracts (the §6 discipline from the chore
14
+ port).
15
+
16
+ Three workflows prove the machinery: Chore (first spec, equivalence proof),
17
+ Standard (validated against Park 14's known trajectory), Bugfix. The
18
+ composition UI comes after the spec is proven — it is a view over the spec,
19
+ not a second source of truth.
20
+
21
+ ## 2. The declarative gradient — recommendation
22
+
23
+ **Recommendation: declarative about composition, code-backed phases.**
24
+
25
+ The spec declares *structure*: which phases run, in what order, under what
26
+ guards, with what typed handoffs between them. The phases themselves are code
27
+ — they already are, and they stay that way. The UI composes specs visually
28
+ and renders each phase as a node showing its I/O contract, never its
29
+ internals.
30
+
31
+ This is not derived from first principles. It is where the incumbent
32
+ landscape converged:
33
+
34
+ - **n8n, Node-RED, Retool, Langflow, Botpress** (visual-first): the canvas
35
+ owns structure — sequencing, branching, wiring. Code lives inside opaque
36
+ nodes with typed edges (n8n's Code node, Node-RED's Function node,
37
+ Retool's Code block). The UI never breaks on code; it renders the node
38
+ and its contract.
39
+ - **Temporal, Airflow, Dagster, LangGraph Studio** (code-first): the UI is
40
+ read-only observability. Nobody draws boxes to define logic.
41
+ - **Nobody does code↔canvas round-trip authoring well.** Every system picks
42
+ a primary side. The visual-first systems keep the canvas as the durable
43
+ artifact and code as a contained inline escape.
44
+
45
+ Eric's P0/P1 maps onto this exactly:
46
+
47
+ - **P0** (writers, world-builders, problem-solvers composing powerfully):
48
+ the canvas owns composition. Sequencing, guards, handoffs are all
49
+ declarative and therefore all visual. A non-programmer can string up a
50
+ real workflow.
51
+ - **P1** (power users, capabilities not expressed in the UI): the phase is
52
+ the escape hatch. A code-backed phase appears on the canvas as a node
53
+ with its declared I/O contract visible — typed edges in, typed edges
54
+ out — and its implementation opaque. The UI renders it the way n8n
55
+ renders a Code node: present, honest, unbroken.
56
+
57
+ **What the spec does NOT get:** conditionals, loops, or scripting in the
58
+ spec format. Retool's docs steer users explicitly toward visual blocks for
59
+ control flow and reserve code blocks for transformation — and their
60
+ community's failure mode is "JavaScript glue scattered across the canvas."
61
+ Our three workflows don't need spec-level branching; they need sequencing,
62
+ guards, and rework loops, all of which are structural. If a future workflow
63
+ needs real branching, that arrives as a phase (code) or as a spec-format
64
+ extension with its own design note — not as an accident.
65
+
66
+ ## 3. Spec format sketch
67
+
68
+ A workflow spec is data, not code. Sketch (exact syntax TBD in
69
+ implementation):
70
+
71
+ ```
72
+ workflow: standard
73
+ phases:
74
+ - triage (contract: reads task, writes classification)
75
+ - capture (contract: reads classification, writes evidence bundle)
76
+ - map (contract: reads bundle, writes plan)
77
+ - build (contract: reads plan, writes worktree + merge record)
78
+ - review (contract: reads merge record, writes verdict)
79
+ - integrate (contract: reads verdict, writes merge commit)
80
+ - qa (contract: reads merge commit, writes qa verdict)
81
+ - publish (contract: reads qa verdict, writes published artifact)
82
+ transitions:
83
+ - on: review.rework -> to: map (rework loop)
84
+ - on: qa.stale -> to: park (the Park 14 gate, declarative)
85
+ - on: phase.failed -> to: failed
86
+ guards:
87
+ - qa-bundle-fresh (named guard, implemented once, referenced by name)
88
+
89
+ rejectedResume: Build # Phase B: where a rejected latest session resumes
90
+ # (rework entry point — workflow knowledge, spec-declared;
91
+ # undeclared + rejected session = fail closed)
92
+ ```
93
+
94
+ What makes this the right level:
95
+
96
+ - Every transition our three workflows need is expressible: advance,
97
+ rework-to-earlier-phase, park, fail. These are the typed results the
98
+ chore driver already handles (ADVANCE / REWIND / PARK / FAILED /
99
+ NEED_SPAWN / WAIT).
100
+ - Guards are named and implemented once (e.g. the QA staleness check),
101
+ referenced by name. The spec says *which* guard; code says *how*.
102
+ - Phase I/O contracts are the same §6 contracts the chore port already
103
+ captured. No new concept.
104
+
105
+ ## 4. Generic driver extraction
106
+
107
+ `chore-run.js`'s `driveLoop` is already generic. The extraction:
108
+
109
+ - **Stays in the driver (generic):** derive start phase from ledger state,
110
+ run phase module, interpret typed result, apply transition table, handle
111
+ NEED_SPAWN via the ferry, WAIT/re-drive semantics, terminal closeout
112
+ (DONE/parked/failed), pin lifecycle, the pre-refusal protocol.
113
+ - **Moves to the spec (per-workflow):** phase order, transition table,
114
+ guard references, phase-to-module mapping.
115
+ - **Stays in phase modules (per-phase):** everything else — the actual
116
+ work, prompts, guards' implementations.
117
+
118
+ Transition enforcement note (Phase B, 2026-09-26): the spec's
119
+ `transitions` table is validated (endpoints must name declared phases) but
120
+ not enforced by the driver — the mechanical rule is `result.next ∈
121
+ phaseNames`, which is exact parity with chore-run.js. The table is
122
+ documentation for the future composition UI. Enforcing named transitions
123
+ is a spec-format extension with its own design note, not an accident.
124
+
125
+ Terminal-transition note (Phase C, 2026-09-26): PARK/FAILED/DONE are result
126
+ types, not phase-to-phase edges, and are NOT declared in the spec's
127
+ `transitions` table — the validator requires `to` to name a declared
128
+ phase, and a terminal outcome has no destination phase. The Standard spec
129
+ initially carried a `QA→park` transition (with a `reason_prefix` the closed
130
+ schema also rejected); both were removed. The park behavior lives in the
131
+ phase module (QA's staleness gate) and the driver's `handlePhaseResult`;
132
+ the spec documents it via the phase's contract string. If the future
133
+ composition UI needs to render terminal edges, that's a spec-format
134
+ extension with its own design note.
135
+
136
+ Runtime-library note (Phase B): the driver's run helpers (`deriveRunEnv`,
137
+ `pinLifecycle`, `deriveReworkCount`, `projectGuard`, `parkTask`) still
138
+ come from `lib/chore/phase-lib.js`. Generalizing that library is Phase
139
+ C/D work. Known workflow-shaped behavior inside it (must be generalized
140
+ or spec-driven when Standard/Bugfix arrive):
141
+ - `projectGuard`'s abort path writes a failed session with `step: "Build"`
142
+ and tells the dispatcher to re-launch from Build — chore's rework
143
+ entry, not a generic one.
144
+ - The experiential seed reads the `"Triage"` session by name.
145
+
146
+ The driver never imports a workflow by name. It takes a spec path. This is
147
+ what makes the future UI possible: the UI edits specs, the driver runs
148
+ them, and neither knows about the other beyond the spec format.
149
+
150
+ ## 5. Migration path
151
+
152
+ Same discipline as the chore cutover — additive, proven, then flipped:
153
+
154
+ 1. Build the generic driver + spec loader alongside `chore-run.js`.
155
+ Nothing changes at runtime.
156
+ 2. Express Chore as the first spec. Prove equivalence on scratch: same
157
+ inputs → same phase sequence → same terminal states. Suite green.
158
+ 3. Cut Chore over to the generic driver on the soak cell (one release,
159
+ same release machinery as 817241c). `chore-run.js`'s bespoke loop
160
+ retires.
161
+ 4. Standard as spec, validated against Park 14's recorded trajectory
162
+ (Phase C). Bugfix as spec (Phase D).
163
+ 5. Read-only visualization (Phase E). Composition editor later, as its
164
+ own project.
165
+
166
+ At no point do two drivers run the same workflow. At no point does the
167
+ soak cell run unproven code.
168
+
169
+ ## 6. Open questions for Eric
170
+
171
+ 1. **Gradient:** is "declarative about composition, code-backed phases"
172
+ the right landing? The alternative is thinner (ordered phase list only,
173
+ transitions hardcoded in the driver) or thicker (spec-level
174
+ conditionals). Recommendation is the middle.
175
+ 2. **Spec syntax:** JSON, YAML, or a small DSL? Recommendation: start with
176
+ JSON (mechanical, validatable, no parser to maintain); revisit if
177
+ humans hand-author specs often.
178
+ 3. **Guard implementation:** guards as phase-module exports, or as
179
+ first-class spec-referenced functions? Recommendation: phase-module
180
+ exports — fewer concepts, guards live next to the phases they protect.
@@ -766,6 +766,12 @@ The fix is in the lifecycle, not the prompt. The old merged-but-unpushed recover
766
766
 
767
767
  Applies to: lib/worktree-lifecycle.sh; standard, bugfix, chore (one marker line).
768
768
 
769
+ ### Fast-forward recovery (2026-09-27, Park 14)
770
+
771
+ Park 14's Integrate failed 3x with STALE_MERGE: the archive work had landed on soak/main by fast-forward (no merge commit exists), so `current_branch_merge()` found nothing — but a merge WAS recorded and the branch tip is exactly the target tip. The `ahead==0` recovery now also recognizes this shape via `ff_integrated()`: branch tip == target tip, exact equality only. Ancestry alone is NOT accepted — a branch reset backward after integration remains an ancestor of the target, and accepting it would defeat blocker-35 (fixture F3). The ff recovery fires only when a merge is recorded (genuine empty diffs still take MERGED_EMPTY with no lock). No merge record is appended for the ff shape: the state is self-describing (tip == target tip).
772
+
773
+ Applies to: lib/worktree-lifecycle.sh (ff_integrated, cmd_integrate, cmd_push_target).
774
+
769
775
  <a id="publish-diff-base"></a>
770
776
  ## Publish diff base (BASE..HEAD)
771
777
 
@@ -193,14 +193,15 @@ Applies to: standard, bugfix, chore.
193
193
  <a id="discarded-reason"></a>
194
194
  ## Discarded reason
195
195
 
196
- Invariant: the runtime threw the output away; reason is discarded.
196
+ Invariant: a throw is classified — only the runtime scanner's "output was not JSON" throw is a brace discard; any other throw is an agent error (2026-09-27, Park 14: Integrate errored on a real STALE_MERGE while the trailer claimed a brace discard). Reason is "discarded" for scanner discards, "errored" for agent errors — the retry must not be told to reword prose when the underlying command failed.
197
197
 
198
- Applies to: standard, bugfix, chore.
198
+ Applies to: standard, bugfix, chore, docs.
199
199
 
200
200
  ```
201
201
  // reason: "discarded" (the runtime threw the output away - it could not be
202
- // machine-read), "empty" (agent() returned without throwing but produced
203
- // nothing usable), "no-tools" (the worker's TOOL CHECK reported
202
+ // machine-read), "errored" (the attempt errored before producing a usable
203
+ // report — not a brace discard), "empty" (agent() returned without throwing
204
+ // but produced nothing usable), "no-tools" (the worker's TOOL CHECK reported
204
205
  // artifact_tools: missing), or "no-transport" (the worker's TOOL CHECK
205
206
  // reported shell_transport: unavailable).
206
207
  // The trailer tells the retry what to expect, not just to try again.
package/lib/AGENTS.md CHANGED
@@ -10,9 +10,10 @@ Shipped library: ESM JavaScript CLIs and import-safe modules, shell scripts for
10
10
 
11
11
  - `build-registry.js` — deterministic extractor that generates `workflows/registry.json` (workflow step registry) from the workflow files' `meta` blocks at release time; invoked by `crew-release.sh` deploy
12
12
  - `crew-api.js` — the crew-owned task-service API (dependency inversion, 2026-09-11): a zero-dependency Node CLI implementing the API.md contract against `$CREW_HOME/crew-state.db` (schema in `schema.sql`). Workflows call it through their agents' shell; the dashboard delegates to it. All state-machine invariants live as CHECK constraints in the schema, never in client prose. Includes the `record-phase` composite (session + event in one transaction; 2026-09-21 optionally carries a structured Review verdict payload — validated at the API boundary, written to the `verdicts` table in the same transaction, session-keyed first-write-wins idempotency — plus the `get-verdicts` read surface) and a one-time `migrate` import from a dashboard app.db. Active-release resolution is split in two (room #15): `resolveActiveReleaseName` (symlink-only — writers like the initial-provenance stamp record the active release without proving they are it) and `resolveActiveRelease` (symlink + self-path cross-check — verifiers like `scan-ack-pending` refuse to stamp claims when the running code isn't the active release's own). Provenance is per-project (2026-09-18, room #15 blocker 8): nullable `provenance_*` columns on the projects row, a single `stampProvenance()` writer, `set-provenance`/`get-provenance` require `project_id` (no silent global fallback), and a watermarked openDb backfill that attributes the legacy `config.provenance.*` triple to exactly-one ancestor match — never fabricated, otherwise deferred.
13
- - `crew-dispatch-worker.js` — worker-layer port of the sandboxed dispatch dispatcher (Piece 1, 2026-09-26): full-privilege Node, reads the board through crew-api.js via execFile (argv only, no shell), reproduces the eligibility / retry / reservation / playtest / tick-release / dispatch-decision logic, and tags every claim `executor: "worker"`. In Piece 1 it runs `--read-only` as the tick's shadow BEFORE the authoritative sandboxed dispatcher — both read the same board state; the shadow never mutates it, never launches, never writes the tick-release or dispatch-decision logs. Appends its own shadow evidence line to `$CREW_HOME/.dispatch-shadow.jsonl` (machine-written, no agent transcription). Claims are recommendations only: `executor` names the layer that computed the recommendation, not the layer that launched it — in Piece 1 no `executor: "worker"` claim is ever launched.
13
+ - `crew-dispatch-worker.js` — worker-layer port of the sandboxed dispatch dispatcher (Piece 1, 2026-09-26): full-privilege Node, reads the board through crew-api.js via execFile (argv only, no shell), reproduces the eligibility / retry / reservation / playtest / tick-release / dispatch-decision logic, and tags every claim `executor: "worker"`. Modes: default is the safe shadow (compute claims, no writes — `--read-only` is an explicit alias); `--authoritative` acquires claims via `reserve-dispatch` (the same atomic UPSERT the sandbox uses), acknowledges the poll, and writes the tick-release/dispatch-decision logs. In Piece 1 it runs `--read-only` as the tick's shadow BEFORE the authoritative sandboxed dispatcher — both read the same board state; the shadow never mutates it, never launches, never writes the tick-release or dispatch-decision logs. Appends its own shadow evidence line to `$CREW_HOME/.dispatch-shadow.jsonl` (machine-written, no agent transcription). Claims are recommendations only: `executor` names the layer that computed the recommendation, not the layer that launched it — in Piece 1 no `executor: "worker"` claim is ever launched.
14
14
  - `compare-dispatch-shadow.js` — mechanical shadow verdict (Piece 1, 2026-09-26): pairs each `.dispatch-shadow.jsonl` line with the authoritative `.dispatch-decisions.jsonl` line for the same tick (`decision.tick_seq === shadow.tick_seq_before + 1` — the shadow runs first, the sandbox dispatcher then writes its tick-release line and its decision) and compares the sorted `task_id|workflow|step` claim triples, ignoring `executor`. Emits `SHADOW_MATCH` / `SHADOW_DIVERGE` (names the differing triples) / `SHADOW_UNPAIRED` (decision missing — the authoritative dispatcher died) and appends the verdict to `$CREW_HOME/.dispatch-shadow-verdicts.jsonl`; idempotent (already-verdict shadows are skipped). Exit 0 on a verdict — divergence is evidence for the cutover review, not a tick failure; exit 2 on usage/IO errors.
15
15
  - `spawn-boundary.js` — fail-closed spawn-bounds validator (Piece 1, 2026-09-26): import-safe ESM, `validateSpawnBounds()` requires prompt/schema co-location, cwd inside the workdir, rejection of cwd inside crew state/observer paths or crewHome, and rejection of sensitive env-key patterns; throws `BOUNDS_REJECTED`. Bare module execution exits 0 (release entry-gate contract). The production spawn/request wrapper is Piece-3 work; this is only the validator.
16
+ - `chore-run.js` — Chore driver loop (sandbox-exit Piece 3, 2026-09-26): the worker-layer port of `workflows/chore.js`. Shebang CLI, per-boundary stateless invocation — `node lib/chore-run.js --crew-home <dir> --task-id <id> --tick-seq <n>`; stdout carries exactly one JSON frame (`NEED_SPAWN` | `WAIT` | `DONE`), all logs on stderr. Sequences the seven phase modules in `lib/chore/phases/` (Triage → Publish) with ADVANCE/REWIND routing, fresh per-phase claims (one session per phase, source parity), task-scoped spawn keys, and fail-closed derivation (a derivation failure emits `DONE failed`, never a partial advance). No `agent()` calls; deterministic work is `execFile` to the pinned `crew-api.js` (argv only, no shell); creative boundaries are emitted as `NEED_SPAWN` for the spawn bridge. `--help` usage on stdout, exit 0 (shebang⇔CLI contract).
16
17
  - `schema.sql` — the crew-owned state schema: projects, tasks, poll_state, config, agent_sessions, events. Vocabularies enforced by CHECK constraints; `rejected` is a valid event type (the 2026-09-11 crash was a stored session whose event was rejected). Column names match the historical dashboard tables for a verbatim migration.
17
18
  - `crew-release.sh` — immutable release manager: deploy, rollback, prune
18
19
  - `merge-lock.sh` — serialized merge lock for concurrent agents: time-based holder lease (bug 2fc8f52f — an unexpired lease is held regardless of process liveness; only an expired lease may be broken). Requires both `CREW_REPO` and `CREW_HOME` (fail closed: BLOCKED, exit 2 when either is unset). Lock file is key=value: task_id, opaque holder identity (never a PID), acquired_at epoch, lease_seconds (default 600, override via MERGE_LOCK_LEASE_SECONDS). acquire/refresh/release/status/force-release; holder-only refresh and release; every op appends to $CREW_HOME/.merge-lock.log. Stale-lease breaks serialize on a sidecar `$LOCK_FILE.flock` with an in-critical-section lease re-read (R-B1, 2026-09-21); refresh rewrites via temp-file + atomic rename so readers never see a torn file, and takes the same sidecar flock with an identity-only re-check inside the critical section — a reclaim always changes `task_id`, so identity alone closes the clobber (no expiry check on refresh: a long build that outran the lease legitimately revives its lock); release takes the same flock with the identity re-check inside (review pass 2, 2026-09-21) so a release can never `rm` a reclaimer's fresh lock.
@@ -0,0 +1,168 @@
1
+ // lib/bugfix/phases/build.js — Bugfix Build phase (worker layer).
2
+ //
3
+ // Wren implements Mara's spec in a git worktree, commits as
4
+ // "fix: <title>", and declares a machine-readable verdict. Derived from
5
+ // workflows/bugfix.js (the Bugfix Build step).
6
+ //
7
+ // Contract:
8
+ // - Read: mapper spec (re-derived from the latest completed Map session
9
+ // notes — the durable handoff); rejection notes (re-derived from the
10
+ // latest rejected Review/QA session notes on rework); task title;
11
+ // project surface/description; publish type.
12
+ // - Instructions: exact Bugfix prompt — worktree prepare, heartbeat,
13
+ // mapper spec, worktree hint, UX bars, public docs, npm release
14
+ // declaration (when publish type is npm), commit as "fix: <title>",
15
+ // repo_diff:none allowance, rework notes, worktree + VERDICT markers.
16
+ // - Worktree confinement (verdict PASS only): the declared `worktree:`
17
+ // path must match the worktree hint exactly, else the phase fails for
18
+ // dispatcher retry.
19
+ // - Verdict FAIL still falls through to Review (recorded "rejected") —
20
+ // Build never parks and never rewinds on its own verdict.
21
+ // - The release decision (release:/version_bump:) is extracted by Review
22
+ // from these durable session notes (marker lines survive the summary
23
+ // truncation).
24
+ // - Advance: Review (on both PASS and FAIL).
25
+ //
26
+ // Phase I/O contract (Phase D):
27
+ // read: Map session notes (completed), Review/QA session notes
28
+ // (rejected, for rework), task title, project surface/desc,
29
+ // publish type
30
+ // write: session (completed | rejected | failed [confinement]),
31
+ // event (completed | rejected | failed [confinement])
32
+ // out: ADVANCE → Review | NEED_SPAWN / STANDBY / FAILED (boundary /
33
+ // confinement)
34
+
35
+ import {
36
+ runWorkBoundary, recordPhase, buildEventPreamble, summarizeReport,
37
+ ensureClaimed, log, latestSessionNotes, crewApi, lifecycleEnvPrefix,
38
+ } from "../../workflow-lib.js";
39
+ import { extractWorktree } from "../../extract.js";
40
+
41
+ export const PHASE = { name: "Build", identity: "wren" };
42
+
43
+ // buildInstructions — verbatim from workflows/bugfix.js (variable
44
+ // references remapped; logic and prose unchanged).
45
+ // ctx: { env, state, mapperSpec, rejectionNotes, safeTitle }.
46
+ export function buildInstructions(ctx) {
47
+ var env = ctx.env, state = ctx.state;
48
+ var instructions = "STEP 1: Prepare your worktree.\n" +
49
+ "Run: "+ lifecycleEnvPrefix(env) + " prepare " + env.taskId + "\n" +
50
+ "If the output says CREATED or REUSED, proceed. If it says ERROR, stop and report the failure clearly.\n\n" +
51
+ "HEARTBEAT: Start a background heartbeat loop NOW (before STEP 2) to signal you are still alive during this build. Run this once:\n" +
52
+ "(while node " + env.crewApiPinned + " --crew-home " + env.crewHome + " heartbeat-session --json '{\"id\": \"" + state.activeSessionId + "\"}' >/dev/null 2>&1; do sleep 900; done) &\n" +
53
+ "The loop heartbeats every 15 minutes and exits on its own when the session ends. This prevents the dispatcher from mistaking a long build for a dead session and launching a duplicate.\n\n" +
54
+ "STEP 2: Edit source files to implement the mapper's spec below.\n" +
55
+ (ctx.mapperSpec ? "MAPPER'S SPEC (implement exactly this):\n" + ctx.mapperSpec + "\n\n" : "") +
56
+ "Your working directory: " + env.worktreeHint + "/\n" +
57
+ "This is the project source: " + env.projectDesc + "\n" +
58
+ "Edit the TypeScript source files directly. Do NOT use artifact_edit — that happens in the Publish phase.\n" +
59
+ "Do not add unrequested features. Build exactly what the spec calls for.\n" +
60
+ (env.surfaceTerminal ? "TERMINAL UX: build to the shared bar at " + env.uxDoctrinePath + " — --help text, error messages, and exit codes are user-facing and ship in this commit.\n" : "") +
61
+ (env.surfaceArtifact ? "ARTIFACT UX: build to the shared bar at " + env.uxDoctrinePath + " — the rendered result is what the user sees; it ships in this commit.\n" : "") +
62
+ "PUBLIC DOCS: If your change is public-affecting (it alters anything a user or consumer can observe: API actions, parameters, behavior, or errors), update the public docs in the same commit — API.md for API changes. Documentation and implementation ship together.\n\n" +
63
+ (env.publishType === "npm" ? "PACKAGE VERSION: this project publishes to the npm registry. Versions are assigned at PUBLISH time — never in your branch. Do NOT touch the `version` field in package.json (or package-lock). Instead, end your report with exactly these two lines:\n" +
64
+ "release: yes|no — 'yes' if this change warrants a published release (anything a consumer can observe: workflow behavior, phase lists, identities, published docs, API); 'no' if internal-only.\n" +
65
+ "version_bump: patch|minor|major — patch for fixes (default), minor for new behavior, major for breaking changes. Omit this line only when release is no.\n" +
66
+ "Example: release: yes\\nversion_bump: minor\n\n" : "") +
67
+ "STEP 3: Commit your changes.\n" +
68
+ "cd " + env.worktreeHint + "\n" +
69
+ "git add -A\n" +
70
+ "git commit -m \"fix: " + ctx.safeTitle + "\"\n\n" +
71
+ "If the task's deliverable is runtime state (a cron definition, scheduler change, or dashboard/config state created outside the repo) and the repository genuinely needs no change, do NOT fabricate a commit: leave the branch with no commits ahead of the integration target and declare `repo_diff: none` in your report, naming the runtime-state deliverable. If you verified the deliverable is already on the integration target (a prior merge landed it — do NOT re-implement working code), make no commit and declare `repo_diff: none` in your report. The workflow verifies the claim mechanically from git — the task's own merge records, never your declaration, are the proof of delivery. Otherwise commit your changes normally.\n\n" +
72
+ (ctx.rejectionNotes ? "This is REWORK after rejection. Address these specific issues:\n" + ctx.rejectionNotes + "\n\n" : "") +
73
+ "Report back in plain prose: what you built and the outcome." +
74
+ (env.publishType === "npm" ? " End your report with the release: and version_bump: lines exactly as specified above — keep them on their own lines, lowercase, unrephrased — then a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then a final line with exactly: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not." : " End your report with a line `worktree: ` followed by the exact working directory path from above (copy it verbatim \u2014 it must match character-for-character), then exactly one line: VERDICT: PASS if the build is complete, VERDICT: FAIL if it is not.");
75
+
76
+
77
+ return instructions;
78
+ }
79
+
80
+ // latestRejectionNotes — the notes of the latest rejected Review or QA
81
+ // session (the worker-layer equivalent of the source's in-memory
82
+ // rejectionNotes, set when Review/QA rejects and read by the next Build).
83
+ async function latestRejectionNotes(env) {
84
+ var st = await crewApi(env, "get-state", { events_limit: 1 });
85
+ var sessions = (st && st.sessions) || [];
86
+ var best = null;
87
+ for (var i = 0; i < sessions.length; i++) {
88
+ var s = sessions[i];
89
+ if (s.task_id === env.taskId && (s.step === "Review" || s.step === "QA") && s.status === "rejected") {
90
+ if (!best || String(s.started_at || "") > String(best.started_at || "")) best = s;
91
+ }
92
+ }
93
+ return best ? (best.notes || "") : "";
94
+ }
95
+
96
+ export async function runPhase(ctx) {
97
+ var env = ctx.env, state = ctx.state;
98
+
99
+ // Mapper spec: the durable handoff from Map (the truncated summary Map
100
+ // recorded — marker lines preserved).
101
+ ctx.mapperSpec = await latestSessionNotes(env, "Map", "completed");
102
+ // Rework: the rejecting reviewer's notes, when this Build follows a
103
+ // Review/QA rejection.
104
+ ctx.rejectionNotes = await latestRejectionNotes(env);
105
+ // safeTitle: the commit-message-safe title (source: safeTitle).
106
+ ctx.safeTitle = String(env.taskTitle || "").replace(/"/g, "'").replace(/\\/g, "\\\\").replace(/`/g, "'");
107
+
108
+ var claimed = await ensureClaimed(env, state, PHASE);
109
+ if (claimed.type !== "CLAIMED") return claimed;
110
+
111
+ var boundary = await runWorkBoundary(env, state, {
112
+ phase: PHASE.name,
113
+ identity: PHASE.identity,
114
+ instructions: buildInstructions(ctx),
115
+ eventPreamble: buildEventPreamble(env, PHASE.name),
116
+ crewApiLine: true,
117
+ verdictStep: true,
118
+ });
119
+ if (boundary.type !== "BOUNDARY_DONE") return boundary;
120
+
121
+ var workerText = boundary.workerText;
122
+ var verdictPassed = boundary.verdictPassed === true;
123
+
124
+ // Worktree confinement (verdict PASS only): the declared worktree path
125
+ // must match the hint exactly. A builder that worked outside the
126
+ // configured checkout fails the phase for dispatcher retry.
127
+ if (verdictPassed) {
128
+ var wt = extractWorktree(workerText);
129
+ if (!wt.ok || wt.path !== env.worktreeHint) {
130
+ var declared = wt.ok ? wt.path : "<none>";
131
+ log("Build worktree confinement failed — declared: " + declared + ", expected: " + env.worktreeHint + " — marking failed for retry");
132
+ await recordPhase(env, {
133
+ task_id: env.taskId,
134
+ session: {
135
+ id: state.activeSessionId, task_id: env.taskId, identity: PHASE.identity,
136
+ step: PHASE.name, status: "failed",
137
+ notes: "Build declared worktree " + declared + " — expected " + env.worktreeHint + ". The builder worked outside the configured repo checkout; phase failed for retry",
138
+ },
139
+ event: {
140
+ task_id: env.taskId, type: "failed",
141
+ message: "Build worktree confinement failed — builder worked outside " + env.worktreeHint + ", phase failed, dispatcher will retry",
142
+ },
143
+ });
144
+ return {
145
+ type: "FAILED",
146
+ reason: "Build worked outside the configured repo checkout",
147
+ detail: "The builder declared worktree " + declared + " but the task's worktree is " + env.worktreeHint + ". The phase is marked failed; the dispatcher will retry Build against the configured repo.",
148
+ };
149
+ }
150
+ }
151
+
152
+ // Closeout: a FAIL still falls through to Review (recorded "rejected").
153
+ // The release decision rides in the session notes' marker lines for
154
+ // Review to extract.
155
+ var status = verdictPassed ? "completed" : "rejected";
156
+ await recordPhase(env, {
157
+ task_id: env.taskId,
158
+ session: {
159
+ id: state.activeSessionId, task_id: env.taskId, identity: PHASE.identity,
160
+ step: PHASE.name, status: status, notes: summarizeReport(workerText),
161
+ },
162
+ event: {
163
+ task_id: env.taskId, type: status, identity: PHASE.identity,
164
+ message: "Build " + status + " by " + PHASE.identity,
165
+ },
166
+ });
167
+ return { type: "ADVANCE", next: "Review" };
168
+ }
@@ -0,0 +1,170 @@
1
+ // lib/bugfix/phases/capture.js — Capture (hazel) phase module (sandbox
2
+ // exit, Phase D). Import-safe: no side effects on import, bare `node`
3
+ // exits 0.
4
+ //
5
+ // The capture mechanism is identical to the standard workflow's (same
6
+ // baseline note protocol, same terminal/artifact paths, same park path);
7
+ // the only bugfix difference is the advance target: Capture feeds
8
+ // Reproduce, not Map.
9
+ //
10
+ // Reads: experiential status (durable Triage session notes), surface
11
+ // classification, baseline note events, terminal targets.
12
+ // Writes: exact-prefix `baseline:` note events, completed/failed session +
13
+ // event. The durable cross-run handoff is the `baseline:` notes themselves
14
+ // (Map's gate reads them).
15
+ // Transitions: ADVANCE → Reproduce on every path except the park path
16
+ // (PARK with the baseline-request reason).
17
+ //
18
+ // Worker-layer shape: the artifact-surface paths are fully mechanical
19
+ // (direct Crew API calls). The terminal-surface baseline is creative
20
+ // (Hazel drives the CLI herself, reads transcripts) and crosses the work
21
+ // boundary via NEED_SPAWN; the phase owns the deterministic recording
22
+ // (log-event + record-phase) after the boundary resolves. Map-gate
23
+ // bounce re-visits pass state.gateBounceCount as the spawn visitSuffix so
24
+ // the second visit mints fresh keys.
25
+
26
+ import {
27
+ resolveExperiential, baselineStatus, recordPhase, logEvent, parkTask,
28
+ crewApi, runWorkBoundary, buildEventPreamble, closeoutPassed,
29
+ ensureClaimed, log,
30
+ } from "../../workflow-lib.js";
31
+
32
+ export const PHASE = { name: "Capture", identity: "hazel" };
33
+
34
+ // buildInstructions — the creative terminal-baseline prompt, shared with
35
+ // the standard workflow's Capture (same steps, same machine-read
36
+ // declaration contract). Steps 1–5 verbatim from workflows/bugfix.js
37
+ // (Capture branch); the recording tail is adapted: the worker layer owns
38
+ // log-event / record-phase deterministically, so the agent ends with a
39
+ // machine-read baseline declaration instead of running the recording
40
+ // commands itself.
41
+ export function buildInstructions(ctx) {
42
+ var env = ctx.env, state = ctx.state;
43
+ var termBaseDir = env.taskEvidenceDir + "/baseline";
44
+ var terminalTargets = state.memo.terminalTargets || "";
45
+ return "Run the pre-change terminal baseline yourself — there is no parent capture protocol for terminal-surface projects.\n" +
46
+ "1. The project's repo is at " + env.repoPath + " (pre-change state; the task branch does not exist yet). The CLI under test is the project's own command-line interface in that tree — start like a new user with --help. You are code-blind: you may RUN the CLI, never READ its source.\n" +
47
+ "2. Terminal targets for this task: " + (terminalTargets || "not declared — derive them from --help and the task description") + ".\n" +
48
+ "3. Run in shell: mkdir -p " + termBaseDir + "\n" +
49
+ "4. For each target, run the command with stdout AND stderr captured to " + termBaseDir + "/<nn>-<short-slug>.txt (number them 01, 02, ...), appending the exit code as the final line. Pattern: <cmd> > " + termBaseDir + "/01-<slug>.txt 2>&1; echo \"exit=$?\" >> " + termBaseDir + "/01-<slug>.txt\n" +
50
+ "5. READ every transcript file you wrote — an unread transcript is not evidence.\n" +
51
+ "End your report with exactly one line on its own: either `baseline: captured (terminal transcripts: <comma-separated filenames>)` listing the transcript files you archived, or `baseline: none (terminal targets not runnable: <reason>)` when the CLI cannot run from the pre-change tree — never fabricate a transcript. This line is machine-read.\n" +
52
+ "Report back in plain prose: which commands you ran and what the pre-change baseline looks like.";
53
+ }
54
+
55
+ // buildTerminalBaselineInstructions — legacy alias for buildInstructions
56
+ // (the creative terminal-baseline prompt).
57
+ export function buildTerminalBaselineInstructions(ctx) {
58
+ return buildInstructions(ctx);
59
+ }
60
+
61
+ export async function runPhase(ctx) {
62
+ var env = ctx.env, state = ctx.state;
63
+ var taskId = env.taskId;
64
+ var claimed = await ensureClaimed(env, state, PHASE);
65
+ if (claimed.type !== "CLAIMED") return claimed;
66
+
67
+ var capExp = await resolveExperiential(env, state);
68
+ var surfaceClassified = env.surfaceArtifact || env.surfaceTerminal;
69
+ if (capExp !== "yes" || !surfaceClassified) {
70
+ log("Capture skipped for task " + taskId + " — " + (capExp !== "yes" ? "not experiential" : "surface unclassified"));
71
+ await recordPhase(env, {
72
+ task_id: taskId,
73
+ session: { id: state.activeSessionId, task_id: taskId, identity: PHASE.identity, step: PHASE.name, status: "completed", notes: "Capture skipped — not an experiential task on a classified surface" },
74
+ event: { task_id: taskId, type: "completed", identity: PHASE.identity, message: "Capture completed by " + PHASE.identity },
75
+ });
76
+ return { type: "ADVANCE", next: "Reproduce" };
77
+ }
78
+
79
+ // Terminal-surface Capture: Hazel runs the CLI herself against the
80
+ // pre-change tree and archives attempt-scoped transcripts. No parent
81
+ // protocol exists for terminal projects (there is no see-act loop to
82
+ // drive), so the baseline is agent-driven in one shot — the Map gate
83
+ // reads the same "baseline: captured/none" note prefixes either way.
84
+ if (env.surfaceTerminal) {
85
+ log("Capture: terminal-surface baseline for task " + taskId + " — Hazel drives the CLI herself");
86
+ var visitSuffix = (state.gateBounceCount > 0) ? "-g" + state.gateBounceCount : "";
87
+ var boundary = await runWorkBoundary(env, state, {
88
+ phase: PHASE.name, identity: PHASE.identity,
89
+ instructions: buildInstructions(ctx),
90
+ eventPreamble: buildEventPreamble(env, PHASE.name),
91
+ crewApiLine: true, verdictStep: false,
92
+ visitSuffix: visitSuffix,
93
+ });
94
+ if (boundary.type !== "BOUNDARY_DONE") return boundary;
95
+ var workerText = boundary.workerText || "";
96
+ var decls = [];
97
+ var dm;
98
+ var dRe = /^baseline:\s*(captured|none)\b.*$/gim;
99
+ while ((dm = dRe.exec(workerText)) !== null) decls.push(dm[0].trim());
100
+ if (decls.length === 0 || !closeoutPassed(boundary)) {
101
+ var failNotes = "Terminal baseline agent produced no machine-read baseline declaration — cannot record evidence";
102
+ log("Capture: " + failNotes + " for task " + taskId);
103
+ await recordPhase(env, {
104
+ task_id: taskId,
105
+ session: { id: state.activeSessionId, task_id: taskId, identity: PHASE.identity, step: PHASE.name, status: "failed", notes: failNotes },
106
+ event: { task_id: taskId, type: "failed", identity: PHASE.identity, message: "Capture failed for task " + taskId + " — no baseline declaration" },
107
+ });
108
+ return { type: "FAILED", reason: failNotes };
109
+ }
110
+ var declaration = decls[decls.length - 1];
111
+ var captured = /^baseline:\s*captured\b/i.test(declaration);
112
+ await logEvent(env, { task_id: taskId, type: "note", identity: PHASE.identity, message: declaration });
113
+ await recordPhase(env, {
114
+ task_id: taskId,
115
+ session: { id: state.activeSessionId, task_id: taskId, identity: PHASE.identity, step: PHASE.name, status: "completed",
116
+ notes: captured ? "Terminal baseline captured by the QA agent (transcripts archived)" : "baseline: none — terminal targets not runnable, final QA judges on rubric alone" },
117
+ event: { task_id: taskId, type: "completed", identity: PHASE.identity, message: "Capture completed by " + PHASE.identity },
118
+ });
119
+ return { type: "ADVANCE", next: "Reproduce" };
120
+ }
121
+
122
+ var capStatus = await baselineStatus(env);
123
+ // See docs/decisions/qa-reproduce.md#stale-decision-guard: a stale baseline decision parks fail-closed.
124
+ var baselineLatestMessage = ("baseline: " + capStatus.baseline_kind + capStatus.baseline_refs).trim();
125
+ var baselineStale = env.visualProtocolAvailable && baselineLatestMessage === "baseline: none (visual protocol unavailable)";
126
+ if (capStatus.baseline_found && !baselineStale) {
127
+ var evidenceN = capStatus.evidence_count + 1;
128
+ var carryNote = "baseline: " + capStatus.baseline_kind + " (#" + evidenceN + " carries forward prior:" + capStatus.baseline_refs + ")";
129
+ log("Capture: baseline evidence already recorded for task " + taskId + " (" + capStatus.baseline_kind + ") — logging per-attempt carry-forward note (evidence #" + evidenceN + ")");
130
+ await logEvent(env, { task_id: taskId, type: "note", identity: PHASE.identity, message: carryNote });
131
+ await recordPhase(env, {
132
+ task_id: taskId,
133
+ session: { id: state.activeSessionId, task_id: taskId, identity: PHASE.identity, step: PHASE.name, status: "completed",
134
+ notes: "Baseline evidence already recorded: " + capStatus.baseline_kind + " " + capStatus.baseline_refs + " (evidence #" + evidenceN + ")" },
135
+ event: { task_id: taskId, type: "completed", identity: PHASE.identity, message: "Capture completed by " + PHASE.identity },
136
+ });
137
+ return { type: "ADVANCE", next: "Reproduce" };
138
+ }
139
+
140
+ var attemptN = capStatus.requested_count + 1;
141
+ if (capStatus.requested_count >= 2) {
142
+ log("Capture: baseline capture unavailable after 2 requests for task " + taskId + " — recording baseline:none");
143
+ await logEvent(env, { task_id: taskId, type: "note", identity: PHASE.identity, message: "baseline: none (capture unavailable after 2 attempts)" });
144
+ await recordPhase(env, {
145
+ task_id: taskId,
146
+ session: { id: state.activeSessionId, task_id: taskId, identity: PHASE.identity, step: PHASE.name, status: "completed", notes: "baseline: none — final QA judges on rubric alone" },
147
+ event: { task_id: taskId, type: "completed", identity: PHASE.identity, message: "Capture completed by " + PHASE.identity },
148
+ });
149
+ return { type: "ADVANCE", next: "Reproduce" };
150
+ }
151
+
152
+ log("Capture: requesting baseline capture (attempt " + attemptN + ") for task " + taskId);
153
+ await logEvent(env, { task_id: taskId, type: "note", identity: PHASE.identity, message: "baseline: requested (attempt " + attemptN + ")" });
154
+ await crewApi(env, "upsert-session", {
155
+ id: state.activeSessionId, task_id: taskId, identity: PHASE.identity, step: PHASE.name, status: "failed",
156
+ notes: "Baseline capture requested (attempt " + attemptN + ") — parent protocol: run the baseline capture protocol in docs/visual-verdict.md, then re-queue the task at Map. The Capture step re-checks baseline evidence when re-run.",
157
+ });
158
+ if (!env.visualProtocolAvailable) {
159
+ log("Capture: visual protocol not available (VISUAL_PROTOCOL_AVAILABLE=false) — recording baseline:none instead of parking for task " + taskId);
160
+ await logEvent(env, { task_id: taskId, type: "note", identity: PHASE.identity, message: "baseline: none (visual protocol unavailable)" });
161
+ await recordPhase(env, {
162
+ task_id: taskId,
163
+ session: { id: state.activeSessionId, task_id: taskId, identity: PHASE.identity, step: PHASE.name, status: "completed", notes: "baseline: none — visual protocol unavailable, final QA judges on rubric alone" },
164
+ event: { task_id: taskId, type: "completed", identity: PHASE.identity, message: "Capture completed by " + PHASE.identity },
165
+ });
166
+ return { type: "ADVANCE", next: "Reproduce" };
167
+ }
168
+ var parked = await parkTask(env, "Baseline capture requested (attempt " + attemptN + ") — parent: run the baseline capture protocol in docs/visual-verdict.md, then re-queue the task at Map");
169
+ return parked;
170
+ }