muse-crew 0.14.6 → 0.14.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/merge-lock.sh CHANGED
@@ -19,6 +19,20 @@
19
19
  # (or malformed) lease is broken with logging and re-acquired. Every op
20
20
  # appends one line to $CREW_HOME/.merge-lock.log (the audit trail).
21
21
  #
22
+ # Stale-break atomicity (R-B1, 2026-09-21): the break's read→rm→create is
23
+ # serialized on a sidecar flock ($LOCK_FILE.flock) and the lease is re-read
24
+ # inside the critical section — the old unconditional rm before the noclobber
25
+ # create let two concurrent reclaimers both win. Refresh rewrites via
26
+ # temp-file + atomic rename so a concurrent reader never sees a torn file,
27
+ # and takes the same sidecar flock so a refresh can never clobber a
28
+ # reclaimer's fresh lock in its read→mv gap (2026-09-21): refresh's
29
+ # identity-only re-check closes the clobber because a reclaim always changes
30
+ # task_id. Refresh never checks expiry — a long build that outran the lease
31
+ # legitimately revives its lock here (publish-npm.sh step 5b relies on it).
32
+ # Release (review pass 2, 2026-09-21) takes the same sidecar flock with the
33
+ # identity re-check inside the critical section, so a release can never rm
34
+ # a reclaimer's fresh lock in its read→rm gap.
35
+ #
22
36
  # Usage:
23
37
  # merge-lock.sh acquire <task_id> <holder> — acquire the lock (0 ok, 1 held)
24
38
  # merge-lock.sh refresh <task_id> — extend the lease (holder task only)
@@ -44,6 +58,7 @@ fi
44
58
 
45
59
  LOCK_DIR="$CREW_REPO/.worktrees"
46
60
  LOCK_FILE="$LOCK_DIR/.merge-lock"
61
+ FLOCK_FILE="$LOCK_FILE.flock" # sidecar serializing stale-lease breaks (R-B1) and refresh rewrites (2026-09-21)
47
62
  LOG_FILE="$CREW_HOME/.merge-lock.log"
48
63
  LEASE_SECONDS="${MERGE_LOCK_LEASE_SECONDS:-600}"
49
64
 
@@ -127,55 +142,152 @@ case "$cmd" in
127
142
  exit 1
128
143
  fi
129
144
  # Expired or malformed lease — break it (logged) and re-acquire.
145
+ # R-B1 (2026-09-21): read→rm→create is serialized on the sidecar flock.
146
+ # The rm used to run before the noclobber create outside any atomic op,
147
+ # so two concurrent reclaimers could both win: B read the stale lease,
148
+ # A reclaimed (rm + create, exit 0), B's rm deleted A's fresh lock, and
149
+ # B's create then succeeded too. The lease is re-read INSIDE the
150
+ # critical section — a contender may have refreshed, released, or
151
+ # reclaimed while we waited on the flock.
130
152
  echo "STALE: lease for ${lock_task_id:-unknown} (holder ${lock_holder:-unknown}) expired — breaking" >&2
131
153
  log_op acquire "task_id=$task_id holder=$holder result=STALE-BREAK previous_task=${lock_task_id:-unknown} previous_holder=${lock_holder:-unknown}"
132
- rm -f "$LOCK_FILE"
133
- if write_lock "$task_id" "$holder"; then
134
- log_op acquire "task_id=$task_id holder=$holder result=ACQUIRED-RECLAIMED previous_task=${lock_task_id:-unknown}"
135
- echo "ACQUIRED by $task_id (holder $holder; reclaimed expired lease from ${lock_task_id:-unknown})"
136
- exit 0
137
- fi
138
- # Lost the race: whoever won the create holds it now.
139
- read_lock 2>/dev/null || true
140
- log_op acquire "task_id=$task_id holder=$holder result=HELD-RACE holder_task=${lock_task_id:-unknown}"
141
- echo "HELD by ${lock_task_id:-unknown} (holder ${lock_holder:-unknown}) — lost reclaim race"
142
- exit 1
154
+ stale_task="${lock_task_id:-unknown}"
155
+ mkdir -p "$LOCK_DIR"
156
+ # Subshell exit codes: 0 reclaimed; 10 the re-read found a live lease;
157
+ # 11 lost the noclobber create to a fresh acquirer racing the break.
158
+ reclaim_code=0
159
+ (
160
+ exec 200>"$FLOCK_FILE"
161
+ flock -x 200
162
+ if read_lock; then
163
+ recheck="$(lock_remaining)"
164
+ if [ "$recheck" != "unknown" ] && [ "$recheck" -gt 0 ]; then
165
+ exit 10
166
+ fi
167
+ fi
168
+ rm -f "$LOCK_FILE"
169
+ if write_lock "$task_id" "$holder"; then
170
+ exit 0
171
+ else
172
+ exit 11
173
+ fi
174
+ ) || reclaim_code=$?
175
+ case "$reclaim_code" in
176
+ 0)
177
+ log_op acquire "task_id=$task_id holder=$holder result=ACQUIRED-RECLAIMED previous_task=$stale_task"
178
+ echo "ACQUIRED by $task_id (holder $holder; reclaimed expired lease from $stale_task)"
179
+ exit 0
180
+ ;;
181
+ 10)
182
+ # Re-read found a live lease inside the critical section: another
183
+ # contender refreshed or reclaimed while we waited on the flock.
184
+ read_lock 2>/dev/null || true
185
+ if [ -n "$lock_task_id" ]; then
186
+ live_remaining="$(lock_remaining)"
187
+ log_op acquire "task_id=$task_id holder=$holder result=HELD-RECHECK holder_task=$lock_task_id holder_id=$lock_holder remaining=${live_remaining}s"
188
+ echo "HELD by $lock_task_id (holder $lock_holder, ${live_remaining}s of lease remaining)"
189
+ else
190
+ log_op acquire "task_id=$task_id holder=$holder result=HELD-RECHECK holder_task=unknown"
191
+ echo "HELD — lost the reclaim race to another contender"
192
+ fi
193
+ exit 1
194
+ ;;
195
+ *)
196
+ # Lost the noclobber create to a fresh acquirer racing the break:
197
+ # whoever won the create holds the lock now.
198
+ read_lock 2>/dev/null || true
199
+ log_op acquire "task_id=$task_id holder=$holder result=HELD-RACE holder_task=${lock_task_id:-unknown}"
200
+ echo "HELD by ${lock_task_id:-unknown} (holder ${lock_holder:-unknown}) — lost reclaim race"
201
+ exit 1
202
+ ;;
203
+ esac
143
204
  ;;
144
205
  refresh)
145
206
  [ -z "$task_id" ] && { echo "ERROR: task_id required"; exit 1; }
146
- if ! read_lock; then
147
- log_op refresh "task_id=$task_id result=NO-LOCK"
148
- echo "ERROR: no lock to refresh"
149
- exit 1
150
- fi
151
- if [ "$lock_task_id" != "$task_id" ]; then
152
- log_op refresh "task_id=$task_id result=NOT-HOLDER holder_task=$lock_task_id"
153
- echo "ERROR: lock held by $lock_task_id, not $task_id"
154
- exit 1
155
- fi
156
- # Extend the lease: acquired_at=now, holder and lease window unchanged.
157
- printf 'task_id=%s\nholder=%s\nacquired_at=%s\nlease_seconds=%s\n' \
158
- "$lock_task_id" "$lock_holder" "$(date +%s)" "${lock_lease_seconds:-$LEASE_SECONDS}" > "$LOCK_FILE"
159
- log_op refresh "task_id=$task_id holder=$lock_holder result=REFRESHED"
160
- echo "REFRESHED by $task_id (holder $lock_holder)"
161
- exit 0
207
+ # 2026-09-21: read→identity-check→rewrite is serialized on the sidecar
208
+ # flock against acquire's reclaim — a holder refreshing after lease
209
+ # expiry could otherwise clobber a reclaimer's fresh lock inside the
210
+ # read→mv gap. The re-check inside the critical section is IDENTITY-ONLY:
211
+ # a reclaim always changes task_id, so identity alone closes the clobber.
212
+ # There is deliberately NO expiry fail-closed check — a long build that
213
+ # outran the lease legitimately revives its lock here (publish-npm.sh
214
+ # step 5b and the workflow Publish prompt rely on it). Fail closed (exit
215
+ # 1) only if identity no longer matches.
216
+ refresh_code=0
217
+ (
218
+ exec 200>"$FLOCK_FILE"
219
+ flock -x 200
220
+ if ! read_lock; then
221
+ log_op refresh "task_id=$task_id result=NO-LOCK"
222
+ echo "ERROR: no lock to refresh"
223
+ exit 1
224
+ fi
225
+ if [ "$lock_task_id" != "$task_id" ]; then
226
+ log_op refresh "task_id=$task_id result=NOT-HOLDER holder_task=$lock_task_id"
227
+ echo "ERROR: lock held by $lock_task_id, not $task_id"
228
+ exit 1
229
+ fi
230
+ # Extend the lease: acquired_at=now, holder and lease window unchanged.
231
+ # Write-temp + atomic rename (2026-09-21): a truncating rewrite lets a
232
+ # concurrent reader observe a torn (partially written) file and misread
233
+ # the lease as malformed — funneling it spuriously into the reclaim path.
234
+ tmp_lock="$(mktemp "$LOCK_DIR/.merge-lock.tmp.XXXXXX")" \
235
+ || { echo "ERROR: cannot stage lock refresh"; exit 1; }
236
+ printf 'task_id=%s\nholder=%s\nacquired_at=%s\nlease_seconds=%s\n' \
237
+ "$lock_task_id" "$lock_holder" "$(date +%s)" "${lock_lease_seconds:-$LEASE_SECONDS}" > "$tmp_lock"
238
+ mv -f "$tmp_lock" "$LOCK_FILE"
239
+ log_op refresh "task_id=$task_id holder=$lock_holder result=REFRESHED"
240
+ echo "REFRESHED by $task_id (holder $lock_holder)"
241
+ exit 0
242
+ ) || refresh_code=$?
243
+ exit "$refresh_code"
162
244
  ;;
163
245
  release)
164
246
  [ -z "$task_id" ] && { echo "ERROR: task_id required"; exit 1; }
165
- if ! read_lock; then
166
- log_op release "task_id=$task_id result=NOT-LOCKED"
167
- echo "RELEASED (was not locked)"
168
- exit 0
169
- fi
170
- if [ "$lock_task_id" = "$task_id" ]; then
171
- rm -f "$LOCK_FILE"
172
- log_op release "task_id=$task_id holder=$lock_holder result=RELEASED"
173
- echo "RELEASED by $task_id"
174
- exit 0
175
- fi
176
- log_op release "task_id=$task_id result=NOT-HOLDER holder_task=$lock_task_id"
177
- echo "ERROR: lock held by $lock_task_id, not $task_id"
178
- exit 1
247
+ # 2026-09-21 (review pass 2): read→identity-check→rm is serialized on the
248
+ # sidecar flock — the same double-hold race class as R-B1 is reachable
249
+ # here: releaser reads, reclaimer breaks and creates, releaser's rm then
250
+ # deletes the reclaimer's fresh lock. The identity re-check runs INSIDE
251
+ # the critical section.
252
+ release_code=0
253
+ # The identity check and rm run in a subshell, so the holder name is
254
+ # ferried out on stdout (subshell variables never reach the parent).
255
+ release_out="$(
256
+ (
257
+ exec 200>"$FLOCK_FILE"
258
+ flock -x 200
259
+ if ! read_lock; then
260
+ exit 10
261
+ fi
262
+ if [ "$lock_task_id" != "$task_id" ]; then
263
+ exit 11
264
+ fi
265
+ printf 'holder=%s\n' "$lock_holder"
266
+ rm -f "$LOCK_FILE"
267
+ exit 0
268
+ )
269
+ )" || release_code=$?
270
+ release_holder="${release_out#holder=}"
271
+ case "$release_code" in
272
+ 0)
273
+ log_op release "task_id=$task_id holder=$release_holder result=RELEASED"
274
+ echo "RELEASED by $task_id"
275
+ exit 0
276
+ ;;
277
+ 10)
278
+ log_op release "task_id=$task_id result=NOT-LOCKED"
279
+ echo "RELEASED (was not locked)"
280
+ exit 0
281
+ ;;
282
+ *)
283
+ # Re-read inside the critical section: identity checked against the
284
+ # lock as it stands now, not as it stood before the flock.
285
+ read_lock 2>/dev/null || true
286
+ log_op release "task_id=$task_id result=NOT-HOLDER holder_task=${lock_task_id:-unknown}"
287
+ echo "ERROR: lock held by ${lock_task_id:-unknown}, not $task_id"
288
+ exit 1
289
+ ;;
290
+ esac
179
291
  ;;
180
292
  status)
181
293
  if ! read_lock; then
package/lib/qa-db.js ADDED
@@ -0,0 +1,132 @@
1
+ // qa-db.js — fresh per-run QA database for the local artifact server.
2
+ //
3
+ // Import-safe: no side effects on import. `node qa-db.js` with no args exits
4
+ // 0 (the release entry gate executes bare non-CLI lib modules).
5
+ //
6
+ // `openQaDb(spaceDir)` opens a drizzle db over a FRESH per-run SQLite
7
+ // database, migrated from the space's own `drizzle/` migrations in journal
8
+ // order — the same migration path a fresh production install takes — and
9
+ // never a copy of the shipped app.db. The database file lives in a per-run
10
+ // temp dir, so the server stays read-only w.r.t. the space directory and QA
11
+ // writes can never contaminate production data (the B38 contamination
12
+ // class: audit sessions writing verification data into the live artifact's
13
+ // production DB).
14
+ //
15
+ // Driver: drizzle-orm/sqlite-proxy resolved from the space's own
16
+ // node_modules (so the driver matches the artifact's drizzle version), over
17
+ // node:sqlite (built-in). Zero extra dependencies in the crew release.
18
+ //
19
+ // Proxy contract, verified mechanically against drizzle-orm 0.45.2's
20
+ // COMPILED runtime (sqlite-proxy/driver.js, sqlite-proxy/session.js,
21
+ // utils.js — not just the .d.ts):
22
+ // - utils.js mapResultRow(columns, row, ...) reads row[columnIndex]:
23
+ // rows must be POSITIONAL value arrays in SQL column order. The proxy
24
+ // prepares with { returnArrays: true }, so node:sqlite returns rows
25
+ // as positional arrays — duplicate column names (self-joins, t.*,
26
+ // joins of overlapping schemas) survive by ordinal. The object-keyed
27
+ // default would silently collapse them.
28
+ // - session.js get(): mapGetResult(clientResult.rows) — a miss must be
29
+ // FALSY rows (returns undefined); a hit must be the single row as a
30
+ // positional array (it is NOT wrapped — mapGetResult treats rows as the
31
+ // row itself).
32
+ // - session.js all()/values(): { rows } is destructured and .map'ed, so
33
+ // rows must be an array of positional arrays.
34
+ // - session.js run(): the callback's return is handed to the caller; the
35
+ // declared type is Promise<{rows: any[]}> — {rows: []} is the honest
36
+ // empty shape.
37
+ // - session.js batch(): batchResults.map((result, i) =>
38
+ // preparedQueries[i].mapResult(result, true)) — each item must be one
39
+ // per-query {rows} object in order; mapResult with isFromBatch unwraps
40
+ // rows.rows before the same get/all mapping above.
41
+ // - driver.js drizzle(callback, batchCallback?, config?): the second
42
+ // positional arg is the batch callback when it is a function.
43
+
44
+ import { DatabaseSync } from "node:sqlite";
45
+ import { createRequire } from "node:module";
46
+ import { pathToFileURL } from "node:url";
47
+ import { join } from "node:path";
48
+ import { existsSync } from "node:fs";
49
+ import { readFile } from "node:fs/promises";
50
+ import { mkdtemp } from "node:fs/promises";
51
+ import { tmpdir } from "node:os";
52
+
53
+ // sqlite-proxy callback over a node:sqlite DatabaseSync. See the contract
54
+ // note at the top of this file.
55
+ export function sqliteProxyCallback(sqlite) {
56
+ const one = async (sql, params, method) => {
57
+ // node:sqlite Statements have no close(); the DatabaseSync owns them
58
+ // and they are reclaimed with it.
59
+ // returnArrays: rows come back as POSITIONAL value arrays, so
60
+ // duplicate column names (self-joins, t.*, joins of overlapping
61
+ // schemas) survive by ordinal. The object-keyed default would
62
+ // silently collapse them — Object.values() would drop the earlier
63
+ // columns entirely.
64
+ const stmt = sqlite.prepare(sql, { returnArrays: true });
65
+ if (method === "run") {
66
+ stmt.run(...params);
67
+ return { rows: [] };
68
+ }
69
+ if (method === "all" || method === "values") {
70
+ return { rows: stmt.all(...params) };
71
+ }
72
+ if (method === "get") {
73
+ const row = stmt.get(...params);
74
+ return row === undefined ? { rows: undefined } : { rows: row };
75
+ }
76
+ throw new Error("sqlite-proxy: unknown method " + method);
77
+ };
78
+ return one;
79
+ }
80
+
81
+ // Fresh per-run QA database for spaceDir. Resolves to
82
+ // { db, close } — db is the drizzle instance handed to ctx.db(); close()
83
+ // releases the sqlite handle (best effort on kill). Throws with a clear
84
+ // reason when the space has no drizzle migrations or drizzle-orm cannot be
85
+ // resolved from its node_modules; callers degrade gracefully (the server
86
+ // still boots, ctx.db throws the reason when called).
87
+ export async function openQaDb(spaceDir) {
88
+ const drizzleDir = join(spaceDir, "drizzle");
89
+ const journalPath = join(drizzleDir, "meta", "_journal.json");
90
+ if (!existsSync(journalPath)) {
91
+ throw new Error("no drizzle migrations at " + journalPath);
92
+ }
93
+ const journal = JSON.parse(await readFile(journalPath, "utf8"));
94
+ const entries = [...(journal.entries || [])].sort((a, b) => a.idx - b.idx);
95
+ if (entries.length === 0) {
96
+ throw new Error("empty drizzle journal at " + journalPath);
97
+ }
98
+ // Per-run temp dir — never the space dir, never the shipped app.db.
99
+ const dbDir = await mkdtemp(join(tmpdir(), "serve-artifact-db-"));
100
+ const sqlite = new DatabaseSync(join(dbDir, "qa.db"));
101
+ try {
102
+ sqlite.exec("PRAGMA foreign_keys = ON;");
103
+ for (const entry of entries) {
104
+ const sqlPath = join(drizzleDir, entry.tag + ".sql");
105
+ sqlite.exec(await readFile(sqlPath, "utf8"));
106
+ }
107
+ } catch (err) {
108
+ try { sqlite.close(); } catch { /* best effort */ }
109
+ throw new Error("migration failed: " + String((err && err.message) || err).slice(0, 200));
110
+ }
111
+ // drizzle-orm/sqlite-proxy resolved from the space's own node_modules, so
112
+ // the driver matches the artifact's drizzle version.
113
+ let db;
114
+ try {
115
+ const spaceRequire = createRequire(join(spaceDir, "package.json"));
116
+ const proxyEntry = spaceRequire.resolve("drizzle-orm/sqlite-proxy");
117
+ const { drizzle } = await import(pathToFileURL(proxyEntry).href);
118
+ const one = sqliteProxyCallback(sqlite);
119
+ db = drizzle(one, async (items) => {
120
+ const out = [];
121
+ for (const item of items) out.push(await one(item.sql, item.params, item.method));
122
+ return out;
123
+ });
124
+ } catch (err) {
125
+ try { sqlite.close(); } catch { /* best effort */ }
126
+ throw new Error("cannot build drizzle db: " + String((err && err.message) || err).slice(0, 200));
127
+ }
128
+ return {
129
+ db,
130
+ close() { try { sqlite.close(); } catch { /* best effort on kill */ } },
131
+ };
132
+ }
package/lib/schema.sql CHANGED
@@ -64,6 +64,17 @@ CREATE TABLE IF NOT EXISTS tasks (
64
64
  deps TEXT NOT NULL DEFAULT '[]',
65
65
  filed_by TEXT,
66
66
  retry_reset_at TEXT,
67
+ -- Cascade-park attribution (room #26 blocker 36, 2026-09-21). When a task
68
+ -- parks, its todo dependents cascade-park with structured attribution on
69
+ -- the existing 'parked' state — never a new state, never silent todo,
70
+ -- never auto-waived. park_reason is a closed single-column CHECK
71
+ -- ('dep_parked' only); park_dep_id names the parked dep. NULL on both =
72
+ -- ordinary park (human hold). The pairing invariant (reason<->dep)
73
+ -- cannot be a schema CHECK — SQLite forbids cross-column CHECKs on
74
+ -- ADD COLUMN — so it is enforced by crew-api.js, the single writer of
75
+ -- these columns.
76
+ park_reason TEXT CHECK (park_reason IS NULL OR park_reason = 'dep_parked'),
77
+ park_dep_id TEXT,
67
78
  created_at TEXT NOT NULL,
68
79
  updated_at TEXT NOT NULL
69
80
  );
@@ -108,6 +119,45 @@ CREATE TABLE IF NOT EXISTS agent_sessions (
108
119
  (already_merged_sha NOT GLOB '*[^0-9a-f]*' AND length(already_merged_sha) BETWEEN 7 AND 40))
109
120
  );
110
121
 
122
+ -- Room #26 blocker 33 (2026-09-21): structured, non-lossy verdict records.
123
+ -- Review verdict grounds were destroyed by the 2000-char session-note
124
+ -- truncation (the workflow slices the worker report before record-phase
125
+ -- writes it as notes) — a park on a verified-correct implementation was
126
+ -- unauditable because the grounds past char 2000 existed nowhere. The fib:
127
+ -- "the review notes are the review record." The notes are lossy by design
128
+ -- (downstream consumers read them: rejectionNotes, the QA backstop, the
129
+ -- dashboard); this table is the lossless record. record-phase optionally
130
+ -- carries the full grounds and writes the row in the same transaction as
131
+ -- the session note, so the record can never be missing when the note
132
+ -- exists. `summary` is the truncated note actually recorded (provenance of
133
+ -- what the lossy path carried); `grounds` is the full worker report,
134
+ -- uncapped. One verdict per recording session: a re-recorded phase for the
135
+ -- same session is the same evidence, so the first write wins (ON CONFLICT
136
+ -- DO NOTHING) and rows are never revised. A new Review execution claims a
137
+ -- new session, so rework rounds and dispatcher retries each get their own
138
+ -- row, ordered by created_at.
139
+ -- `verdict` is the machine-extracted verdict; INDETERMINATE is the
140
+ -- fail-closed extraction failure (re-ask exhausted) — the grounds are still
141
+ -- preserved even though no verdict could be read.
142
+ CREATE TABLE IF NOT EXISTS verdicts (
143
+ id TEXT PRIMARY KEY,
144
+ task_id TEXT NOT NULL REFERENCES tasks(id) ON DELETE CASCADE,
145
+ step TEXT NOT NULL,
146
+ attempt INTEGER NOT NULL CHECK (attempt >= 0),
147
+ reviewer TEXT NOT NULL,
148
+ verdict TEXT NOT NULL CHECK (verdict IN ('PASS', 'FAIL', 'INDETERMINATE')),
149
+ grounds TEXT NOT NULL,
150
+ summary TEXT NOT NULL,
151
+ -- Room #26 blocker 34 (redesign): what the review actually examined —
152
+ -- mechanical-fail | branch-diff | runtime-state-none | frozen-merge:<sha>.
153
+ -- NULL = not a classified review (e.g. docs.js path, legacy rows).
154
+ review_basis TEXT,
155
+ session_id TEXT NOT NULL,
156
+ created_at TEXT NOT NULL,
157
+ UNIQUE(session_id)
158
+ );
159
+ CREATE INDEX IF NOT EXISTS idx_verdicts_task ON verdicts(task_id);
160
+
111
161
  CREATE TABLE IF NOT EXISTS events (
112
162
  id TEXT PRIMARY KEY,
113
163
  type TEXT NOT NULL CHECK (type IN
@@ -14,9 +14,17 @@
14
14
  //
15
15
  // Fidelity notes (what this is and isn't):
16
16
  // - The served client and action handlers are the artifact's own built code.
17
- // - The Ctx is locally built: privileged handlers run from the space's own
17
+ // - The Ctx is locally built: ctx.db is a drizzle db over a fresh per-run
18
+ // SQLite database, migrated from the space's own drizzle/ migrations (the
19
+ // same migration path a fresh production install takes) — never the
20
+ // shipped app.db, so the server stays read-only w.r.t. the space
21
+ // directory and QA writes can never contaminate production data;
22
+ // privileged handlers run from the space's own
18
23
  // server/dist/privileged.js when present; blobs are stored in a per-run
19
- // temp dir and served back at /__blobs/<key>.
24
+ // temp dir and served back at /__blobs/<key>. When the space has no
25
+ // drizzle migrations (or drizzle-orm can't be resolved from its
26
+ // node_modules), the server still boots and ctx.db throws a clear error
27
+ // when called.
20
28
  // - Environment (CREW_HOME and friends) is inherited from the caller — export
21
29
  // what the artifact's server needs before starting this.
22
30
  //
@@ -27,6 +35,7 @@ import { join, normalize, dirname, sep } from "node:path";
27
35
  import { existsSync } from "node:fs";
28
36
  import { readFile, writeFile, mkdir, mkdtemp } from "node:fs/promises";
29
37
  import { tmpdir } from "node:os";
38
+ import { openQaDb } from "./qa-db.js";
30
39
 
31
40
  function parseArgs(argv) {
32
41
  const out = { spaceDir: null, port: 0, tag: null };
@@ -85,6 +94,23 @@ async function main() {
85
94
  privilegedByName = new Map(privilegedHandlers.entries.map((e) => [e.contract.name, e.handler]));
86
95
  }
87
96
 
97
+ // ctx.db: best-effort. A space with drizzle/ migrations gets a working db
98
+ // (fresh per-run, migrated from the space's own migrations — the harness
99
+ // can finally drive db-backed success paths); anything else still boots
100
+ // and ctx.db throws a clear error when called, instead of 500ing as
101
+ // "not a function".
102
+ let dbAccessor;
103
+ let qaHandle = null;
104
+ try {
105
+ const qa = await openQaDb(args.spaceDir);
106
+ dbAccessor = () => qa.db;
107
+ qaHandle = qa;
108
+ } catch (err) {
109
+ const reason = String((err && err.message) || err);
110
+ process.stderr.write("serve-artifact.js: ctx.db unavailable: " + reason.slice(0, 200) + "\n");
111
+ dbAccessor = () => { throw new Error("ctx.db is not available in the local QA server: " + reason); };
112
+ }
113
+
88
114
  // Local blob store (outside the space dir — the server stays read-only
89
115
  // w.r.t. the space). Keys are path-safe segments; anything else is rejected.
90
116
  const BLOB_DIR = await mkdtemp(join(tmpdir(), "serve-artifact-blobs-"));
@@ -100,6 +126,7 @@ async function main() {
100
126
  function makeCtx(def) {
101
127
  const declared = new Set(((def && def.privileged) || []).map((c) => c.name));
102
128
  return {
129
+ db: dbAccessor,
103
130
  async executePrivileged(contract, actionArgs) {
104
131
  if (!contract || typeof contract.name !== "string") throw new Error("executePrivileged: bad contract descriptor");
105
132
  if (!declared.has(contract.name)) throw new Error("executePrivileged(" + contract.name + ") was not declared by this action");
@@ -201,6 +228,23 @@ async function main() {
201
228
  server.listen(args.port, "127.0.0.1", () => {
202
229
  process.stdout.write("READY port=" + server.address().port + "\n");
203
230
  });
231
+
232
+ // Graceful shutdown: SIGTERM/SIGINT close the listener and the QA db, then
233
+ // exit. (An earlier revision only closed the db on signal — the process
234
+ // ignored termination and leaked. Caught by tests/serve-artifact-db.test.js.)
235
+ let shuttingDown = false;
236
+ for (const sig of ["SIGTERM", "SIGINT"]) {
237
+ process.on(sig, () => {
238
+ if (shuttingDown) return;
239
+ shuttingDown = true;
240
+ const finish = () => {
241
+ try { if (qaHandle) qaHandle.close(); } catch {}
242
+ process.exit(0);
243
+ };
244
+ server.close(finish);
245
+ setTimeout(finish, 1500).unref();
246
+ });
247
+ }
204
248
  }
205
249
 
206
250
  main().catch((e) => {
@@ -50,6 +50,18 @@
50
50
  # crash): re-running integrate takes the lock and pushes instead of
51
51
  # parking on MERGED_EMPTY.
52
52
  #
53
+ # Fixture F3 (stale-record recovery — blocker 35, 2026-09-21): rework moved
54
+ # the task branch backward after the merge was recorded, so the recorded
55
+ # merge carries abandoned work. Re-running integrate must fail closed
56
+ # with STALE_MERGE — never push the stale state, never fall through to
57
+ # MERGED_EMPTY (an unpushed merge sits on the line).
58
+ #
59
+ # Fixture I (push-target identity — blocker 35, 2026-09-21): the
60
+ # ERROR-after-MERGED retry path. I1: stale record → STALE_MERGE and the
61
+ # remote ref does not move. I2: branch restored to the merged tip → the
62
+ # legitimate retry still PUSHEDs. I3: the manual R5 shape (stale record
63
+ # plus a newer manual merge of the current tip on the line) → PUSHEDs.
64
+ #
53
65
  # Fixture H (detached HEAD with an origin — 2026-09-19 detached-HEAD audit
54
66
  # REDO): H1 proves push-destination fails closed when origin/HEAD is unset
55
67
  # (genuinely unknowable destination); H2-H5 prove the detached line pushes
@@ -327,6 +339,116 @@ echo "$out" | grep -q '^MERGED_EMPTY' && fail "F2: wrongly MERGED_EMPTY while th
327
339
 
328
340
  echo "fixture F (merged-but-unpushed recovery): all checks passed"
329
341
 
342
+ # ---------------- Fixture F3: stale record — branch moved after the merge ----------------
343
+ # (blocker 35, room #26 J3, 2026-09-21): rework moved the task branch
344
+ # BACKWARD after the merge was recorded (the recorded merge now carries
345
+ # abandoned work). The recovery must NOT push the stale recorded state
346
+ # labeled as the deliverable — STALE_MERGE fails closed, and it must not
347
+ # fall through to MERGED_EMPTY either (an unpushed merge sits on the
348
+ # line; MERGED_EMPTY would report PASS while a later publish diff would
349
+ # still carry the abandoned work).
350
+ # Reuse fixture D's repo (no remote): new task, merge, then reset the
351
+ # branch to its pre-work base.
352
+ cd "$D_DIR"
353
+ export CREW_REPO="$D_DIR"
354
+ export CREW_HOME="$T/home-d"
355
+ tid="stale1"
356
+ # Fixture F2's integrate left noremote1's lock held — release it so this
357
+ # task's integrate doesn't back off on a foreign holder.
358
+ "$REPO_DIR/lib/merge-lock.sh" release "noremote1" >/dev/null 2>&1 || fail "F3: lock release failed"
359
+ out=$(bash "$LIFECYCLE" prepare "$tid") || fail "F3 prepare failed: $out"
360
+ BRANCH=$(bash "$LIFECYCLE" resolve-branch "$tid") || fail "F3 resolve-branch failed"
361
+ F3_BASE=$(git rev-parse "$BRANCH")
362
+ echo work >> "$D_DIR/.worktrees/$tid/f.txt"
363
+ (cd "$D_DIR/.worktrees/$tid" && git commit -qam "work w1")
364
+ F3_W1=$(git rev-parse "$BRANCH")
365
+ git worktree remove --force "$D_DIR/.worktrees/$tid" || fail "F3: worktree remove failed"
366
+ out=$(bash "$LIFECYCLE" integrate "$tid" "merge: $tid" 2>&1) || fail "F3 integrate failed: $out"
367
+ echo "$out" | grep -q '^MERGED:' || fail "F3: expected MERGED, got: $out"
368
+ F3_M1=$(git rev-parse HEAD)
369
+ # Rework abandons w1: the branch goes back to its pre-work base.
370
+ git branch -q -f "$BRANCH" "$F3_BASE" || fail "F3: branch reset failed"
371
+ [ "$(git rev-parse "$BRANCH")" = "$F3_BASE" ] || fail "F3: branch not reset"
372
+ "$REPO_DIR/lib/merge-lock.sh" release "$tid" >/dev/null 2>&1 || fail "F3: lock release failed"
373
+ out=$(bash "$LIFECYCLE" integrate "$tid" "merge: $tid" 2>&1) && \
374
+ fail "F3: integrate with a stale record should fail, got: $out"
375
+ echo "$out" | grep -q '^STALE_MERGE' || fail "F3: expected STALE_MERGE, got: $out"
376
+ echo "$out" | grep -q '^MERGED_EMPTY' && fail "F3: fell through to MERGED_EMPTY on a stale record: $out"
377
+ echo "$out" | grep -q 'recovered:' && fail "F3: recovered a stale merge: $out"
378
+ [ "$(git rev-parse HEAD)" = "$F3_M1" ] || fail "F3: HEAD moved during STALE_MERGE"
379
+ echo "fixture F3 (stale record fails closed on re-integrate): all checks passed"
380
+
381
+ # ---------------- Fixture I: branch + origin — stale push-target and the R5-manual shape ----------------
382
+ # (blocker 35): the ERROR-after-MERGED retry path (push-target) pushed live
383
+ # HEAD with no check. I1: stale record → STALE_MERGE, the remote ref does
384
+ # not move. I2: branch restored to the merged tip → the legitimate retry
385
+ # still PUSHEDs (identity check is not a refusal policy). I3: the manual
386
+ # R5 shape (stale record + a newer manual merge of the CURRENT tip on the
387
+ # line) → PUSHEDs — the identity check must not strand a good line.
388
+ I_DIR="$T/fixture-i"
389
+ I_REMOTE="$T/remote-i.git"
390
+ git init -q --bare "$I_REMOTE"
391
+ mkdir -p "$I_DIR"
392
+ cd "$I_DIR"
393
+ git init -q -b main .
394
+ git config user.email test@test.t
395
+ git config user.name test
396
+ echo base > f.txt
397
+ git add -A
398
+ git commit -qm "base"
399
+ git remote add origin "$I_REMOTE"
400
+ git push -q origin main
401
+ git symbolic-ref refs/remotes/origin/HEAD refs/remotes/origin/main
402
+ export CREW_REPO="$I_DIR"
403
+ export CREW_HOME="$I_DIR/home"
404
+ export CREW_LIB="$REPO_DIR/lib"
405
+ tid="staleorigin1"
406
+ out=$(bash "$LIFECYCLE" prepare "$tid") || fail "I prepare failed: $out"
407
+ BRANCH=$(bash "$LIFECYCLE" resolve-branch "$tid") || fail "I resolve-branch failed"
408
+ I_BASE=$(git rev-parse "$BRANCH")
409
+ echo work >> "$I_DIR/.worktrees/$tid/f.txt"
410
+ (cd "$I_DIR/.worktrees/$tid" && git commit -qam "work w1")
411
+ I_W1=$(git rev-parse "$BRANCH")
412
+ git worktree remove --force "$I_DIR/.worktrees/$tid" || fail "I: worktree remove failed"
413
+ out=$(bash "$LIFECYCLE" integrate "$tid" "merge: $tid" 2>&1) || fail "I integrate failed: $out"
414
+ echo "$out" | grep -q '^MERGED:' || fail "I: expected MERGED, got: $out"
415
+ echo "$out" | grep -q '^PUSHED: origin/main' || fail "I: expected inline PUSHED, got: $out"
416
+ I_M1=$(git rev-parse HEAD)
417
+ [ "$(git --git-dir="$I_REMOTE" rev-parse refs/heads/main)" = "$I_M1" ] || fail "I: remote not at M1"
418
+
419
+ # I1: rework abandons w1 (branch back to base); push-target must refuse.
420
+ git branch -q -f "$BRANCH" "$I_BASE" || fail "I1: branch reset failed"
421
+ out=$(bash "$LIFECYCLE" push-target "$tid" 2>&1) && \
422
+ fail "I1: push-target with a stale record should fail, got: $out"
423
+ echo "$out" | grep -q '^STALE_MERGE' || fail "I1: expected STALE_MERGE, got: $out"
424
+ [ "$(git --git-dir="$I_REMOTE" rev-parse refs/heads/main)" = "$I_M1" ] || \
425
+ fail "I1: remote moved despite STALE_MERGE"
426
+ echo "fixture I1 (stale push-target fails closed): passed"
427
+
428
+ # I2: branch restored to the merged tip — the legitimate retry pushes.
429
+ git branch -q -f "$BRANCH" "$I_W1" || fail "I2: branch restore failed"
430
+ out=$(bash "$LIFECYCLE" push-target "$tid" 2>&1) || fail "I2 push-target failed: $out"
431
+ echo "$out" | grep -q '^PUSH_IDENTITY_OK' || fail "I2: expected PUSH_IDENTITY_OK, got: $out"
432
+ echo "$out" | grep -q '^PUSHED: origin/main' || fail "I2: expected PUSHED, got: $out"
433
+ echo "fixture I2 (legitimate retry still pushes): passed"
434
+
435
+ # I3: the manual R5 shape — stale record M1, but the line carries a newer
436
+ # manual merge of the CURRENT tip w2. push-target must PUSH (no regression).
437
+ echo work2 >> "$I_DIR/.worktrees/$tid/f.txt"
438
+ (cd "$I_DIR/.worktrees/$tid" && git commit -qam "work w2")
439
+ I_W2=$(git rev-parse "$BRANCH")
440
+ git merge -q --no-ff "$BRANCH" -m "manual r5 merge" || fail "I3: manual merge failed"
441
+ I_R=$(git rev-parse HEAD)
442
+ [ "$(git rev-list --parents -n 1 "$I_R" | awk '{print $3}')" = "$I_W2" ] || \
443
+ fail "I3: manual merge ^2 is not the branch tip"
444
+ out=$(bash "$LIFECYCLE" push-target "$tid" 2>&1) || fail "I3 push-target failed: $out"
445
+ echo "$out" | grep -q '^PUSHED: origin/main' || fail "I3: expected PUSHED for the R5-manual shape, got: $out"
446
+ [ "$(git --git-dir="$I_REMOTE" rev-parse refs/heads/main)" = "$I_R" ] || \
447
+ fail "I3: remote not at the manual merge"
448
+ echo "fixture I3 (R5-manual shape still pushes): passed"
449
+
450
+ echo "fixture I (stale push-target identity): all checks passed"
451
+
330
452
  # ---------------- Fixture G: push-target after MERGED_EMPTY ----------------
331
453
  # (2026-09-19 REVIEW #8: MERGED_EMPTY takes no lock and writes no record —
332
454
  # push-target must skip loudly, not report a misleading "lock lost".)