shapeup-sdlc 3.4.0 → 3.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/AGENTS.md +16 -4
- package/README.md +7 -3
- package/SECURITY.md +1 -1
- package/hooks/sandbox-guard.mjs +69 -6
- package/kernel/compile.mjs +33 -12
- package/kernel/harness.mjs +10 -4
- package/kernel/init/run-args.mjs +206 -0
- package/kernel/init/run.mjs +10 -0
- package/kernel/lib/contract.mjs +68 -1
- package/kernel/lib/paths.mjs +10 -0
- package/kernel/probe/concurrency.mjs +31 -6
- package/kernel/probe/digest.mjs +15 -1
- package/kernel/probe/owner.mjs +4 -1
- package/kernel/probe/requirements.mjs +296 -0
- package/kernel/probe/resume.mjs +195 -6
- package/kernel/probe/rounds.mjs +104 -0
- package/kernel/reduce/graph.mjs +5 -2
- package/kernel/reduce/ingest.mjs +69 -15
- package/kernel/reduce/ship.mjs +52 -31
- package/kernel/reduce/snapshot.mjs +23 -2
- package/kernel/report/export.mjs +54 -2
- package/kernel/report/facts.mjs +24 -2
- package/{skills/tech-lead → kernel}/schemas/domain.schema.json +20 -12
- package/kernel/verify/envelope.mjs +2 -2
- package/kernel/verify/skills.mjs +1 -1
- package/kernel/verify/spec.mjs +130 -4
- package/kernel/verify/trace.mjs +16 -7
- package/package.json +1 -1
- package/skills/ba-pitch-analyzer/SKILL.md +16 -1
- package/skills/coach/SKILL.md +8 -2
- package/skills/hill-chart/SKILL.md +3 -4
- package/skills/scope-architect/SKILL.md +16 -1
- package/skills/scope-hammer/SKILL.md +11 -2
- package/skills/spec-evaluator/SKILL.md +12 -1
- package/skills/tech-lead/SKILL.md +10 -10
- package/skills/tech-lead/references/gates.md +70 -12
- package/skills/tech-lead/references/protocol.md +4 -2
- package/skills/tech-lead/workflows/shapeup-run.js +176 -38
- package/skills/translator/SKILL.md +1 -1
- /package/{skills/tech-lead → kernel}/schemas/gate-answers.schema.json +0 -0
- /package/{skills/tech-lead → kernel}/schemas/work-order.schema.json +0 -0
- /package/{skills/tech-lead → kernel}/schemas/work-result.schema.json +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "shapeup-sdlc-plugin",
|
|
3
3
|
"displayName": "ShapeUp SDLC Plugin",
|
|
4
|
-
"version": "3.
|
|
4
|
+
"version": "3.6.0",
|
|
5
5
|
"description": "Shape Up SDLC harness for Claude Code: shaping, intake, orient, scope-mapping, building (T0-verified, sandboxed, scope-contracted), evaluation and QA skills orchestrated by a tech-lead.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "Liberty Nguyen",
|
package/AGENTS.md
CHANGED
|
@@ -28,9 +28,10 @@ Betting Table: PO decides; rejected pitches loop back to raw idea.
|
|
|
28
28
|
|------|------|--------|
|
|
29
29
|
| Kick-off | ⏸ **L0** — Intake & Config (L0.8 model/budget matrix) + worker roster ✧ | `/translator` if non-English |
|
|
30
30
|
| Orient (Scout) | ⏸ **L1a** — Orient Review | `/orient` |
|
|
31
|
-
|
|
|
31
|
+
| Requirements | — (reviewed at L1b) | `/ba-pitch-analyzer` (`coverage`): the pitch's clauses → committed `requirements.md`, one atomic clause per `REQ-<n>` row naming the pitch clause it came from, ids assigned once and frozen; dispatched once, ahead of Analyze because the acceptance criteria are what cite its ids. A registry already on disk is not re-dispatched |
|
|
32
|
+
| Analyze | — (reviewed at L1b) | `/ba-pitch-analyzer` (`analyze`): spec tree + board (UC + Invariants + Test Surface ★); before Wire (needs its use cases). An acceptance criterion that grades a requirement carries `(covers: REQ-…)` — that clause is the edge the verdict travels back along |
|
|
32
33
|
| Wire | ⏸ **L1a.5** — Wiring Review ✚ | `/solution-architect` (`wire`): sole writer of committed `wiring-map.md` — per-UC engine → seam → entry-point call site → affordance, per `project-profile.md` |
|
|
33
|
-
| Map Scopes | ⏸ **L1b** — Board Review (+ substrate disjointness lint) | `/scope-architect` (scope contracts ✦ — sole writer); traceability oracle advisory
|
|
34
|
+
| Map Scopes | ⏸ **L1b** — Board Review (+ substrate disjointness lint) | `/scope-architect` (scope contracts ✦ — sole writer); traceability oracle advisory ✚. A registered requirement that no acceptance criterion grades and no scope claims is **red** here, and L1b prints the `REQ → AC` table: the two ways out are an AC carrying `(covers: REQ-…)` or the PO marking the clause `CUT (PO-approved)`. Red only where the plan is still cheap to change — after L1b nobody re-reads the pitch |
|
|
34
35
|
| Build Vertically | ⏸ **L2** — Board 100% ✅ + T0-green ✦ | per dispatch: compile order → `/task-executor` (--order) → ingest result; T0-verified per attempt (fixtures + DB probe + seesaw ✦), substrate-sandboxed ✦. Scopes build **concurrently** ✦ — `--parallel-scopes N` caps it (default 4), a scope is released the moment its own dependencies are green, and a scope green in this round is skipped rather than rebuilt. Then the **round build gate** ⚙: the ledger's run command, then the profile's `build_probe` and `launch_probe`, run once per round before EVAL — a red gate ends the round with no verdict and its failing step is compiled into the next round's orders as bugs; a `mobile` profile with no `launch_probe` is warned about every round, so the install/launch risk has an owner |
|
|
35
36
|
| EVAL (once per round) | ⏸ **L3** — Verdict | `/spec-evaluator` (--order), only over a round whose build gate ⚙ is not red: spec- + test-surface-conformance ★, T0 citation ✦; refuted boxes/verdict applied by ingest |
|
|
36
37
|
| FAIL → round r+1 | — | regression rule ★: bugs + full Test Surface of touched UC |
|
|
@@ -58,16 +59,27 @@ Everything discovered funnels into `.shapeup/<slug>/discovery/ledger.md` (Orient
|
|
|
58
59
|
- **Ledger = single source of truth** — every discovery flow writes only its own section.
|
|
59
60
|
- **QA is a level-up, not a gate** — `--no-qa` skips it; circuit breaker outranks the Hunter.
|
|
60
61
|
- **Role separation** — Evaluator grades, task-executor fixes, QA discovers.
|
|
62
|
+
- **The requirements matrix is a projection, never a verdict** — `REQ → AC → criterion → verdict` is derived from files for one named run (the registry, the board's `covers:` clauses, the run's verdict rows, the T0 citations), never narrated and never passed in. `covers:` is the authoritative join: a criterion anchored to a requirement no AC covers is printed for reconciliation and counted as nothing. L4 reads one line off it, GATE H's census takes the clauses with no evidence, `REPORT.md` freezes the table — and none of that blocks a ship. A clause with no PASS evidence is a fact the baseline comparison weighs, not a veto.
|
|
61
63
|
- **Hill phase is mechanical ✦** — derived only from T0/T1/seesaw artifacts, never self-reported, and a T0-green from a round whose build gate ⚙ is red moves no dot (a green fixture in a round the feature did not build is evidence about the fixture); the evaluator cites a T0 artifact it re-hashes itself, from the list its order carries. A scoped verdict citing none is refused: its round stays open and is evaluated again, never advanced.
|
|
62
64
|
- **Envelope port (v1.0)** — every dispatch is WorkOrder in / WorkResult out; shared state has exactly one writer (the ingest step); malformed envelopes are hook-denied. Workers: stateless, craft-only, pipeline-blind.
|
|
63
65
|
|
|
64
66
|
## Setup & Execution
|
|
65
67
|
|
|
66
|
-
- Orders/results live in `.shapeup/<slug>/orders|results/`; the envelope schemas ship
|
|
68
|
+
- Orders/results live in `.shapeup/<slug>/orders|results/`; the envelope schemas ship with the plugin runtime, not with any individual skill, so every worker validates against the same copy.
|
|
67
69
|
- The plugin's run entry points need a one-time permission grant — `npx shapeup-sdlc init` writes it into `.claude/settings.json` (`permissions.allow`); without it a headless run stalls at step one. That grant is necessary, not sufficient: it covers the run's own deterministic entry points, not the generic file edits every worker skill makes constantly, or any command a worker reaches for beyond the grant's own exact shape. A truly unattended run also needs a Claude Code permission mode that covers those (`acceptEdits` at minimum) — the plugin cannot grant that on your behalf.
|
|
70
|
+
- The grant is necessary but sits under two more layers this plugin cannot reach either. A fresh
|
|
71
|
+
checkout is an **untrusted workspace**, and Claude Code discards the whole permission grant — every
|
|
72
|
+
rule in it, not only this one — until the workspace is trusted; the installer detects that state and
|
|
73
|
+
tells you, because trusting a directory to run code from is your decision to make, never a package's
|
|
74
|
+
to make for you. Above workspace trust sits Claude Code's own **auto-mode classifier**, which can
|
|
75
|
+
still block a call the grant already covers, invisibly to anything this plugin ships — no rule or
|
|
76
|
+
hook here can see it, let alone override it. When either layer stops an automated run, the
|
|
77
|
+
documented fallback is to drive Build by hand: the same per-dispatch cycle named in the Build
|
|
78
|
+
Vertically step above — compile, dispatch, ingest, verify — run one call at a time instead of
|
|
79
|
+
through the chained launch, until the run can resume unattended again.
|
|
68
80
|
- Two storage tiers (ADR-0001): COMMITTED `shapeup/<slug>/` (shaping, spec, scopes, wiring-map, project-profile, requirements, hill, `REPORT.md` frozen at L4) vs GITIGNORED `.shapeup/` (board, orders/results, T0/eval/QA artifacts, ledgers, metrics, gate answers, and the run scripts staged for launch).
|
|
69
81
|
- **The run launches from a copy inside your project, and it has to.** The Workflow tool loads a script only from a directory the session may already read; the plugin installs outside your project, so naming the shipped path is refused before the run begins and no permission rule repairs it — the grant authorises the tool, not what it may read. Opening a run therefore re-copies the run scripts to `.shapeup/workflows/` and reports the path the launch names. A run in flight keeps the copy it started with: an upgrade reaches the next run, not the current round.
|
|
70
|
-
- **A write the substrate does not cover is denied for as long as the dispatch is in flight, and no longer.** A dispatch is live from the moment its order is compiled until a result for *that* dispatch lands, and liveness is read off the run's order set — not off the pointer that names the run, which outlives it.
|
|
82
|
+
- **A write the substrate does not cover is denied for as long as the dispatch is in flight, and no longer.** A dispatch is live from the moment its order is compiled until a result for *that* dispatch lands, and liveness is read off the run's order set — not off the pointer that names the run, which outlives it. A run closed any other way than shipping — escalated, aborted, killed outright — can leave one still open, and while the run's pointer is still on disk the fence holds on that order exactly as if the run were live. The pointer is what actually switches it off, though: the fence is enforced only while that pointer exists on disk, so a close that removes it releases the fence whatever the order set still says — restoring the pointer flips it straight back to denying. A **ship** close does not itself answer what it leaves outstanding — a ship close can retire the pointer over an order that never got a result — so it is not special because nothing is left unanswered; it is special only because retiring the pointer is the one lever every close needs pulled, and `reduce ship` pulls it for you. Whichever way a run closes, an order it leaves unanswered stays genuinely unresolved — not merely un-fenced — until `init run --force` runs: it writes a synthetic result for every order the closed run left unanswered, so the fence lifts without waiting on a worker that is never coming back. What the fence actually gates is narrower than "the project," too — it is this assistant's own edit path (a direct file write or edit call) outside the live order's substrate; a shell command, `git`, or any other editor still writes straight through it. Re-dispatch is fenced again, and the committed tier is never a worker's to write either way: those files belong to the orchestrator, whose window is a phase boundary rather than the middle of somebody else's order.
|
|
71
83
|
- Every run has a `run_id` — the receipt mints it, and orders, T0 artifacts, trial rows, agent-call journal rows and hook decisions all carry it. It is the only key that separates two runs of the same feature: everything else (`order_id`, round/attempt) repeats. It is **not** a time boundary — a relaunch resumes the same run and reuses the key, so one `run_id` legitimately spans every launch after a paused gate or a kill, with hours of wall clock between them, and `orders/<id>.json` is rewritten by each. Anything measuring elapsed time reads the append-only records, never the span of a key. SHIP S.7 exports the run's records as fact tables under `.shapeup/exports/<run_id>/` before the run trace is superseded; a WorkResult carries no `run_id` and reaches it through `order_id`.
|
|
72
84
|
- Every run projects a **run graph** — `.shapeup/<slug>/graph.jsonl`, append-only, written only by
|
|
73
85
|
`reduce graph`. Two families kept separate: work lineage (Run, Order, Result, Verdict, Trial,
|
package/README.md
CHANGED
|
@@ -221,8 +221,12 @@ the layer that carries it, and the three layers here fail differently:
|
|
|
221
221
|
- `PreToolUse` (`Edit|Write|MultiEdit`) — **`hooks/sandbox-guard.mjs` blocks a write that no LIVE
|
|
222
222
|
order's substrate permits.** It reads every order that is compiled and not yet answered — a result
|
|
223
223
|
at least as new as the order itself — rather than a pointer to one, so scopes building
|
|
224
|
-
concurrently are each held to their own contract and a finished run fences nothing
|
|
225
|
-
|
|
224
|
+
concurrently are each held to their own contract and a finished run fences nothing. A phase
|
|
225
|
+
dispatch the run has already moved past stops fencing too: once a later phase compiles its order,
|
|
226
|
+
an evaluation or a QA leg that never returned a result no longer holds the board. `frozen` is
|
|
227
|
+
checked first and outranks everything, across every live contract — including the carve-out that
|
|
228
|
+
otherwise keeps the active feature's own run trace writable, so a path a live order froze stays
|
|
229
|
+
frozen wherever it lives.
|
|
226
230
|
- `PreToolUse` (`Bash|Read|Write|Edit|MultiEdit`) — **`hooks/safety-spine.mjs` denies destructive
|
|
227
231
|
commands** (`rm -rf` on unrecoverable targets, force-push/push-to-main, `git reset --hard`,
|
|
228
232
|
`DROP TABLE`) and secret-file reads. A machine guard, not a pipeline guard; the escape hatch is
|
|
@@ -328,11 +332,11 @@ claude --plugin-dir . # load this working copy without installing
|
|
|
328
332
|
plugin.json # plugin manifest
|
|
329
333
|
marketplace.json # marketplace listing (points at this repo)
|
|
330
334
|
skills/<name>/SKILL.md # the 13 harness skills (+ references/ and assets/)
|
|
331
|
-
skills/tech-lead/schemas/ # the envelope port: WorkOrder, WorkResult, domain registry
|
|
332
335
|
skills/tech-lead/workflows/shapeup-run.js # the BUILD-phase pipeline, on the native Workflow runtime
|
|
333
336
|
kernel/harness.mjs # ONE entry point for every deterministic step; the whole permission grant
|
|
334
337
|
kernel/{verify,reduce,probe,init,report}/ # its subcommands, plus compile and gate at the root
|
|
335
338
|
kernel/lib/ # argv (the typed CLI boundary), paths (+ the run key), contract (shape)
|
|
339
|
+
kernel/schemas/ # the envelope port: WorkOrder, WorkResult, domain registry
|
|
336
340
|
commands/*.md # slash commands (/ship + the 9 phase commands)
|
|
337
341
|
hooks/ # hooks.json + the four walls: safety-spine, gate-intake, sandbox-guard
|
|
338
342
|
# (PreToolUse) + gate-zerowork (Stop, the one blocking hook)
|
package/SECURITY.md
CHANGED
|
@@ -72,7 +72,7 @@ sitting, and reading them is the recommended review.
|
|
|
72
72
|
| [`safety-spine.mjs`](hooks/safety-spine.mjs) | PreToolUse (`Bash\|Read\|Write\|Edit\|MultiEdit`) | The proposed command/path; `.shapeup/safety-overrides.json` | Yes — provably destructive ops only: `rm -rf` on unrecoverable targets, `git push --force` / push to main, `git reset --hard`, `git clean -fdx`, `DROP TABLE`/`TRUNCATE`, reads of `.env`/keys/cloud credentials, and any write to its own overrides file | Never blocks an unmatched command; `--force-with-lease` stays allowed |
|
|
73
73
|
| [`gate-intake.mjs`](hooks/gate-intake.mjs) | PreToolUse (`Skill`) | The `tech-lead` dispatch's own arguments | Yes — an orchestrator dispatch carrying no resolvable intake (no pitch, spec, resume or requirement text) | Fails open on `--order` and on any ambiguous arg shape |
|
|
74
74
|
| [`harness verify envelope`](kernel/verify/envelope.mjs) | PreToolUse (`Skill\|Agent`) | The `--order` file named in the dispatch; the JSON schemas | Yes — a worker dispatch whose order file is missing or schema-invalid | Never gates a dispatch that carries no `--order` (standalone skill use stays free) |
|
|
75
|
-
| [`sandbox-guard.mjs`](hooks/sandbox-guard.mjs) | PreToolUse (`Edit\|Write\|MultiEdit`) | The target path; the `substrate` block of every LIVE order — compiled, with no result at least as new as the order's own `compiled_at
|
|
75
|
+
| [`sandbox-guard.mjs`](hooks/sandbox-guard.mjs) | PreToolUse (`Edit\|Write\|MultiEdit`) | The target path; the `substrate` block of every LIVE order — compiled, with no result at least as new as the order's own `compiled_at`, and, for a run-level phase dispatch, not yet superseded by a later phase's order | Yes — any write no live order permits: inside any `frozen`, outside every `allowed`/`shared`, or a `Write` to an `append_only` path. `frozen` is checked FIRST, so it also outranks the run-trace carve-out | No-op unless an order is live, which a finished run no longer is: the pointer names the run, never a dispatch. The active feature's own `.shapeup/<slug>/` run trace is writable EXCEPT where a live order freezes a path inside it. Appends denials to the local pathology log |
|
|
76
76
|
| [`dispatch-receipt.mjs`](hooks/dispatch-receipt.mjs) | PostToolUse (`Skill\|Agent`) | The `--order` file named in the dispatch; the tool result's own report of which skill ran | **No — it has no deny path at all.** It records that the shipped skill ran, so `harness reduce ingest` can refuse a result no dispatch produced | Never writes an attestation for a result that does not name a resolved skill; never fails the call it observes (every write is inside `try`/`catch`) |
|
|
77
77
|
| [`gate-zerowork.mjs`](hooks/gate-zerowork.mjs) | Stop | Run receipts on disk; the session transcript; the decision ledger | **Yes — the one blocking hook.** Returns `decision:"block"` when the session dispatched the orchestrator and produced no run receipt | Defers the moment any receipt exists; `stop_hook_active` caps it at one block per stop chain |
|
|
78
78
|
|
package/hooks/sandbox-guard.mjs
CHANGED
|
@@ -42,6 +42,21 @@
|
|
|
42
42
|
// the order's own `compiled_at`, the stamp the compiler writes INTO the order, which a copy or a
|
|
43
43
|
// touch cannot perturb. An order carrying no stamp falls back to presence, which is all it ever had.
|
|
44
44
|
//
|
|
45
|
+
// A RUN-LEVEL ORDER RETIRES AT ITS PHASE BOUNDARY, which is the other half of that same question.
|
|
46
|
+
// Liveness is "compiled, no result yet", so a PHASE dispatch whose worker never returned — a killed
|
|
47
|
+
// evaluation, a QA leg that escalated without a result, a failed scope mapping — would stay live for
|
|
48
|
+
// the rest of the run, and everything its `frozen` list names (the board, the spec tree) would be
|
|
49
|
+
// fenced from then on. The board is the one that bites: the next round's doer cannot tick its own
|
|
50
|
+
// acceptance criteria, and the run wedges with no dispatch actually in flight. The orchestrator has
|
|
51
|
+
// long since moved on by the time that matters, and the move itself is the signal: the next phase
|
|
52
|
+
// compiles its own order. So an order for a RUN-LEVEL operation stops being live once an order for a
|
|
53
|
+
// DIFFERENT operation has been compiled after it — the window the committed tier already has, a
|
|
54
|
+
// phase boundary rather than the middle of somebody else's dispatch. Build legs are exempt: they run
|
|
55
|
+
// concurrently, finish out of order, and an abandoned one is resolved by ``harness init run --force``
|
|
56
|
+
// rather than by a sibling's compile stamp. Two concurrent legs of the SAME operation never retire
|
|
57
|
+
// each other, for that same reason. An order carrying no operation or no `compiled_at` keeps
|
|
58
|
+
// fencing — an unreadable claim is not a retired one.
|
|
59
|
+
//
|
|
45
60
|
// That is the same question as "the writer's own contract" because scope substrates are disjoint by
|
|
46
61
|
// construction — `harness verify spec`'s DISJOINT rule fails a spec where two scopes claim the same
|
|
47
62
|
// path, and it runs at GATE L1b before any build starts. `frozen` is checked across all of them, so
|
|
@@ -133,7 +148,37 @@ function answered(resultPath, order) {
|
|
|
133
148
|
}
|
|
134
149
|
|
|
135
150
|
/**
|
|
136
|
-
*
|
|
151
|
+
* Operations whose order is a BUILD leg — one scope, one attempt, dispatched alongside its siblings.
|
|
152
|
+
*
|
|
153
|
+
* Everything else the compiler emits is a RUN-LEVEL phase dispatch, and only those retire at a phase
|
|
154
|
+
* boundary (see the banner). The distinction is by operation rather than by "does it carry a scope",
|
|
155
|
+
* because the single-task lane compiles an `execute` order with no scope contract on it and that leg
|
|
156
|
+
* is still a build.
|
|
157
|
+
*/
|
|
158
|
+
const BUILD_OPERATIONS = new Set(["execute", "fix", "spike"]);
|
|
159
|
+
|
|
160
|
+
/**
|
|
161
|
+
* Has the run moved past this order's phase — i.e. did a LATER order for a different operation get
|
|
162
|
+
* compiled while this one was still unanswered?
|
|
163
|
+
*
|
|
164
|
+
* Conservative in both directions it cannot read: an order with no `operation`, a build leg, or an
|
|
165
|
+
* order with no parseable `compiled_at` keeps fencing.
|
|
166
|
+
*
|
|
167
|
+
* @param {object} order - The parsed, unanswered order.
|
|
168
|
+
* @param {Array<{operation:(string|undefined), at:number}>} stamps - Every compiled order's
|
|
169
|
+
* operation and parsed `compiled_at`, unparseable stamps excluded.
|
|
170
|
+
* @returns {boolean} True when this order's phase is over and it should stop being enforced.
|
|
171
|
+
*/
|
|
172
|
+
function pastItsPhase(order, stamps) {
|
|
173
|
+
const op = order?.operation;
|
|
174
|
+
if (!op || BUILD_OPERATIONS.has(op)) return false;
|
|
175
|
+
const at = Date.parse(order?.compiled_at ?? "");
|
|
176
|
+
if (Number.isNaN(at)) return false;
|
|
177
|
+
return stamps.some((s) => s.at > at && s.operation !== op);
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/**
|
|
181
|
+
* Every order for this run that has been compiled, not yet answered, and whose phase is still open.
|
|
137
182
|
*
|
|
138
183
|
* @param {string} cwd - Project root.
|
|
139
184
|
* @param {string} slug - The run named by the pointer.
|
|
@@ -143,14 +188,17 @@ function liveOrders(cwd, slug) {
|
|
|
143
188
|
const dir = ordersDir(cwd, slug);
|
|
144
189
|
if (!existsSync(dir)) return [];
|
|
145
190
|
const rDir = resultsDir(cwd, slug);
|
|
146
|
-
const
|
|
191
|
+
const unanswered = [];
|
|
192
|
+
const stamps = [];
|
|
147
193
|
for (const f of readdirSync(dir)) {
|
|
148
194
|
if (!f.endsWith(".json")) continue;
|
|
149
195
|
const order = readJSON(join(dir, f));
|
|
150
196
|
if (!order) continue;
|
|
151
|
-
|
|
197
|
+
const at = Date.parse(order.compiled_at ?? "");
|
|
198
|
+
if (!Number.isNaN(at)) stamps.push({ operation: order.operation, at });
|
|
199
|
+
if (!answered(join(rDir, f), order)) unanswered.push(order);
|
|
152
200
|
}
|
|
153
|
-
return
|
|
201
|
+
return unanswered.filter((o) => !pastItsPhase(o, stamps));
|
|
154
202
|
}
|
|
155
203
|
|
|
156
204
|
function extractPaths(toolInput) {
|
|
@@ -231,22 +279,33 @@ async function main() {
|
|
|
231
279
|
const runTracePrefix = join(LOCAL, active.slug) + sep;
|
|
232
280
|
const violations = [];
|
|
233
281
|
const blockReasons = [];
|
|
282
|
+
let frozenHits = 0;
|
|
234
283
|
|
|
235
284
|
for (const raw of targetPaths) {
|
|
236
285
|
const abs = resolve(cwd, raw);
|
|
237
286
|
const rel = relative(root, abs);
|
|
238
|
-
if (rel.startsWith(runTracePrefix)) continue;
|
|
239
287
|
|
|
240
288
|
// Frozen takes absolute precedence, and it is checked across EVERY live contract: a path one
|
|
241
289
|
// scope froze stays frozen while another scope is in flight, which is the whole point of
|
|
242
290
|
// declaring it.
|
|
291
|
+
//
|
|
292
|
+
// IT IS CHECKED BEFORE THE RUN-TRACE CARVE-OUT, and that order is load-bearing. The carve-out
|
|
293
|
+
// below exists so the doer can keep its own bookkeeping current; it was never a licence to
|
|
294
|
+
// overwrite a file a live contract declared read-only. Checked after it, every `frozen` glob
|
|
295
|
+
// naming a path inside the run trace was inert — the board an evaluation froze, the staged pitch
|
|
296
|
+
// a planner is graded against — so the compiler emitted a declaration with no enforcer, which is
|
|
297
|
+
// the exact state this hook exists to end. A path a live contract freezes is a violation
|
|
298
|
+
// wherever it lives.
|
|
243
299
|
const freezer = contracts.find((c) => matchesAny(rel, c.frozen));
|
|
244
300
|
if (freezer) {
|
|
245
301
|
violations.push(rel);
|
|
302
|
+
frozenHits++;
|
|
246
303
|
blockReasons.push(`${rel} is frozen by ${freezer.order_id}`);
|
|
247
304
|
continue;
|
|
248
305
|
}
|
|
249
306
|
|
|
307
|
+
if (rel.startsWith(runTracePrefix)) continue;
|
|
308
|
+
|
|
250
309
|
if (contracts.some((c) => matchesAny(rel, c.allowed))) continue; // inside a live contract
|
|
251
310
|
|
|
252
311
|
if (contracts.some((c) => matchesAny(rel, c.appendOnly))) {
|
|
@@ -291,7 +350,11 @@ async function main() {
|
|
|
291
350
|
|
|
292
351
|
return {
|
|
293
352
|
verdict: "deny", event: "PreToolUse", tool: p.tool_name, subject: active.order_path, cwd: root,
|
|
294
|
-
|
|
353
|
+
// THE LEDGER NAMES THE CAUSE, not just the verdict. A write refused because a live contract
|
|
354
|
+
// FROZE the path and a write refused because no contract covers it are different facts with
|
|
355
|
+
// different remedies, and a single rule string cannot tell the reader which one happened —
|
|
356
|
+
// which is how a frozen declaration can stop being enforced without a single row moving.
|
|
357
|
+
rule: frozenHits === violations.length ? "frozen" : "outside-substrate",
|
|
295
358
|
reason: `${violations.length} write(s) rejected by substrate boundaries: ${blockReasons.join("; ")}`,
|
|
296
359
|
payload: {
|
|
297
360
|
hookSpecificOutput: {
|
package/kernel/compile.mjs
CHANGED
|
@@ -58,7 +58,7 @@ export const COACHABLE = new Set([
|
|
|
58
58
|
]);
|
|
59
59
|
|
|
60
60
|
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
61
|
-
const ORDER_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "
|
|
61
|
+
const ORDER_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "./schemas/work-order.schema.json"), "utf8"));
|
|
62
62
|
|
|
63
63
|
// --- tiny frontmatter reader (scalar keys + [a, b] inline lists) --------------------------
|
|
64
64
|
/**
|
|
@@ -207,14 +207,29 @@ export function substrateFor(operation, { slug, specDir, scope } = {}) {
|
|
|
207
207
|
// `feedback.md`, `api-feasibility.md` and `integration.md` are analysis, not contract.
|
|
208
208
|
const working = `${local}/working`;
|
|
209
209
|
const FROZEN_SPEC_CORE = [`${spec}/domain-model.md`, `${spec}/usecases/*.md#Steps`, `${spec}/contracts/**`, `${spec}/ux-behavior.md`];
|
|
210
|
+
// THE STAGED PITCH IS THE RUN'S OWN INPUT TRUTH, and it is a separate constant from the spec core
|
|
211
|
+
// on purpose: the spec core is what a planner PRODUCES and a judge grades against, while these two
|
|
212
|
+
// are what the run was asked for. `init run` stages them beside the receipt that digests them, and
|
|
213
|
+
// every planning and evaluating operation reads them — so a worker that can rewrite either can
|
|
214
|
+
// rewrite the question it is about to be measured on, and the receipt's digest stops describing
|
|
215
|
+
// the file next to it. Frozen, not merely unlisted: `analyze` already carries the whole run trace
|
|
216
|
+
// in its `allowed` globs, so only a `frozen` entry denies the write. `translate` is the one
|
|
217
|
+
// operation that legitimately rewrites a pitch, and it writes the COMMITTED copy, not this one.
|
|
218
|
+
const FROZEN_INTAKE = [`${local}/intake.md`, `${local}/breadboard.md`];
|
|
210
219
|
switch (operation) {
|
|
211
220
|
case "execute": case "fix": case "spike":
|
|
221
|
+
// Build legs are the widest window on FROZEN_INTAKE, not an exemption from it: they are the
|
|
222
|
+
// most numerous and longest-lived dispatches in a run, so a doer that can rewrite the staged
|
|
223
|
+
// pitch can rewrite the run's own input truth mid-build. `init run` stages these before any
|
|
224
|
+
// order is live (no live contract yet — nothing to violate) and `translate` writes the
|
|
225
|
+
// COMMITTED copy, not this one, so neither legitimate write is touched by this line.
|
|
212
226
|
return {
|
|
213
227
|
allowed: [...(scope?.allowed_file_substrate || []), `${local}/spikes/**`],
|
|
214
228
|
shared: scope?.shared_substrate || [],
|
|
229
|
+
frozen: [...FROZEN_INTAKE],
|
|
215
230
|
};
|
|
216
231
|
case "analyze":
|
|
217
|
-
return { allowed: [`${spec}/**`, `${local}/**`], frozen: [] };
|
|
232
|
+
return { allowed: [`${spec}/**`, `${local}/**`], frozen: [...FROZEN_INTAKE] };
|
|
218
233
|
|
|
219
234
|
case "reconcile":
|
|
220
235
|
return {
|
|
@@ -229,24 +244,24 @@ export function substrateFor(operation, { slug, specDir, scope } = {}) {
|
|
|
229
244
|
// The covers-closure input truth. Writes ONLY the derived registry: the REQ source it
|
|
230
245
|
// extracts from is frozen alongside the spec core, because a planner that may edit the
|
|
231
246
|
// requirements it is being measured against is not measuring anything.
|
|
232
|
-
return { allowed: [globShared(slug, "requirements.md")], frozen: FROZEN_SPEC_CORE };
|
|
247
|
+
return { allowed: [globShared(slug, "requirements.md")], frozen: [...FROZEN_SPEC_CORE, ...FROZEN_INTAKE] };
|
|
233
248
|
|
|
234
249
|
case "map-scopes":
|
|
235
250
|
return {
|
|
236
251
|
allowed: [`${scopesDir}/*.md`, globShared(slug, "scope-board.md")],
|
|
237
|
-
frozen: [...FROZEN_SPEC_CORE, `${local}/tasks/**`],
|
|
252
|
+
frozen: [...FROZEN_SPEC_CORE, ...FROZEN_INTAKE, `${local}/tasks/**`],
|
|
238
253
|
};
|
|
239
254
|
case "wire":
|
|
240
255
|
// solution-architect writes the SHARED wiring map DIRECTLY (precedent: scope-architect
|
|
241
256
|
// writes scopes/*.md). The spec core, the scopes, and the profile stay frozen.
|
|
242
257
|
return {
|
|
243
258
|
allowed: [globShared(slug, "wiring-map.md")],
|
|
244
|
-
frozen: [...FROZEN_SPEC_CORE, `${scopesDir}/**`, globShared(slug, "project-profile.md")],
|
|
259
|
+
frozen: [...FROZEN_SPEC_CORE, ...FROZEN_INTAKE, `${scopesDir}/**`, globShared(slug, "project-profile.md")],
|
|
245
260
|
};
|
|
246
261
|
case "evaluate":
|
|
247
|
-
return { allowed: [`${local}/evaluation/**`], frozen: [`${spec}/**`, `${local}/tasks/**`] };
|
|
262
|
+
return { allowed: [`${local}/evaluation/**`], frozen: [`${spec}/**`, ...FROZEN_INTAKE, `${local}/tasks/**`] };
|
|
248
263
|
case "hunt":
|
|
249
|
-
return { allowed: [`${local}/qa/**`], frozen: [`${spec}/**`, `${local}/tasks/**`] };
|
|
264
|
+
return { allowed: [`${local}/qa/**`], frozen: [`${spec}/**`, ...FROZEN_INTAKE, `${local}/tasks/**`] };
|
|
250
265
|
case "orient":
|
|
251
266
|
return { allowed: [`${local}/orient/**`], frozen: [`${spec}/**`] };
|
|
252
267
|
case "translate":
|
|
@@ -509,24 +524,30 @@ export function bugLocations(bug) {
|
|
|
509
524
|
*
|
|
510
525
|
* OWNERSHIP IS BY SUBSTRATE, because that is what the sandbox enforces: a scope is exactly the set
|
|
511
526
|
* of files its worker may write, so a scope whose substrate excludes the cited line cannot fix it
|
|
512
|
-
* however well it understands the bug.
|
|
527
|
+
* however well it understands the bug. The substrate a scope may write is `allowed ∪ shared` — the
|
|
528
|
+
* same union `sandbox-guard` composes at the fence — so a path declared ONLY in a contract's
|
|
529
|
+
* `shared` list is a file that scope may legitimately write, and the election must see it too —
|
|
530
|
+
* filtering on `allowed` alone elects no one for a shared-only path, and `bugsForScope` then reads
|
|
531
|
+
* that null as "no scope owns this" and fans the bug out to every scope instead of the one or two
|
|
532
|
+
* that declared it.
|
|
513
533
|
*
|
|
514
534
|
* BUT A MATCH IS NOT AN ELECTION. An entry point is routinely SHARED — on the measured run
|
|
515
535
|
* `bin/todo.js` sits in five scopes' substrate at once — so "address it to every scope that
|
|
516
536
|
* matches" hands the same one-line fix to five workers building concurrently against one file.
|
|
517
537
|
* That is a write race the harness sets up itself, and four of the five fixes are waste even when
|
|
518
538
|
* it resolves. So: prefer a scope that owns the file EXCLUSIVELY (allowed, not shared), and among
|
|
519
|
-
* equals
|
|
520
|
-
*
|
|
539
|
+
* equals — every remaining candidate declares it shared, exclusive or not — take the lowest scope
|
|
540
|
+
* id: a rule that needs no coordination to agree with itself, since each leg compiles its own
|
|
541
|
+
* order in its own process.
|
|
521
542
|
*
|
|
522
543
|
* @param {string} path - Repo-relative file the bug cites.
|
|
523
544
|
* @param {Array<{scope_id:string, allowed:string[], shared:string[]}>} scopes - Every scope.
|
|
524
545
|
* @returns {string|null} The elected scope id, or null when no scope may write that file.
|
|
525
546
|
*/
|
|
526
547
|
export function electOwner(path, scopes) {
|
|
527
|
-
const can = (scopes || []).filter((s) => matchesAny(path, s.allowed));
|
|
548
|
+
const can = (scopes || []).filter((s) => matchesAny(path, s.allowed) || matchesAny(path, s.shared || []));
|
|
528
549
|
if (!can.length) return null;
|
|
529
|
-
const exclusive = can.filter((s) => !matchesAny(path, s.shared || []));
|
|
550
|
+
const exclusive = can.filter((s) => matchesAny(path, s.allowed) && !matchesAny(path, s.shared || []));
|
|
530
551
|
return (exclusive.length ? exclusive : can).map((s) => s.scope_id).sort()[0];
|
|
531
552
|
}
|
|
532
553
|
|
package/kernel/harness.mjs
CHANGED
|
@@ -31,7 +31,8 @@
|
|
|
31
31
|
// ship · board · verdict · graph
|
|
32
32
|
// gate An answer file with a source, not a vibe.
|
|
33
33
|
// probe resume · t0 · stats · digest · Read-only queries over run state. `concurrency`
|
|
34
|
-
// concurrency · leg · eval ·
|
|
34
|
+
// concurrency · leg · eval · answers how many legs ran at once and what the
|
|
35
|
+
// owner · requirements
|
|
35
36
|
// fan-out bought, and refuses a figure the record set
|
|
36
37
|
// cannot support rather than printing a plausible one.
|
|
37
38
|
// `leg` answers whether a scope's work reached the
|
|
@@ -45,8 +46,12 @@
|
|
|
45
46
|
// own end-of-turn summary of its own verdict. `owner`
|
|
46
47
|
// answers which scope may write a path, elected from
|
|
47
48
|
// the contracts — so a census cites it instead of
|
|
48
|
-
// asserting ownership from memory.
|
|
49
|
-
//
|
|
49
|
+
// asserting ownership from memory. `requirements`
|
|
50
|
+
// answers which pitch clause a verdict reached, joined
|
|
51
|
+
// through the plan's own covers: edge — the L4 line and
|
|
52
|
+
// GATE H's census cite it for the same reason.
|
|
53
|
+
// init run · fit · run-args Opens a run, or refuses it (exit 3). `run-args`
|
|
54
|
+
// writes GATE L0.9b's launch record and echoes it.
|
|
50
55
|
// report export Projects the run's records as fact tables.
|
|
51
56
|
// compile The WorkOrder: schema-valid or nothing is dispatched.
|
|
52
57
|
//
|
|
@@ -81,8 +86,9 @@ export const ROUTES = {
|
|
|
81
86
|
resume: "./probe/resume.mjs", t0: "./probe/t0.mjs", stats: "./probe/stats.mjs",
|
|
82
87
|
digest: "./probe/digest.mjs", concurrency: "./probe/concurrency.mjs",
|
|
83
88
|
leg: "./probe/leg.mjs", eval: "./probe/eval.mjs", owner: "./probe/owner.mjs",
|
|
89
|
+
requirements: "./probe/requirements.mjs",
|
|
84
90
|
},
|
|
85
|
-
init: { run: "./init/run.mjs", fit: "./init/fit.mjs" },
|
|
91
|
+
init: { run: "./init/run.mjs", fit: "./init/fit.mjs", "run-args": "./init/run-args.mjs" },
|
|
86
92
|
report: { export: "./report/export.mjs", _default: "export" },
|
|
87
93
|
gate: "./gate.mjs",
|
|
88
94
|
compile: "./compile.mjs",
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// GATE L0.9b — THE LAUNCH RECORD, GIVEN A WRITER.
|
|
3
|
+
//
|
|
4
|
+
// WHY THIS EXISTS. `domain.schema.json`'s `RunArgs` entry calls `.shapeup/<slug>/run-args.json`
|
|
5
|
+
// "the only artifact that records what a run was configured with" and names tech-lead as the
|
|
6
|
+
// writer — but "tech-lead" meant a paragraph of prose telling the orchestrating session to
|
|
7
|
+
// assemble a JSON object by hand and `Write` it. Outside a structural-test fixture, nothing ever
|
|
8
|
+
// did: no kernel module wrote the file, so `probe concurrency`'s `dialFrom()` read a fan-out dial
|
|
9
|
+
// that was never recorded and reported the effective default every time, indistinguishable from a
|
|
10
|
+
// run that genuinely chose it. A registry entry is not an instruction, and an instruction with no
|
|
11
|
+
// enforcer is how a documented record becomes a file that simply never exists.
|
|
12
|
+
//
|
|
13
|
+
// THE FIX. This is the single writer. It takes the resolved GATE L0 values as flags, builds the
|
|
14
|
+
// exact `RunArgs` object the schema describes, writes it to `.shapeup/<slug>/run-args.json`, and
|
|
15
|
+
// prints that SAME object on stdout — so tech-lead passes the printed value straight to
|
|
16
|
+
// `Workflow({args: ...})` rather than re-typing it a second time. One construction, not two: the
|
|
17
|
+
// file on disk and the value the workflow actually launches with can no longer disagree.
|
|
18
|
+
//
|
|
19
|
+
// `wallClockS` is NOT one of this command's flags and never lands in the object it writes.
|
|
20
|
+
// `--wall-clock-budget` is consumed earlier, by `harness init run` (see `RECEIPT_VERSION` and
|
|
21
|
+
// `config.wall_clock_budget_s` in `./run.mjs`) — the deadline breaker reads that receipt field
|
|
22
|
+
// directly and was never going to see this launch's `RunArgs` at all. Naming a third field here
|
|
23
|
+
// that nothing reads is the exact defect this command exists to close, not one to reintroduce.
|
|
24
|
+
//
|
|
25
|
+
// USAGE
|
|
26
|
+
// node `harness init run-args` --slug <slug> --auto-level interactive|auto|unattended \
|
|
27
|
+
// --exec-model <name> [--eval-model <name>] [--qa-model <name>] \
|
|
28
|
+
// --max-rounds N --attempts N --plugin-root <dir> \
|
|
29
|
+
// [--run-id <id>] [--answers <preset|path>] [--lane full|tiny] \
|
|
30
|
+
// [--no-eval] [--no-qa] [--adversarial-verify] [--parallel-scopes N] [--cwd <dir>]
|
|
31
|
+
//
|
|
32
|
+
// `--eval-model` is required unless `--no-eval` is set — an EVAL-skipping run never resolves one.
|
|
33
|
+
// `--run-id`, given no explicit value, is read off the run's own receipt (the run must already be
|
|
34
|
+
// open — see `harness init run`), never invented.
|
|
35
|
+
//
|
|
36
|
+
// Prints the written `RunArgs` object as JSON on stdout. Exit 0 on success, 2 on a usage error,
|
|
37
|
+
// 3 when no run is open at `--slug` (the receipt is what makes this artifact meaningful at all).
|
|
38
|
+
|
|
39
|
+
import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
|
|
40
|
+
import { dirname, join, resolve } from "node:path";
|
|
41
|
+
import { fileURLToPath } from "node:url";
|
|
42
|
+
import { runArgs } from "../lib/argv.mjs";
|
|
43
|
+
import { localRoot, RECEIPT_FILE, runArgsPath, runIdFromReceipt } from "../lib/paths.mjs";
|
|
44
|
+
|
|
45
|
+
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
46
|
+
/** The plugin root, resolved from where this file actually is (`kernel/init/` → repo root). */
|
|
47
|
+
export const PLUGIN_ROOT = resolve(HERE, "../..");
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* The `autoLevel` enum, read from the schema that defines it — never hand-typed here.
|
|
51
|
+
*
|
|
52
|
+
* `domain.schema.json`'s `$defs/RunArgs.properties.autoLevel.enum` is the one place this set is
|
|
53
|
+
* declared. A `new Set([...])` beside it, inside the very module written to close run-argument
|
|
54
|
+
* contract drift (see this file's own banner), would be that drift repeating one level down —
|
|
55
|
+
* the same class of failure this module exists to close, applied here to an *enum's values*
|
|
56
|
+
* instead of RunArgs *field* names. Mirrors `kernel/verify/skills.mjs`'s `roster()`, which
|
|
57
|
+
* derives `WorkerName` the same way.
|
|
58
|
+
*
|
|
59
|
+
* @param {string} [root=PLUGIN_ROOT] - Plugin root the schema is read from — overridable so a test
|
|
60
|
+
* can point this at a scratch copy of the schema and prove the result grows and shrinks with it,
|
|
61
|
+
* never with an edit to this function.
|
|
62
|
+
* @returns {string[]} The enum, in schema order.
|
|
63
|
+
* @throws {Error} When the schema is missing or does not carry the enum.
|
|
64
|
+
*/
|
|
65
|
+
export function autoLevels(root = PLUGIN_ROOT) {
|
|
66
|
+
const schemaPath = join(root, "kernel/schemas/domain.schema.json");
|
|
67
|
+
const schema = JSON.parse(readFileSync(schemaPath, "utf8"));
|
|
68
|
+
const levels = schema?.$defs?.RunArgs?.properties?.autoLevel?.enum;
|
|
69
|
+
if (!Array.isArray(levels) || !levels.length) {
|
|
70
|
+
throw new Error(`${schemaPath} carries no $defs/RunArgs.properties.autoLevel.enum — the auto-level set cannot be derived`);
|
|
71
|
+
}
|
|
72
|
+
return levels;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** The typed argv contract (see `../lib/argv.mjs`). */
|
|
76
|
+
export const ARGV_SPEC = {
|
|
77
|
+
usage: `harness.mjs init run-args --slug <slug> --auto-level ${autoLevels().join("|")} ` +
|
|
78
|
+
'--exec-model <name> [--eval-model <name>] [--qa-model <name>] --max-rounds N --attempts N ' +
|
|
79
|
+
'--plugin-root <dir> [--run-id <id>] [--answers <preset|path>] [--lane full|tiny] ' +
|
|
80
|
+
'[--no-eval] [--no-qa] [--adversarial-verify] [--parallel-scopes N] [--cwd <dir>]',
|
|
81
|
+
_: { arity: 0, max: 0, name: "(no positional operands)" },
|
|
82
|
+
cwd: { type: "path" },
|
|
83
|
+
slug: { type: "str", required: true },
|
|
84
|
+
"run-id": { type: "str" },
|
|
85
|
+
"auto-level": { type: "str", required: true },
|
|
86
|
+
answers: { type: "str" },
|
|
87
|
+
lane: { type: "str" },
|
|
88
|
+
"exec-model": { type: "str", required: true },
|
|
89
|
+
"eval-model": { type: "str" },
|
|
90
|
+
"qa-model": { type: "str" },
|
|
91
|
+
"max-rounds": { type: "int", min: 1, required: true },
|
|
92
|
+
attempts: { type: "int", min: 1, required: true },
|
|
93
|
+
"plugin-root": { type: "path", required: true },
|
|
94
|
+
"no-eval": { type: "flag" },
|
|
95
|
+
"no-qa": { type: "flag" },
|
|
96
|
+
"adversarial-verify": { type: "flag" },
|
|
97
|
+
"parallel-scopes": { type: "int", min: 1 },
|
|
98
|
+
};
|
|
99
|
+
|
|
100
|
+
function fail(code, msg) {
|
|
101
|
+
console.error(msg);
|
|
102
|
+
process.exit(code);
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Drop `undefined`-valued keys, one level deep on the named nested objects — so the JSON this
|
|
107
|
+
* writes never carries a literal `"eval": undefined` for a skipped role or an unset switch.
|
|
108
|
+
*
|
|
109
|
+
* @param {object} o - The candidate RunArgs object.
|
|
110
|
+
* @returns {object} The same shape with every `undefined` leaf and empty nested object removed.
|
|
111
|
+
*/
|
|
112
|
+
function pruned(o) {
|
|
113
|
+
const out = {};
|
|
114
|
+
for (const [k, v] of Object.entries(o)) {
|
|
115
|
+
if (v === undefined) continue;
|
|
116
|
+
if (v && typeof v === "object" && !Array.isArray(v)) {
|
|
117
|
+
const inner = pruned(v);
|
|
118
|
+
if (Object.keys(inner).length) out[k] = inner;
|
|
119
|
+
continue;
|
|
120
|
+
}
|
|
121
|
+
out[k] = v;
|
|
122
|
+
}
|
|
123
|
+
return out;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* Build the `RunArgs` object — pure, so the structural suite can assert its shape without a
|
|
128
|
+
* filesystem. Mirrors `domain.schema.json` `$defs/RunArgs` exactly: this is the one place that
|
|
129
|
+
* shape is constructed, so a field this function does not carry cannot reach the file either.
|
|
130
|
+
*
|
|
131
|
+
* @param {object} o - Resolved inputs (destructured); optional fields may be `undefined`.
|
|
132
|
+
* @returns {object} The RunArgs object, pruned of unset optionals.
|
|
133
|
+
*/
|
|
134
|
+
export function buildRunArgs({
|
|
135
|
+
slug, runId, autoLevel, answers, lane, execModel, evalModel, qaModel,
|
|
136
|
+
maxRounds, attemptBudget, pluginRoot, startedAt,
|
|
137
|
+
noEval, noQa, adversarialVerify, maxParallelScopes,
|
|
138
|
+
}) {
|
|
139
|
+
return pruned({
|
|
140
|
+
slug, runId, autoLevel, answers, lane,
|
|
141
|
+
models: { exec: execModel, eval: evalModel, qa: qaModel },
|
|
142
|
+
budgets: { maxRounds, attemptBudget },
|
|
143
|
+
pluginRoot, startedAt,
|
|
144
|
+
noEval, noQa, adversarialVerify, maxParallelScopes,
|
|
145
|
+
});
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/**
|
|
149
|
+
* Write `.shapeup/<slug>/run-args.json`, or refuse (exit 3) when no run is open at `--slug`.
|
|
150
|
+
*
|
|
151
|
+
* @param {string[]} rawArgv - The subcommand's own arguments (harness.mjs strips the verb words).
|
|
152
|
+
* @returns {(Promise<void>|void)} Settles when the subcommand has written its output; every path
|
|
153
|
+
* calls `process.exit()` with the subcommand's documented code rather than returning.
|
|
154
|
+
*/
|
|
155
|
+
export function cli(rawArgv) {
|
|
156
|
+
const args = runArgs(ARGV_SPEC, rawArgv);
|
|
157
|
+
const cwd = args.cwd || process.cwd();
|
|
158
|
+
|
|
159
|
+
if (!autoLevels().includes(args.autoLevel)) {
|
|
160
|
+
fail(2, `--auto-level must be one of: ${autoLevels().join(", ")}`);
|
|
161
|
+
}
|
|
162
|
+
if (!args.noEval && !args.evalModel) {
|
|
163
|
+
fail(2, "--eval-model is required unless --no-eval is set — an EVAL-skipping run never resolves one.");
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
const runRoot = localRoot(cwd, args.slug);
|
|
167
|
+
const receiptPath = join(runRoot, RECEIPT_FILE);
|
|
168
|
+
if (!existsSync(receiptPath)) {
|
|
169
|
+
fail(3, [
|
|
170
|
+
`✋ init run-args: no open run at ${runRoot} — run "harness init run" first (GATE L0.1).`,
|
|
171
|
+
"",
|
|
172
|
+
"`run-args.json` records what an OPEN run was launched with; writing one with no receipt",
|
|
173
|
+
"behind it would leave a launch record for a run that, everywhere else in the harness, never",
|
|
174
|
+
"started.",
|
|
175
|
+
].join("\n"));
|
|
176
|
+
}
|
|
177
|
+
let receipt = null;
|
|
178
|
+
try { receipt = JSON.parse(readFileSync(receiptPath, "utf8")); } catch { /* handled below */ }
|
|
179
|
+
const runId = args.runId ?? runIdFromReceipt(receipt) ?? undefined;
|
|
180
|
+
|
|
181
|
+
const runArgsObj = buildRunArgs({
|
|
182
|
+
slug: args.slug,
|
|
183
|
+
runId,
|
|
184
|
+
autoLevel: args.autoLevel,
|
|
185
|
+
answers: args.answers ?? undefined,
|
|
186
|
+
lane: args.lane ?? undefined,
|
|
187
|
+
execModel: args.execModel,
|
|
188
|
+
evalModel: args.evalModel ?? undefined,
|
|
189
|
+
qaModel: args.qaModel ?? undefined,
|
|
190
|
+
maxRounds: args.maxRounds,
|
|
191
|
+
attemptBudget: args.attempts,
|
|
192
|
+
pluginRoot: args.pluginRoot,
|
|
193
|
+
startedAt: new Date().toISOString(),
|
|
194
|
+
noEval: args.noEval || undefined,
|
|
195
|
+
noQa: args.noQa || undefined,
|
|
196
|
+
adversarialVerify: args.adversarialVerify || undefined,
|
|
197
|
+
maxParallelScopes: args.parallelScopes ?? undefined,
|
|
198
|
+
});
|
|
199
|
+
|
|
200
|
+
mkdirSync(runRoot, { recursive: true });
|
|
201
|
+
writeFileSync(runArgsPath(cwd, args.slug), JSON.stringify(runArgsObj, null, 2) + "\n", "utf8");
|
|
202
|
+
|
|
203
|
+
// Printed, not just written — so the caller's `Workflow({args: ...})` reads this stdout instead
|
|
204
|
+
// of re-assembling the object from the flags it just typed.
|
|
205
|
+
console.log(JSON.stringify(runArgsObj, null, 2));
|
|
206
|
+
}
|
package/kernel/init/run.mjs
CHANGED
|
@@ -224,6 +224,16 @@ export function runFrontmatter({ slug, config, startedAt }) {
|
|
|
224
224
|
"deploy: ~",
|
|
225
225
|
`started_at: ${startedAt}`,
|
|
226
226
|
"closed_at: ~",
|
|
227
|
+
// The cause a terminal status ended on, written alongside `closed_at` by
|
|
228
|
+
// `probe resume --close` (kernel/probe/resume.mjs's closeRun) — the two land in one write, so a
|
|
229
|
+
// closed run's ledger never carries a timestamp with no reason beside it.
|
|
230
|
+
"close_cause: ~",
|
|
231
|
+
// The once-only guard's OWN record of what closed this run — deliberately separate from
|
|
232
|
+
// `status:` above, which every phase rewrites (`setRunStatus`) for the life of the run,
|
|
233
|
+
// including the product's own ship path immediately before `probe resume --close` runs. Only
|
|
234
|
+
// `closeRun` ever writes this line, so it is the one field an intervening `status:` rewrite
|
|
235
|
+
// cannot move (kernel/probe/resume.mjs's closeRun docblock has the measured scenario).
|
|
236
|
+
"closed_status: ~",
|
|
227
237
|
"---",
|
|
228
238
|
"",
|
|
229
239
|
`# Harness run — ${slug}`,
|