shapeup-sdlc 3.5.0 → 3.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/AGENTS.md +15 -4
- package/README.md +1 -1
- package/kernel/compile.mjs +55 -7
- package/kernel/harness.mjs +11 -4
- package/kernel/init/run-args.mjs +206 -0
- package/kernel/init/run.mjs +10 -0
- package/kernel/lib/paths.mjs +10 -0
- package/kernel/probe/attempts.mjs +135 -0
- package/kernel/probe/concurrency.mjs +31 -6
- package/kernel/probe/digest.mjs +15 -1
- package/kernel/probe/owner.mjs +4 -1
- package/kernel/probe/resume.mjs +314 -5
- package/kernel/probe/rounds.mjs +104 -0
- package/kernel/reduce/ingest.mjs +53 -12
- package/kernel/reduce/ship.mjs +15 -30
- package/kernel/reduce/snapshot.mjs +23 -2
- package/kernel/report/export.mjs +54 -2
- package/kernel/report/facts.mjs +24 -2
- package/{skills/tech-lead → kernel}/schemas/domain.schema.json +15 -12
- package/kernel/verify/envelope.mjs +2 -2
- package/kernel/verify/skills.mjs +1 -1
- package/package.json +1 -1
- package/skills/ba-pitch-analyzer/SKILL.md +1 -1
- package/skills/ba-pitch-analyzer/references/doc-schemas.md +6 -0
- package/skills/coach/SKILL.md +8 -2
- package/skills/hill-chart/SKILL.md +3 -4
- package/skills/scope-hammer/SKILL.md +11 -3
- package/skills/tech-lead/SKILL.md +10 -10
- package/skills/tech-lead/references/gates.md +48 -11
- package/skills/tech-lead/references/protocol.md +4 -2
- package/skills/tech-lead/workflows/shapeup-run.js +164 -40
- package/skills/translator/SKILL.md +1 -1
- /package/{skills/tech-lead → kernel}/schemas/gate-answers.schema.json +0 -0
- /package/{skills/tech-lead → kernel}/schemas/work-order.schema.json +0 -0
- /package/{skills/tech-lead → kernel}/schemas/work-result.schema.json +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "shapeup-sdlc-plugin",
|
|
3
3
|
"displayName": "ShapeUp SDLC Plugin",
|
|
4
|
-
"version": "3.
|
|
4
|
+
"version": "3.7.0",
|
|
5
5
|
"description": "Shape Up SDLC harness for Claude Code: shaping, intake, orient, scope-mapping, building (T0-verified, sandboxed, scope-contracted), evaluation and QA skills orchestrated by a tech-lead.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "Liberty Nguyen",
|
package/AGENTS.md
CHANGED
|
@@ -11,8 +11,9 @@ Skills and commands are named short throughout this file; every one of them reso
|
|
|
11
11
|
- Attested, not assumed: a dispatch leaves a receipt naming the skill that ran, and ingest refuses an orchestrated result that has none. A dispatch that fails is answered by the sub-agent improvising the craft, which every other check accepts — so "the artifact exists" is not evidence that the shipped skill produced it. `--no-receipt-check` is the way through when the receipt channel itself fails.
|
|
12
12
|
- GATE L2 is advisory — warns when EVAL runs over unfinished tasks, permits the call (per-machine board, operator asked; ADR-0001) — a signal, not a bug.
|
|
13
13
|
- Sign-off is a file: each gate resolves from the answer set (`ci`/`guarded`/`interactive`) — cross, stop for the PO, or abort; the decision's source is ledgered.
|
|
14
|
+
- **Every ending a run does not come back from records a close** — its status, its cause and its timestamp — derived from the ending itself rather than from a hand-kept pair, so an ending nobody enumerated cannot silently record nothing. A breaker trip that routes to GATE H closes the run as `escalated`, carrying the breaker in its cause; a pause records no close **by design**, because a relaunch resumes it.
|
|
14
15
|
- **Your scope cut decides your concurrency, not the dial.** Two scopes that both declare a write to one path never build at the same time: `shared` substrate is the sanctioned escape from disjointness, and an edit is read-modify-write, so concurrent writers silently drop each other's work. An entry point listed as writable by five scopes therefore builds five scopes one at a time whatever `--parallel-scopes` says. The run states the ceiling its contracts actually permit before dispatching anything, as a BUILD-order line — a peak below that ceiling is a dispatch problem, a ceiling below the dial is a scope-cut problem, and only the second is fixable by re-cutting. The fix is a cut where exactly one scope owns each entry point. Note the ceiling is narrated, not gated: it does not travel in the ⏸ L1b block, so an unattended run carries it only in its log.
|
|
15
|
-
- The build+eval loop breaks only three ways ✦: EVAL PASS → QA → Ship; outer `round_budget` exhausted; opt-in `wall_clock_budget_s` tripped (the wall-clock axis event counters miss). Budget trips route to GATE H — ship what's green, never kill the run from outside. A scope exhausting its per-scope `attempt_budget` (T0 attempts) queues a GATE H proposal, never blocks the round.
|
|
16
|
+
- The build+eval loop breaks only three ways ✦: EVAL PASS → QA → Ship; outer `round_budget` exhausted; opt-in `wall_clock_budget_s` tripped (the wall-clock axis event counters miss). Budget trips route to GATE H — ship what's green, never kill the run from outside. A scope exhausting its per-scope `attempt_budget` (T0 attempts) queues a GATE H proposal, never blocks the round. **An attempt is spent only when the attested channels say so** — a dispatch receipt, plus a leg row or a WorkResult. A compiled order and a T0 verdict are both writable by the scope being judged, so neither counts on its own, and an attempt still in flight holds the breaker open rather than tripping it. Opening the next attempt over an unanswered one is refused outright, because grading a tree the previous attempt may still be writing spends a budget on work nobody did.
|
|
16
17
|
|
|
17
18
|
### Phase 1 — Shaping (`/shapeup`)
|
|
18
19
|
1. Set Boundaries → `/shapeup shaping`
|
|
@@ -65,12 +66,22 @@ Everything discovered funnels into `.shapeup/<slug>/discovery/ledger.md` (Orient
|
|
|
65
66
|
|
|
66
67
|
## Setup & Execution
|
|
67
68
|
|
|
68
|
-
- Orders/results live in `.shapeup/<slug>/orders|results/`; the envelope schemas ship
|
|
69
|
+
- Orders/results live in `.shapeup/<slug>/orders|results/`; the envelope schemas ship with the plugin runtime, not with any individual skill, so every worker validates against the same copy.
|
|
69
70
|
- The plugin's run entry points need a one-time permission grant — `npx shapeup-sdlc init` writes it into `.claude/settings.json` (`permissions.allow`); without it a headless run stalls at step one. That grant is necessary, not sufficient: it covers the run's own deterministic entry points, not the generic file edits every worker skill makes constantly, or any command a worker reaches for beyond the grant's own exact shape. A truly unattended run also needs a Claude Code permission mode that covers those (`acceptEdits` at minimum) — the plugin cannot grant that on your behalf.
|
|
71
|
+
- The grant is necessary but sits under two more layers this plugin cannot reach either. A fresh
|
|
72
|
+
checkout is an **untrusted workspace**, and Claude Code discards the whole permission grant — every
|
|
73
|
+
rule in it, not only this one — until the workspace is trusted; the installer detects that state and
|
|
74
|
+
tells you, because trusting a directory to run code from is your decision to make, never a package's
|
|
75
|
+
to make for you. Above workspace trust sits Claude Code's own **auto-mode classifier**, which can
|
|
76
|
+
still block a call the grant already covers, invisibly to anything this plugin ships — no rule or
|
|
77
|
+
hook here can see it, let alone override it. When either layer stops an automated run, the
|
|
78
|
+
documented fallback is to drive Build by hand: the same per-dispatch cycle named in the Build
|
|
79
|
+
Vertically step above — compile, dispatch, ingest, verify — run one call at a time instead of
|
|
80
|
+
through the chained launch, until the run can resume unattended again.
|
|
70
81
|
- Two storage tiers (ADR-0001): COMMITTED `shapeup/<slug>/` (shaping, spec, scopes, wiring-map, project-profile, requirements, hill, `REPORT.md` frozen at L4) vs GITIGNORED `.shapeup/` (board, orders/results, T0/eval/QA artifacts, ledgers, metrics, gate answers, and the run scripts staged for launch).
|
|
71
82
|
- **The run launches from a copy inside your project, and it has to.** The Workflow tool loads a script only from a directory the session may already read; the plugin installs outside your project, so naming the shipped path is refused before the run begins and no permission rule repairs it — the grant authorises the tool, not what it may read. Opening a run therefore re-copies the run scripts to `.shapeup/workflows/` and reports the path the launch names. A run in flight keeps the copy it started with: an upgrade reaches the next run, not the current round.
|
|
72
|
-
- **A write the substrate does not cover is denied for as long as the dispatch is in flight, and no longer.** A dispatch is live from the moment its order is compiled until a result for *that* dispatch lands, and liveness is read off the run's order set — not off the pointer that names the run, which outlives it.
|
|
73
|
-
- Every run has a `run_id` — the receipt mints it, and orders, T0 artifacts, trial rows, agent-call journal rows and hook decisions all carry it. It is the only key that separates two runs of the same feature: everything else (`order_id`, round/attempt) repeats. It is **not** a time boundary — a relaunch resumes the same run and reuses the key, so one `run_id` legitimately spans every launch after a paused gate or a kill, with hours of wall clock between them, and `orders/<id>.json` is rewritten by each. Anything measuring elapsed time reads the append-only records, never the span of a key. SHIP S.7 exports the run's records as fact tables under `.shapeup/exports/<run_id>/` before the run trace is superseded
|
|
83
|
+
- **A write the substrate does not cover is denied for as long as the dispatch is in flight, and no longer.** A dispatch is live from the moment its order is compiled until a result for *that* dispatch lands, and liveness is read off the run's order set — not off the pointer that names the run, which outlives it. A run closed any other way than shipping — escalated, aborted, killed outright — can leave one still open, and while the run's pointer is still on disk the fence holds on that order exactly as if the run were live. The pointer is what actually switches it off, though: the fence is enforced only while that pointer exists on disk, so a close that removes it releases the fence whatever the order set still says — restoring the pointer flips it straight back to denying. A **ship** close does not itself answer what it leaves outstanding — a ship close can retire the pointer over an order that never got a result — so it is not special because nothing is left unanswered; it is special only because retiring the pointer is the one lever every close needs pulled, and `reduce ship` pulls it for you. Whichever way a run closes, an order it leaves unanswered stays genuinely unresolved — not merely un-fenced — until `init run --force` runs: it writes a synthetic result for every order the closed run left unanswered, so the fence lifts without waiting on a worker that is never coming back. What the fence actually gates is narrower than "the project," too — it is this assistant's own edit path (a direct file write or edit call) outside the live order's substrate; a shell command, `git`, or any other editor still writes straight through it. Re-dispatch is fenced again, and the committed tier is never a worker's to write either way: those files belong to the orchestrator, whose window is a phase boundary rather than the middle of somebody else's order.
|
|
84
|
+
- Every run has a `run_id` — the receipt mints it, and orders, T0 artifacts, trial rows, agent-call journal rows and hook decisions all carry it. It is the only key that separates two runs of the same feature: everything else (`order_id`, round/attempt) repeats. It is **not** a time boundary — a relaunch resumes the same run and reuses the key, so one `run_id` legitimately spans every launch after a paused gate or a kill, with hours of wall clock between them, and `orders/<id>.json` is rewritten by each. Anything measuring elapsed time reads the append-only records, never the span of a key. SHIP S.7 exports the run's records as fact tables under `.shapeup/exports/<run_id>/` before the run trace is superseded — and so does **every other terminal ending**: a run that aborts, escalates or stops at GATE H exports on its close, because the runs whose records are worth most are the ones that never reach the Ship phase. The export is advisory at every one of them: a failed export degrades the trace and is reported in the run's state warnings, but never turns a close into a non-close. A WorkResult carries no `run_id` and reaches it through `order_id`.
|
|
74
85
|
- Every run projects a **run graph** — `.shapeup/<slug>/graph.jsonl`, append-only, written only by
|
|
75
86
|
`reduce graph`. Two families kept separate: work lineage (Run, Order, Result, Verdict, Trial,
|
|
76
87
|
GateDecision) and domain (Scope, UseCase, Requirement, Seam). It is derived from the artifacts,
|
package/README.md
CHANGED
|
@@ -332,11 +332,11 @@ claude --plugin-dir . # load this working copy without installing
|
|
|
332
332
|
plugin.json # plugin manifest
|
|
333
333
|
marketplace.json # marketplace listing (points at this repo)
|
|
334
334
|
skills/<name>/SKILL.md # the 13 harness skills (+ references/ and assets/)
|
|
335
|
-
skills/tech-lead/schemas/ # the envelope port: WorkOrder, WorkResult, domain registry
|
|
336
335
|
skills/tech-lead/workflows/shapeup-run.js # the BUILD-phase pipeline, on the native Workflow runtime
|
|
337
336
|
kernel/harness.mjs # ONE entry point for every deterministic step; the whole permission grant
|
|
338
337
|
kernel/{verify,reduce,probe,init,report}/ # its subcommands, plus compile and gate at the root
|
|
339
338
|
kernel/lib/ # argv (the typed CLI boundary), paths (+ the run key), contract (shape)
|
|
339
|
+
kernel/schemas/ # the envelope port: WorkOrder, WorkResult, domain registry
|
|
340
340
|
commands/*.md # slash commands (/ship + the 9 phase commands)
|
|
341
341
|
hooks/ # hooks.json + the four walls: safety-spine, gate-intake, sandbox-guard
|
|
342
342
|
# (PreToolUse) + gate-zerowork (Stop, the one blocking hook)
|
package/kernel/compile.mjs
CHANGED
|
@@ -31,7 +31,7 @@ import { fileURLToPath } from "node:url";
|
|
|
31
31
|
import { validate } from "./verify/envelope.mjs";
|
|
32
32
|
import { readTrials } from "./verify/t0.mjs";
|
|
33
33
|
import { runArgs } from "./lib/argv.mjs";
|
|
34
|
-
import { readRunId } from "./lib/paths.mjs";
|
|
34
|
+
import { readRunId, dispatchReceipts, legLedger } from "./lib/paths.mjs";
|
|
35
35
|
// `specDir` is aliased: this module has a local `let specDir` holding the resolved, possibly
|
|
36
36
|
// --spec-overridden directory, and the import is the convention-derived default.
|
|
37
37
|
import {
|
|
@@ -41,6 +41,8 @@ import {
|
|
|
41
41
|
import { readContract, readAllContracts, tasksForScope, SCOPE_CONTRACT } from "./lib/contract.mjs";
|
|
42
42
|
import { writeActiveOrder } from "./probe/resume.mjs";
|
|
43
43
|
import { greenVerdict } from "./probe/t0.mjs";
|
|
44
|
+
import { attemptEvidence, readReceipts } from "./probe/attempts.mjs";
|
|
45
|
+
import { readLegs } from "./probe/leg.mjs";
|
|
44
46
|
import { latestRoundBuild } from "./verify/build.mjs";
|
|
45
47
|
// The SAME matcher the sandbox hook enforces with. "Is this cited file inside this scope's
|
|
46
48
|
// substrate" has to mean exactly what the guard means, or a bug is addressed to a scope that is
|
|
@@ -58,7 +60,7 @@ export const COACHABLE = new Set([
|
|
|
58
60
|
]);
|
|
59
61
|
|
|
60
62
|
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
61
|
-
const ORDER_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "
|
|
63
|
+
const ORDER_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "./schemas/work-order.schema.json"), "utf8"));
|
|
62
64
|
|
|
63
65
|
// --- tiny frontmatter reader (scalar keys + [a, b] inline lists) --------------------------
|
|
64
66
|
/**
|
|
@@ -218,9 +220,15 @@ export function substrateFor(operation, { slug, specDir, scope } = {}) {
|
|
|
218
220
|
const FROZEN_INTAKE = [`${local}/intake.md`, `${local}/breadboard.md`];
|
|
219
221
|
switch (operation) {
|
|
220
222
|
case "execute": case "fix": case "spike":
|
|
223
|
+
// Build legs are the widest window on FROZEN_INTAKE, not an exemption from it: they are the
|
|
224
|
+
// most numerous and longest-lived dispatches in a run, so a doer that can rewrite the staged
|
|
225
|
+
// pitch can rewrite the run's own input truth mid-build. `init run` stages these before any
|
|
226
|
+
// order is live (no live contract yet — nothing to violate) and `translate` writes the
|
|
227
|
+
// COMMITTED copy, not this one, so neither legitimate write is touched by this line.
|
|
221
228
|
return {
|
|
222
229
|
allowed: [...(scope?.allowed_file_substrate || []), `${local}/spikes/**`],
|
|
223
230
|
shared: scope?.shared_substrate || [],
|
|
231
|
+
frozen: [...FROZEN_INTAKE],
|
|
224
232
|
};
|
|
225
233
|
case "analyze":
|
|
226
234
|
return { allowed: [`${spec}/**`, `${local}/**`], frozen: [...FROZEN_INTAKE] };
|
|
@@ -518,24 +526,30 @@ export function bugLocations(bug) {
|
|
|
518
526
|
*
|
|
519
527
|
* OWNERSHIP IS BY SUBSTRATE, because that is what the sandbox enforces: a scope is exactly the set
|
|
520
528
|
* of files its worker may write, so a scope whose substrate excludes the cited line cannot fix it
|
|
521
|
-
* however well it understands the bug.
|
|
529
|
+
* however well it understands the bug. The substrate a scope may write is `allowed ∪ shared` — the
|
|
530
|
+
* same union `sandbox-guard` composes at the fence — so a path declared ONLY in a contract's
|
|
531
|
+
* `shared` list is a file that scope may legitimately write, and the election must see it too —
|
|
532
|
+
* filtering on `allowed` alone elects no one for a shared-only path, and `bugsForScope` then reads
|
|
533
|
+
* that null as "no scope owns this" and fans the bug out to every scope instead of the one or two
|
|
534
|
+
* that declared it.
|
|
522
535
|
*
|
|
523
536
|
* BUT A MATCH IS NOT AN ELECTION. An entry point is routinely SHARED — on the measured run
|
|
524
537
|
* `bin/todo.js` sits in five scopes' substrate at once — so "address it to every scope that
|
|
525
538
|
* matches" hands the same one-line fix to five workers building concurrently against one file.
|
|
526
539
|
* That is a write race the harness sets up itself, and four of the five fixes are waste even when
|
|
527
540
|
* it resolves. So: prefer a scope that owns the file EXCLUSIVELY (allowed, not shared), and among
|
|
528
|
-
* equals
|
|
529
|
-
*
|
|
541
|
+
* equals — every remaining candidate declares it shared, exclusive or not — take the lowest scope
|
|
542
|
+
* id: a rule that needs no coordination to agree with itself, since each leg compiles its own
|
|
543
|
+
* order in its own process.
|
|
530
544
|
*
|
|
531
545
|
* @param {string} path - Repo-relative file the bug cites.
|
|
532
546
|
* @param {Array<{scope_id:string, allowed:string[], shared:string[]}>} scopes - Every scope.
|
|
533
547
|
* @returns {string|null} The elected scope id, or null when no scope may write that file.
|
|
534
548
|
*/
|
|
535
549
|
export function electOwner(path, scopes) {
|
|
536
|
-
const can = (scopes || []).filter((s) => matchesAny(path, s.allowed));
|
|
550
|
+
const can = (scopes || []).filter((s) => matchesAny(path, s.allowed) || matchesAny(path, s.shared || []));
|
|
537
551
|
if (!can.length) return null;
|
|
538
|
-
const exclusive = can.filter((s) => !matchesAny(path, s.shared || []));
|
|
552
|
+
const exclusive = can.filter((s) => matchesAny(path, s.allowed) && !matchesAny(path, s.shared || []));
|
|
539
553
|
return (exclusive.length ? exclusive : can).map((s) => s.scope_id).sort()[0];
|
|
540
554
|
}
|
|
541
555
|
|
|
@@ -804,6 +818,40 @@ export async function cli(rawArgv) {
|
|
|
804
818
|
const round = flag("round");
|
|
805
819
|
const attempt = flag("attempt");
|
|
806
820
|
|
|
821
|
+
// AN ATTEMPT MAY NOT OPEN OVER AN UNANSWERED ONE — the write half of the attested-channel rule.
|
|
822
|
+
//
|
|
823
|
+
// `--attempt` is a flag: the CALLER picks the number, and nothing here used to check that the
|
|
824
|
+
// previous one had come back. Measured on a consumer run: attempt 2 was compiled and T0-verified
|
|
825
|
+
// while attempt 1 was still in flight, so the round graded the tree attempt 1 was still writing,
|
|
826
|
+
// counted the attempt, and stopped at GATE H with most of its budget and clock unspent. An order
|
|
827
|
+
// and a T0 verdict are both writable by the very scope being judged; only a dispatch receipt, a
|
|
828
|
+
// leg row or a WorkResult attests that work actually happened.
|
|
829
|
+
//
|
|
830
|
+
// FAILS OPEN, NEVER CLOSED, unless the bad state is positively proven. The proof required is the
|
|
831
|
+
// receipts ledger EXISTING while carrying no row for the previous attempt: a lane that does not
|
|
832
|
+
// attest dispatches at all (no ledger on disk) cannot be judged by this rule and is waved
|
|
833
|
+
// through, so `--tiny`, a prose round loop and a standalone build are untouched.
|
|
834
|
+
if (scope?.scope_id && round && attempt > 1) {
|
|
835
|
+
const receiptsPath = dispatchReceipts(cwd, slug);
|
|
836
|
+
if (existsSync(receiptsPath)) {
|
|
837
|
+
const prev = attemptEvidence(
|
|
838
|
+
cwd, slug, scope.scope_id, round, attempt - 1,
|
|
839
|
+
readReceipts(receiptsPath), readLegs(legLedger(cwd, slug)),
|
|
840
|
+
);
|
|
841
|
+
if (prev.state !== "spent") {
|
|
842
|
+
const why = prev.state === "unattested"
|
|
843
|
+
? "no dispatch receipt was ever written for it"
|
|
844
|
+
: "it was dispatched but has neither a leg-completion row nor a WorkResult";
|
|
845
|
+
console.error(
|
|
846
|
+
`compile-order: refusing to open attempt ${attempt} for "${scope.scope_id}" in round ${round} — ` +
|
|
847
|
+
`attempt ${attempt - 1} (${prev.orderId}) is unanswered: ${why}. Grading a tree the previous ` +
|
|
848
|
+
`attempt may still be writing counts an attempt that never ran, and spends a budget on work ` +
|
|
849
|
+
`nobody did. Wait for it to return, or record its outcome, before opening the next one.`);
|
|
850
|
+
process.exit(3);
|
|
851
|
+
}
|
|
852
|
+
}
|
|
853
|
+
}
|
|
854
|
+
|
|
807
855
|
// The fix round's inbound evidence. Derived here, from the ledgered verdict, for every lane —
|
|
808
856
|
// the workflow, `--tiny`, the prose round loop and a standalone `/build` all compile through
|
|
809
857
|
// this line, and none of them can pass a payload to a build order (see the banner above).
|
package/kernel/harness.mjs
CHANGED
|
@@ -32,7 +32,7 @@
|
|
|
32
32
|
// gate An answer file with a source, not a vibe.
|
|
33
33
|
// probe resume · t0 · stats · digest · Read-only queries over run state. `concurrency`
|
|
34
34
|
// concurrency · leg · eval · answers how many legs ran at once and what the
|
|
35
|
-
// owner · requirements
|
|
35
|
+
// owner · requirements · attempts
|
|
36
36
|
// fan-out bought, and refuses a figure the record set
|
|
37
37
|
// cannot support rather than printing a plausible one.
|
|
38
38
|
// `leg` answers whether a scope's work reached the
|
|
@@ -50,7 +50,14 @@
|
|
|
50
50
|
// answers which pitch clause a verdict reached, joined
|
|
51
51
|
// through the plan's own covers: edge — the L4 line and
|
|
52
52
|
// GATE H's census cite it for the same reason.
|
|
53
|
-
//
|
|
53
|
+
// `attempts` answers how many of a scope's attempts are
|
|
54
|
+
// ATTESTED (a dispatch receipt AND a leg row or a
|
|
55
|
+
// WorkResult), never the order set or the T0 verdict
|
|
56
|
+
// set alone — the round loop's inner breaker and
|
|
57
|
+
// scope-hammer's census both cite it, so they cannot
|
|
58
|
+
// disagree about the same exhaustion again.
|
|
59
|
+
// init run · fit · run-args Opens a run, or refuses it (exit 3). `run-args`
|
|
60
|
+
// writes GATE L0.9b's launch record and echoes it.
|
|
54
61
|
// report export Projects the run's records as fact tables.
|
|
55
62
|
// compile The WorkOrder: schema-valid or nothing is dispatched.
|
|
56
63
|
//
|
|
@@ -85,9 +92,9 @@ export const ROUTES = {
|
|
|
85
92
|
resume: "./probe/resume.mjs", t0: "./probe/t0.mjs", stats: "./probe/stats.mjs",
|
|
86
93
|
digest: "./probe/digest.mjs", concurrency: "./probe/concurrency.mjs",
|
|
87
94
|
leg: "./probe/leg.mjs", eval: "./probe/eval.mjs", owner: "./probe/owner.mjs",
|
|
88
|
-
requirements: "./probe/requirements.mjs",
|
|
95
|
+
requirements: "./probe/requirements.mjs", attempts: "./probe/attempts.mjs",
|
|
89
96
|
},
|
|
90
|
-
init: { run: "./init/run.mjs", fit: "./init/fit.mjs" },
|
|
97
|
+
init: { run: "./init/run.mjs", fit: "./init/fit.mjs", "run-args": "./init/run-args.mjs" },
|
|
91
98
|
report: { export: "./report/export.mjs", _default: "export" },
|
|
92
99
|
gate: "./gate.mjs",
|
|
93
100
|
compile: "./compile.mjs",
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
// GATE L0.9b — THE LAUNCH RECORD, GIVEN A WRITER.
|
|
3
|
+
//
|
|
4
|
+
// WHY THIS EXISTS. `domain.schema.json`'s `RunArgs` entry calls `.shapeup/<slug>/run-args.json`
|
|
5
|
+
// "the only artifact that records what a run was configured with" and names tech-lead as the
|
|
6
|
+
// writer — but "tech-lead" meant a paragraph of prose telling the orchestrating session to
|
|
7
|
+
// assemble a JSON object by hand and `Write` it. Outside a structural-test fixture, nothing ever
|
|
8
|
+
// did: no kernel module wrote the file, so `probe concurrency`'s `dialFrom()` read a fan-out dial
|
|
9
|
+
// that was never recorded and reported the effective default every time, indistinguishable from a
|
|
10
|
+
// run that genuinely chose it. A registry entry is not an instruction, and an instruction with no
|
|
11
|
+
// enforcer is how a documented record becomes a file that simply never exists.
|
|
12
|
+
//
|
|
13
|
+
// THE FIX. This is the single writer. It takes the resolved GATE L0 values as flags, builds the
|
|
14
|
+
// exact `RunArgs` object the schema describes, writes it to `.shapeup/<slug>/run-args.json`, and
|
|
15
|
+
// prints that SAME object on stdout — so tech-lead passes the printed value straight to
|
|
16
|
+
// `Workflow({args: ...})` rather than re-typing it a second time. One construction, not two: the
|
|
17
|
+
// file on disk and the value the workflow actually launches with can no longer disagree.
|
|
18
|
+
//
|
|
19
|
+
// `wallClockS` is NOT one of this command's flags and never lands in the object it writes.
|
|
20
|
+
// `--wall-clock-budget` is consumed earlier, by `harness init run` (see `RECEIPT_VERSION` and
|
|
21
|
+
// `config.wall_clock_budget_s` in `./run.mjs`) — the deadline breaker reads that receipt field
|
|
22
|
+
// directly and was never going to see this launch's `RunArgs` at all. Naming a third field here
|
|
23
|
+
// that nothing reads is the exact defect this command exists to close, not one to reintroduce.
|
|
24
|
+
//
|
|
25
|
+
// USAGE
|
|
26
|
+
// node `harness init run-args` --slug <slug> --auto-level interactive|auto|unattended \
|
|
27
|
+
// --exec-model <name> [--eval-model <name>] [--qa-model <name>] \
|
|
28
|
+
// --max-rounds N --attempts N --plugin-root <dir> \
|
|
29
|
+
// [--run-id <id>] [--answers <preset|path>] [--lane full|tiny] \
|
|
30
|
+
// [--no-eval] [--no-qa] [--adversarial-verify] [--parallel-scopes N] [--cwd <dir>]
|
|
31
|
+
//
|
|
32
|
+
// `--eval-model` is required unless `--no-eval` is set — an EVAL-skipping run never resolves one.
|
|
33
|
+
// `--run-id`, given no explicit value, is read off the run's own receipt (the run must already be
|
|
34
|
+
// open — see `harness init run`), never invented.
|
|
35
|
+
//
|
|
36
|
+
// Prints the written `RunArgs` object as JSON on stdout. Exit 0 on success, 2 on a usage error,
|
|
37
|
+
// 3 when no run is open at `--slug` (the receipt is what makes this artifact meaningful at all).
|
|
38
|
+
|
|
39
|
+
import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
|
|
40
|
+
import { dirname, join, resolve } from "node:path";
|
|
41
|
+
import { fileURLToPath } from "node:url";
|
|
42
|
+
import { runArgs } from "../lib/argv.mjs";
|
|
43
|
+
import { localRoot, RECEIPT_FILE, runArgsPath, runIdFromReceipt } from "../lib/paths.mjs";
|
|
44
|
+
|
|
45
|
+
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
46
|
+
/** The plugin root, resolved from where this file actually is (`kernel/init/` → repo root). */
|
|
47
|
+
export const PLUGIN_ROOT = resolve(HERE, "../..");
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* The `autoLevel` enum, read from the schema that defines it — never hand-typed here.
|
|
51
|
+
*
|
|
52
|
+
* `domain.schema.json`'s `$defs/RunArgs.properties.autoLevel.enum` is the one place this set is
|
|
53
|
+
* declared. A `new Set([...])` beside it, inside the very module written to close run-argument
|
|
54
|
+
* contract drift (see this file's own banner), would be that drift repeating one level down —
|
|
55
|
+
* the same class of failure this module exists to close, applied here to an *enum's values*
|
|
56
|
+
* instead of RunArgs *field* names. Mirrors `kernel/verify/skills.mjs`'s `roster()`, which
|
|
57
|
+
* derives `WorkerName` the same way.
|
|
58
|
+
*
|
|
59
|
+
* @param {string} [root=PLUGIN_ROOT] - Plugin root the schema is read from — overridable so a test
|
|
60
|
+
* can point this at a scratch copy of the schema and prove the result grows and shrinks with it,
|
|
61
|
+
* never with an edit to this function.
|
|
62
|
+
* @returns {string[]} The enum, in schema order.
|
|
63
|
+
* @throws {Error} When the schema is missing or does not carry the enum.
|
|
64
|
+
*/
|
|
65
|
+
export function autoLevels(root = PLUGIN_ROOT) {
|
|
66
|
+
const schemaPath = join(root, "kernel/schemas/domain.schema.json");
|
|
67
|
+
const schema = JSON.parse(readFileSync(schemaPath, "utf8"));
|
|
68
|
+
const levels = schema?.$defs?.RunArgs?.properties?.autoLevel?.enum;
|
|
69
|
+
if (!Array.isArray(levels) || !levels.length) {
|
|
70
|
+
throw new Error(`${schemaPath} carries no $defs/RunArgs.properties.autoLevel.enum — the auto-level set cannot be derived`);
|
|
71
|
+
}
|
|
72
|
+
return levels;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/** The typed argv contract (see `../lib/argv.mjs`). */
|
|
76
|
+
export const ARGV_SPEC = {
|
|
77
|
+
usage: `harness.mjs init run-args --slug <slug> --auto-level ${autoLevels().join("|")} ` +
|
|
78
|
+
'--exec-model <name> [--eval-model <name>] [--qa-model <name>] --max-rounds N --attempts N ' +
|
|
79
|
+
'--plugin-root <dir> [--run-id <id>] [--answers <preset|path>] [--lane full|tiny] ' +
|
|
80
|
+
'[--no-eval] [--no-qa] [--adversarial-verify] [--parallel-scopes N] [--cwd <dir>]',
|
|
81
|
+
_: { arity: 0, max: 0, name: "(no positional operands)" },
|
|
82
|
+
cwd: { type: "path" },
|
|
83
|
+
slug: { type: "str", required: true },
|
|
84
|
+
"run-id": { type: "str" },
|
|
85
|
+
"auto-level": { type: "str", required: true },
|
|
86
|
+
answers: { type: "str" },
|
|
87
|
+
lane: { type: "str" },
|
|
88
|
+
"exec-model": { type: "str", required: true },
|
|
89
|
+
"eval-model": { type: "str" },
|
|
90
|
+
"qa-model": { type: "str" },
|
|
91
|
+
"max-rounds": { type: "int", min: 1, required: true },
|
|
92
|
+
attempts: { type: "int", min: 1, required: true },
|
|
93
|
+
"plugin-root": { type: "path", required: true },
|
|
94
|
+
"no-eval": { type: "flag" },
|
|
95
|
+
"no-qa": { type: "flag" },
|
|
96
|
+
"adversarial-verify": { type: "flag" },
|
|
97
|
+
"parallel-scopes": { type: "int", min: 1 },
|
|
98
|
+
};
|
|
99
|
+
|
|
100
|
+
function fail(code, msg) {
|
|
101
|
+
console.error(msg);
|
|
102
|
+
process.exit(code);
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Drop `undefined`-valued keys, one level deep on the named nested objects — so the JSON this
|
|
107
|
+
* writes never carries a literal `"eval": undefined` for a skipped role or an unset switch.
|
|
108
|
+
*
|
|
109
|
+
* @param {object} o - The candidate RunArgs object.
|
|
110
|
+
* @returns {object} The same shape with every `undefined` leaf and empty nested object removed.
|
|
111
|
+
*/
|
|
112
|
+
function pruned(o) {
|
|
113
|
+
const out = {};
|
|
114
|
+
for (const [k, v] of Object.entries(o)) {
|
|
115
|
+
if (v === undefined) continue;
|
|
116
|
+
if (v && typeof v === "object" && !Array.isArray(v)) {
|
|
117
|
+
const inner = pruned(v);
|
|
118
|
+
if (Object.keys(inner).length) out[k] = inner;
|
|
119
|
+
continue;
|
|
120
|
+
}
|
|
121
|
+
out[k] = v;
|
|
122
|
+
}
|
|
123
|
+
return out;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/**
|
|
127
|
+
* Build the `RunArgs` object — pure, so the structural suite can assert its shape without a
|
|
128
|
+
* filesystem. Mirrors `domain.schema.json` `$defs/RunArgs` exactly: this is the one place that
|
|
129
|
+
* shape is constructed, so a field this function does not carry cannot reach the file either.
|
|
130
|
+
*
|
|
131
|
+
* @param {object} o - Resolved inputs (destructured); optional fields may be `undefined`.
|
|
132
|
+
* @returns {object} The RunArgs object, pruned of unset optionals.
|
|
133
|
+
*/
|
|
134
|
+
export function buildRunArgs({
|
|
135
|
+
slug, runId, autoLevel, answers, lane, execModel, evalModel, qaModel,
|
|
136
|
+
maxRounds, attemptBudget, pluginRoot, startedAt,
|
|
137
|
+
noEval, noQa, adversarialVerify, maxParallelScopes,
|
|
138
|
+
}) {
|
|
139
|
+
return pruned({
|
|
140
|
+
slug, runId, autoLevel, answers, lane,
|
|
141
|
+
models: { exec: execModel, eval: evalModel, qa: qaModel },
|
|
142
|
+
budgets: { maxRounds, attemptBudget },
|
|
143
|
+
pluginRoot, startedAt,
|
|
144
|
+
noEval, noQa, adversarialVerify, maxParallelScopes,
|
|
145
|
+
});
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
/**
|
|
149
|
+
* Write `.shapeup/<slug>/run-args.json`, or refuse (exit 3) when no run is open at `--slug`.
|
|
150
|
+
*
|
|
151
|
+
* @param {string[]} rawArgv - The subcommand's own arguments (harness.mjs strips the verb words).
|
|
152
|
+
* @returns {(Promise<void>|void)} Settles when the subcommand has written its output; every path
|
|
153
|
+
* calls `process.exit()` with the subcommand's documented code rather than returning.
|
|
154
|
+
*/
|
|
155
|
+
export function cli(rawArgv) {
|
|
156
|
+
const args = runArgs(ARGV_SPEC, rawArgv);
|
|
157
|
+
const cwd = args.cwd || process.cwd();
|
|
158
|
+
|
|
159
|
+
if (!autoLevels().includes(args.autoLevel)) {
|
|
160
|
+
fail(2, `--auto-level must be one of: ${autoLevels().join(", ")}`);
|
|
161
|
+
}
|
|
162
|
+
if (!args.noEval && !args.evalModel) {
|
|
163
|
+
fail(2, "--eval-model is required unless --no-eval is set — an EVAL-skipping run never resolves one.");
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
const runRoot = localRoot(cwd, args.slug);
|
|
167
|
+
const receiptPath = join(runRoot, RECEIPT_FILE);
|
|
168
|
+
if (!existsSync(receiptPath)) {
|
|
169
|
+
fail(3, [
|
|
170
|
+
`✋ init run-args: no open run at ${runRoot} — run "harness init run" first (GATE L0.1).`,
|
|
171
|
+
"",
|
|
172
|
+
"`run-args.json` records what an OPEN run was launched with; writing one with no receipt",
|
|
173
|
+
"behind it would leave a launch record for a run that, everywhere else in the harness, never",
|
|
174
|
+
"started.",
|
|
175
|
+
].join("\n"));
|
|
176
|
+
}
|
|
177
|
+
let receipt = null;
|
|
178
|
+
try { receipt = JSON.parse(readFileSync(receiptPath, "utf8")); } catch { /* handled below */ }
|
|
179
|
+
const runId = args.runId ?? runIdFromReceipt(receipt) ?? undefined;
|
|
180
|
+
|
|
181
|
+
const runArgsObj = buildRunArgs({
|
|
182
|
+
slug: args.slug,
|
|
183
|
+
runId,
|
|
184
|
+
autoLevel: args.autoLevel,
|
|
185
|
+
answers: args.answers ?? undefined,
|
|
186
|
+
lane: args.lane ?? undefined,
|
|
187
|
+
execModel: args.execModel,
|
|
188
|
+
evalModel: args.evalModel ?? undefined,
|
|
189
|
+
qaModel: args.qaModel ?? undefined,
|
|
190
|
+
maxRounds: args.maxRounds,
|
|
191
|
+
attemptBudget: args.attempts,
|
|
192
|
+
pluginRoot: args.pluginRoot,
|
|
193
|
+
startedAt: new Date().toISOString(),
|
|
194
|
+
noEval: args.noEval || undefined,
|
|
195
|
+
noQa: args.noQa || undefined,
|
|
196
|
+
adversarialVerify: args.adversarialVerify || undefined,
|
|
197
|
+
maxParallelScopes: args.parallelScopes ?? undefined,
|
|
198
|
+
});
|
|
199
|
+
|
|
200
|
+
mkdirSync(runRoot, { recursive: true });
|
|
201
|
+
writeFileSync(runArgsPath(cwd, args.slug), JSON.stringify(runArgsObj, null, 2) + "\n", "utf8");
|
|
202
|
+
|
|
203
|
+
// Printed, not just written — so the caller's `Workflow({args: ...})` reads this stdout instead
|
|
204
|
+
// of re-assembling the object from the flags it just typed.
|
|
205
|
+
console.log(JSON.stringify(runArgsObj, null, 2));
|
|
206
|
+
}
|
package/kernel/init/run.mjs
CHANGED
|
@@ -224,6 +224,16 @@ export function runFrontmatter({ slug, config, startedAt }) {
|
|
|
224
224
|
"deploy: ~",
|
|
225
225
|
`started_at: ${startedAt}`,
|
|
226
226
|
"closed_at: ~",
|
|
227
|
+
// The cause a terminal status ended on, written alongside `closed_at` by
|
|
228
|
+
// `probe resume --close` (kernel/probe/resume.mjs's closeRun) — the two land in one write, so a
|
|
229
|
+
// closed run's ledger never carries a timestamp with no reason beside it.
|
|
230
|
+
"close_cause: ~",
|
|
231
|
+
// The once-only guard's OWN record of what closed this run — deliberately separate from
|
|
232
|
+
// `status:` above, which every phase rewrites (`setRunStatus`) for the life of the run,
|
|
233
|
+
// including the product's own ship path immediately before `probe resume --close` runs. Only
|
|
234
|
+
// `closeRun` ever writes this line, so it is the one field an intervening `status:` rewrite
|
|
235
|
+
// cannot move (kernel/probe/resume.mjs's closeRun docblock has the measured scenario).
|
|
236
|
+
"closed_status: ~",
|
|
227
237
|
"---",
|
|
228
238
|
"",
|
|
229
239
|
`# Harness run — ${slug}`,
|
package/kernel/lib/paths.mjs
CHANGED
|
@@ -140,6 +140,16 @@ export const receipt = (cwd, slug) => join(localRoot(cwd, slug), RECEIPT_FILE);
|
|
|
140
140
|
export const intake = (cwd, slug) => join(localRoot(cwd, slug), "intake.md");
|
|
141
141
|
/** The breadboard the pitch was shaped with, verbatim, next to its digest in the receipt. */
|
|
142
142
|
export const breadboard = (cwd, slug) => join(localRoot(cwd, slug), "breadboard.md");
|
|
143
|
+
/**
|
|
144
|
+
* The launch record's filename, as a constant rather than a literal at each call site — the
|
|
145
|
+
* writer (`harness init run-args`) and every reader (`probe concurrency`'s `dialFrom()`) resolve
|
|
146
|
+
* the same path through this constant, the same way {@link runIdFromRoot} resolves `RECEIPT_FILE`
|
|
147
|
+
* against a bare run root rather than through {@link receipt}'s `(cwd, slug)` form: a caller that
|
|
148
|
+
* already holds the run root (an archived trace, `--run-root <dir>`) has no slug to reconstruct.
|
|
149
|
+
*/
|
|
150
|
+
export const RUN_ARGS_FILE = "run-args.json";
|
|
151
|
+
/** The launch record — the RunArgs object a run was launched with (GATE L0.9b), kernel-written. */
|
|
152
|
+
export const runArgsPath = (cwd, slug) => join(localRoot(cwd, slug), RUN_ARGS_FILE);
|
|
143
153
|
/** The run ledger — rounds, decisions, status frontmatter. */
|
|
144
154
|
export const harnessRun = (cwd, slug) => join(localRoot(cwd, slug), "harness-run.md");
|
|
145
155
|
/** File-derived mid-run digest, frozen by `reduce snapshot --write` as an audit anchor. */
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
// probe attempts — "how many of this scope's attempts are ATTESTED work, not writable artifacts?"
|
|
2
|
+
//
|
|
3
|
+
// CONTRACT. A bounded, read-only query over one scope's attested channels for one round: dispatch
|
|
4
|
+
// receipts (`receipts/dispatch.jsonl`), leg-completion rows (`legs.jsonl`) and WorkResults
|
|
5
|
+
// (`results/`). Prints `{scope_id, round, attempt_budget, spent, in_flight, unattested, green,
|
|
6
|
+
// tripped, attempts}` on stdout; exits 0 when the breaker holds, 1 when it has tripped, 2 on a bad
|
|
7
|
+
// argv. Writes nothing.
|
|
8
|
+
//
|
|
9
|
+
// WHY IT EXISTS. Measured on a real run: attempt 2 of a scope was compiled and T0-verified
|
|
10
|
+
// several minutes BEFORE attempt 1 ingested — while attempt 1 was still in flight. No worker was
|
|
11
|
+
// ever dispatched for attempt 2:
|
|
12
|
+
// `receipts/dispatch.jsonl`, `legs.jsonl` and `results/` carried no row for it. A derivation keyed
|
|
13
|
+
// off the order set (`orders/`) or the T0 verdict set (`t0/verdicts/`) alone counts a compiled
|
|
14
|
+
// order or a green trial as a spent attempt regardless of whether a worker ever ran — both are
|
|
15
|
+
// WRITABLE by the very leg whose exhaustion is being judged. This module counts an attempt as
|
|
16
|
+
// SPENT only when a dispatch receipt attests it started AND either a leg-completion row or a
|
|
17
|
+
// WorkResult on disk attests it closed. Neither channel alone is enough: a receipt with no result
|
|
18
|
+
// is a leg still in flight (open, not spent — the breaker must not trip on unanswered work), and a
|
|
19
|
+
// result or verdict with no receipt is unattested — no worker ran, which is the shape that
|
|
20
|
+
// produced this module.
|
|
21
|
+
//
|
|
22
|
+
// WHY THE SAME FUNCTION SERVES THE BREAKER AND THE CENSUS. Before this module, the round loop's
|
|
23
|
+
// inner breaker and `scope-hammer`'s GATE H0 census read different evidence for the same question
|
|
24
|
+
// — the loop trusted the worker's own self-reported `attempts_used`/`breaker` (schema-shaped, not
|
|
25
|
+
// re-verified), the census read `t0/verdicts/*.json` directly — and disagreed on the measured run.
|
|
26
|
+
// One shared, attested derivation is what makes "the census and the breaker agree" true by
|
|
27
|
+
// construction rather than by coincidence: `skills/scope-hammer/SKILL.md` cites this same probe
|
|
28
|
+
// the way it already cites `probe owner` for ownership claims.
|
|
29
|
+
|
|
30
|
+
import { existsSync, readdirSync, readFileSync } from "node:fs";
|
|
31
|
+
import { join, resolve } from "node:path";
|
|
32
|
+
import { runArgs } from "../lib/argv.mjs";
|
|
33
|
+
import { dispatchReceipts, legLedger, resultsDir } from "../lib/paths.mjs";
|
|
34
|
+
import { readLegs } from "./leg.mjs";
|
|
35
|
+
import { greenVerdict } from "./t0.mjs";
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Every dispatch-receipt row on disk, tolerant of a torn last line (mirrors {@link readLegs}).
|
|
39
|
+
* @param {string} path - `receipts/dispatch.jsonl`.
|
|
40
|
+
* @returns {object[]} Parsed rows; an unparsable line is skipped rather than fatal.
|
|
41
|
+
*/
|
|
42
|
+
export function readReceipts(path) {
|
|
43
|
+
if (!existsSync(path)) return [];
|
|
44
|
+
return readFileSync(path, "utf8").split("\n").filter(Boolean)
|
|
45
|
+
.map((l) => { try { return JSON.parse(l); } catch { return null; } })
|
|
46
|
+
.filter(Boolean);
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* One attempt's evidence across the three attested channels, and the state it derives to.
|
|
51
|
+
*
|
|
52
|
+
* `unattested` (no receipt at all) counts as though the attempt never happened — the order and any
|
|
53
|
+
* T0 verdict may still be sitting on disk, written by something other than a dispatched worker, and
|
|
54
|
+
* neither is asked here. `in-flight` (a receipt, but no leg row and no result) is a real dispatch
|
|
55
|
+
* whose leg has not yet closed: open, not spent. `spent` is closed either way a leg closes — via
|
|
56
|
+
* `reduce ingest`'s own leg-completion row, or a WorkResult already on disk pending ingest.
|
|
57
|
+
*
|
|
58
|
+
* @param {string} cwd - Project root.
|
|
59
|
+
* @param {string} slug - Feature slug.
|
|
60
|
+
* @param {string} scopeId - Scope contract id (the order id's own address, e.g. `shell-r1-a2`).
|
|
61
|
+
* @param {number} round - Build round.
|
|
62
|
+
* @param {number} attempt - Attempt number within the round.
|
|
63
|
+
* @param {object[]} receipts - Pre-read `receipts/dispatch.jsonl` rows (avoids re-reading per attempt).
|
|
64
|
+
* @param {object[]} legs - Pre-read `legs.jsonl` rows.
|
|
65
|
+
* @returns {{orderId:string, hasReceipt:boolean, hasResult:boolean, hasLeg:boolean, state:("unattested"|"in-flight"|"spent")}}
|
|
66
|
+
*/
|
|
67
|
+
export function attemptEvidence(cwd, slug, scopeId, round, attempt, receipts, legs) {
|
|
68
|
+
const orderId = `${slug}/${scopeId}-r${round}-a${attempt}`;
|
|
69
|
+
const hasReceipt = receipts.some((r) => r?.order_id === orderId);
|
|
70
|
+
const hasLeg = legs.some((r) => r?.order_id === orderId);
|
|
71
|
+
const hasResult = existsSync(join(resultsDir(cwd, slug), `${scopeId}-r${round}-a${attempt}.json`));
|
|
72
|
+
const state = !hasReceipt ? "unattested" : (hasResult || hasLeg) ? "spent" : "in-flight";
|
|
73
|
+
return { orderId, hasReceipt, hasResult, hasLeg, state };
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/**
|
|
77
|
+
* A scope's attempt census for one round, derived ONLY from attested channels — the function both
|
|
78
|
+
* the round loop's inner breaker and scope-hammer's GATE H0 census call, so they cannot drift apart
|
|
79
|
+
* again the way a measured run found them.
|
|
80
|
+
*
|
|
81
|
+
* @param {string} cwd - Project root.
|
|
82
|
+
* @param {string} slug - Feature slug.
|
|
83
|
+
* @param {string} scopeId - Scope contract id.
|
|
84
|
+
* @param {number} round - Build round.
|
|
85
|
+
* @param {number} attemptBudget - The scope's configured `attempt_budget`.
|
|
86
|
+
* @returns {{scope_id:string, round:number, attempt_budget:number, spent:number, in_flight:number,
|
|
87
|
+
* unattested:number, green:boolean, tripped:boolean, attempts:object[]}} `tripped` is true only
|
|
88
|
+
* when every attempt within budget is genuinely SPENT and none produced a green T0 — an attempt
|
|
89
|
+
* still in flight holds the breaker open regardless of how many slots are nominally used.
|
|
90
|
+
*/
|
|
91
|
+
export function scopeAttempts(cwd, slug, scopeId, round, attemptBudget) {
|
|
92
|
+
const receipts = readReceipts(dispatchReceipts(cwd, slug));
|
|
93
|
+
const legs = readLegs(legLedger(cwd, slug));
|
|
94
|
+
const attempts = [];
|
|
95
|
+
let spent = 0, inFlight = 0, unattested = 0;
|
|
96
|
+
for (let a = 1; a <= attemptBudget; a++) {
|
|
97
|
+
const ev = attemptEvidence(cwd, slug, scopeId, round, a, receipts, legs);
|
|
98
|
+
if (ev.state === "spent") spent++;
|
|
99
|
+
else if (ev.state === "in-flight") inFlight++;
|
|
100
|
+
else unattested++;
|
|
101
|
+
attempts.push({ attempt: a, ...ev });
|
|
102
|
+
}
|
|
103
|
+
const { green } = greenVerdict(cwd, slug, scopeId, round);
|
|
104
|
+
return {
|
|
105
|
+
scope_id: scopeId, round, attempt_budget: attemptBudget,
|
|
106
|
+
spent, in_flight: inFlight, unattested, green,
|
|
107
|
+
tripped: !green && spent >= attemptBudget,
|
|
108
|
+
attempts,
|
|
109
|
+
};
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
export const ARGV_SPEC = {
|
|
113
|
+
usage: "harness.mjs probe attempts --slug <slug> --scope <scope-id> --round N --attempt-budget N [--cwd <dir>]",
|
|
114
|
+
_: { arity: 0, max: 0, name: "(no positional operands)" },
|
|
115
|
+
slug: { type: "str", required: true },
|
|
116
|
+
scope: { type: "str", required: true },
|
|
117
|
+
round: { type: "int", min: 1, required: true },
|
|
118
|
+
"attempt-budget": { type: "int", min: 1, required: true },
|
|
119
|
+
cwd: { type: "path" },
|
|
120
|
+
};
|
|
121
|
+
|
|
122
|
+
/**
|
|
123
|
+
* Report one scope's attested attempt census for one round.
|
|
124
|
+
*
|
|
125
|
+
* @param {string[]} rawArgv - The subcommand's own arguments (harness.mjs strips the verb words).
|
|
126
|
+
* @returns {void} Exits 0 when the breaker holds, 1 when it has tripped — the shape a caller (or a
|
|
127
|
+
* census citing this row) can branch on without parsing prose.
|
|
128
|
+
*/
|
|
129
|
+
export function cli(rawArgv) {
|
|
130
|
+
const args = runArgs(ARGV_SPEC, rawArgv);
|
|
131
|
+
const cwd = resolve(args.cwd || process.cwd());
|
|
132
|
+
const r = scopeAttempts(cwd, args.slug, args.scope, args.round, args.attemptBudget);
|
|
133
|
+
console.log(JSON.stringify(r));
|
|
134
|
+
process.exit(r.tripped ? 1 : 0);
|
|
135
|
+
}
|