shapeup-sdlc 3.5.0 → 3.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -47,10 +47,11 @@ import { runArgs } from "../lib/argv.mjs";
47
47
  import { splitFrontmatter } from "../lib/contract.mjs";
48
48
  import { runIdFromReceipt, readReceipt } from "../lib/paths.mjs";
49
49
  import { TABLES, runRow, dispatchFacts } from "./facts.mjs";
50
+ import { deriveRounds } from "../probe/rounds.mjs";
50
51
  import {
51
52
  localDir, activeScope, receipt as receiptPath, harnessRun, ordersDir, resultsDir,
52
53
  trials as trialsPath, verdictsDir, evaluationDir, decisions as decisionsPath,
53
- exportsDir, exportRunDir,
54
+ gates as gatesPath, roundBuildDir, exportsDir, exportRunDir,
54
55
  } from "../lib/paths.mjs";
55
56
 
56
57
  export const EXPORT_SCHEMA_VERSION = 1;
@@ -164,6 +165,51 @@ function criterionRows(dir, runId, t) {
164
165
  return out;
165
166
  }
166
167
 
168
+ /**
169
+ * Flatten one gate-crossing ledger row (`gates.jsonl`, `kernel/gate.mjs`'s sole writer) into a
170
+ * flat `gate_decision` fact row. The row on disk already carries exactly these fields
171
+ * (see `appendGateLedger`), so this is a pass-through with a stamped `run_id` fallback rather than
172
+ * a re-derivation: two readers of "what did this gate decide" must not compute the answer twice.
173
+ * @param {object} g - One parsed line of `gates.jsonl`.
174
+ * @param {(string|null)} runId - Run key for a row written before it carried its own.
175
+ * @returns {object} A flat `gate_decision` row.
176
+ */
177
+ function gateDecisionRow(g, runId) {
178
+ return {
179
+ run_id: g?.run_id ?? runId ?? null,
180
+ gate: g?.gate ?? null,
181
+ decision: g?.decision ?? null,
182
+ status: g?.status ?? null,
183
+ source: g?.source ?? null,
184
+ round: g?.round ?? null,
185
+ has_note: !!(g?.note && String(g.note).trim()),
186
+ };
187
+ }
188
+
189
+ /**
190
+ * Flatten one round build-gate artifact (`kernel/verify/build.mjs`'s `writeRoundBuild`) into a flat
191
+ * `build_gate` fact row. The gate ends a round exactly as EVAL does (AGENTS.md's round
192
+ * build gate ⚙), and until now had no fact table at all.
193
+ * @param {object} a - A parsed round-build artifact.
194
+ * @param {(string|null)} runId - Run key for an artifact written before it carried its own.
195
+ * @returns {object} A flat `build_gate` row.
196
+ */
197
+ function buildGateRow(a, runId) {
198
+ const steps = Array.isArray(a?.steps) ? a.steps : [];
199
+ return {
200
+ run_id: a?.run_id ?? runId ?? null,
201
+ round: a?.round ?? null,
202
+ trial: a?.trial ?? null,
203
+ at: a?.at ?? null,
204
+ overall: a?.overall ?? null,
205
+ archetype: a?.archetype ?? null,
206
+ steps_total: steps.length,
207
+ steps_failed: steps.filter((s) => !s?.skipped && s?.pass === false).length,
208
+ warnings: Array.isArray(a?.warnings) ? a.warnings.length : 0,
209
+ discovered_tasks: Array.isArray(a?.discovered_tasks) ? a.discovered_tasks.length : 0,
210
+ };
211
+ }
212
+
167
213
  // ---------------------------------------------------------------------------
168
214
  // The export itself
169
215
  // ---------------------------------------------------------------------------
@@ -190,7 +236,10 @@ export function collectRun(cwd, slug) {
190
236
  const results = readJsonDir(resultsDir(cwd, slug), t);
191
237
 
192
238
  const { dispatch, ac_result, discovery, file_touched } = dispatchFacts({ orders, results, runId });
193
- const run = runRow({ receipt: rec, ledger, runId });
239
+ // Computed here, once, from the same trace this whole function reads, and handed to
240
+ // runRow rather than re-derived by it: runRow stays pure (no I/O), this function already has cwd.
241
+ const rounds = deriveRounds(cwd, slug, ledger.rounds_used);
242
+ const run = runRow({ receipt: rec, ledger, runId, rounds });
194
243
 
195
244
  // Hook decisions are checkout-wide, so they are FILTERED to this run rather than read from a
196
245
  // per-run file. Rows with a null key belong to no run (a hook that fired outside one) and are
@@ -208,6 +257,9 @@ export function collectRun(cwd, slug) {
208
257
  t0_verdict: readJsonDir(verdictsDir(cwd, slug), t).map((a) => t0Row(a, runId)),
209
258
  criterion_verdict: criterionRows(evaluationDir(cwd, slug), runId, t),
210
259
  hook_decision,
260
+ // The decision that crossed each gate, and the round build gate's own artifact.
261
+ gate_decision: readJsonl(gatesPath(cwd, slug), t).map((g) => gateDecisionRow(g, runId)),
262
+ build_gate: readJsonDir(roundBuildDir(cwd, slug), t).map((a) => buildGateRow(a, runId)),
211
263
  },
212
264
  defects: { records_skipped: t.skipped },
213
265
  };
@@ -27,6 +27,11 @@
27
27
  export const TABLES = [
28
28
  "run", "dispatch", "ac_result", "discovery", "file_touched",
29
29
  "trial", "t0_verdict", "criterion_verdict", "hook_decision",
30
+ // The decision that shipped a run (or any other gate) had a ledger row (`gates.jsonl`)
31
+ // and no table — a reader had to open the LOCAL trace itself, which the export exists so nobody
32
+ // has to. `build_gate` is the round build gate's own artifact (kernel/verify/build.mjs), on the
33
+ // same terms: it ends a round exactly as EVAL does, and had no table either.
34
+ "gate_decision", "build_gate",
30
35
  ];
31
36
 
32
37
  /** Coerce anything to a finite number, or null. Keeps `0` and rejects `NaN`/`""`/undefined. */
@@ -76,9 +81,14 @@ export function parseOrderStem(orderId) {
76
81
  * @param {(object|null)} o.receipt - Parsed `receipt.json`.
77
82
  * @param {(object|null)} [o.ledger] - Parsed `harness-run.md` frontmatter (a flat scalar map).
78
83
  * @param {(string|null)} [o.runId] - The run key, when already resolved.
84
+ * @param {({rounds_used:*, rounds_judged:(number|null)}|null)} [o.rounds] - The two-number
85
+ * derivation (`probe/rounds.mjs`'s `deriveRounds`), computed by the caller because it needs the
86
+ * filesystem and this function stays pure. Falls back to the ledger's own (unreliable — see
87
+ * `deriveRounds`) `rounds_used` line when the caller has not derived one, so an existing caller
88
+ * is unaffected rather than broken.
79
89
  * @returns {(object|null)} The run row, or null when there is no receipt to describe.
80
90
  */
81
- export function runRow({ receipt, ledger = null, runId = null }) {
91
+ export function runRow({ receipt, ledger = null, runId = null, rounds = null }) {
82
92
  if (!receipt) return null;
83
93
  const c = receipt.config || {};
84
94
  const fm = ledger || {};
@@ -87,6 +97,9 @@ export function runRow({ receipt, ledger = null, runId = null }) {
87
97
  slug: receipt.slug ?? null,
88
98
  started_at: receipt.started_at ?? null,
89
99
  closed_at: fm.closed_at && fm.closed_at !== "~" ? fm.closed_at : null,
100
+ // Why the run ended at a terminal status, written by `probe resume --close` alongside
101
+ // `closed_at` — the two facts a trace needs to tell a live run from a dead one apart.
102
+ close_cause: fm.close_cause && fm.close_cause !== "~" ? fm.close_cause : null,
90
103
  intake_sha256: receipt.intake_sha256 ?? null,
91
104
  intake_chars: num(receipt.intake_chars),
92
105
  intake_lines: num(receipt.intake_lines),
@@ -101,8 +114,17 @@ export function runRow({ receipt, ledger = null, runId = null }) {
101
114
  // Copied from the ledger, never re-derived: the run's own status line is the harness's answer,
102
115
  // and a read plane that recomputed it would be asserting a second one.
103
116
  status: fm.status ?? null,
117
+ // The terminal status `closeRun` alone writes, carried BESIDE `status` rather than instead of
118
+ // it. `status:` is ordinary phase traffic and a later phase may move it, so an export that
119
+ // carried only that line could show a run wearing another close's cause. These two disagreeing
120
+ // is itself the fact worth exporting: it says the ledger was written after the close.
121
+ closed_status: fm.closed_status && fm.closed_status !== "~" ? fm.closed_status : null,
104
122
  final_verdict: fm.final_verdict && fm.final_verdict !== "~" ? fm.final_verdict : null,
105
- rounds_used: num(Number(fm.rounds_used)),
123
+ // Two fields, not one: `rounds_used` is the highest round carrying ANY build evidence,
124
+ // `rounds_judged` the highest round EVAL actually returned a verdict for. A caller that has not
125
+ // derived `rounds` falls back to the ledger's own (pre-fix, unreliable) line, non-regression.
126
+ rounds_used: rounds ? num(Number(rounds.rounds_used)) : num(Number(fm.rounds_used)),
127
+ rounds_judged: rounds ? num(rounds.rounds_judged) : null,
106
128
  };
107
129
  }
108
130
 
@@ -668,14 +668,14 @@
668
668
  "string",
669
669
  "null"
670
670
  ],
671
- "description": "Source file of the failure; null/absent when the log line carried no location (raw triple — never invented)."
671
+ "description": "Source file of the failure; null/absent when the log line named no file at all (raw triple — never invented)."
672
672
  },
673
673
  "line": {
674
674
  "type": [
675
675
  "integer",
676
676
  "null"
677
677
  ],
678
- "description": "Line of the failure; null/absent when the log line carried no location. Paired with `file` — a triple has both or neither, and neither is ever invented to satisfy a shape."
678
+ "description": "Line of the failure; null/absent when the diagnostic carried no line number — independently of `file`, since a diagnostic can name a file with no line (a resource-compiler error, for example). Never invented to satisfy a shape."
679
679
  },
680
680
  "core_message": {
681
681
  "type": "string",
@@ -1891,9 +1891,10 @@
1891
1891
  "building",
1892
1892
  "evaluating",
1893
1893
  "shipped",
1894
- "escalated"
1894
+ "escalated",
1895
+ "aborted"
1895
1896
  ],
1896
- "description": "Mirrors harness-run.md frontmatter status."
1897
+ "description": "Mirrors harness-run.md frontmatter status — kernel/probe/resume.mjs's RUN_STATUSES is the source enum this one must not drift from."
1897
1898
  },
1898
1899
  "round": {
1899
1900
  "type": "integer",
@@ -1904,7 +1905,12 @@
1904
1905
  "description": "From the latest t0/verdicts/r<N>-a<M>.json filename."
1905
1906
  },
1906
1907
  "rounds_used": {
1907
- "type": "integer"
1908
+ "type": "integer",
1909
+ "description": "Highest round carrying any build evidence (an order, a T0 verdict, a round build-gate artifact, or an EVAL result) — kernel/probe/rounds.mjs's deriveRounds(), the same derivation the ship report and the export use. Falls back to harness-run.md's literal frontmatter value only when no such evidence exists on disk."
1910
+ },
1911
+ "rounds_judged": {
1912
+ "type": "integer",
1913
+ "description": "Highest round EVAL actually returned a verdict for (an evaluate-r<N>.json result on disk) — its own field, never folded into rounds_used: a round built is not a round judged. Omitted when no round has been judged yet."
1908
1914
  },
1909
1915
  "max_rounds": {
1910
1916
  "type": "integer"
@@ -2704,10 +2710,10 @@
2704
2710
  }
2705
2711
  },
2706
2712
  "RunArgs": {
2707
- "description": "C1 — the launch half of the workflow's only conversation. Compiled ONCE by tech-lead at GATE L0 from harness init run output + the L0.8 model matrix + budgets, written to .shapeup/<slug>/run-args.json and handed to the harness run launch as one JSON literal — the workflow cannot ask follow-ups and cannot read config files itself, so everything a run will ever need travels in this one record. A workflow script validates its own subset of this shape in code (no runtime schema check at the C1 boundary itself); this entry is the central-registry definition the workflow script and the tech-lead skill both read as the one true shape.",
2713
+ "description": "C1 — the launch half of the workflow's only conversation. Resolved ONCE at GATE L0 from harness init run output + the L0.8 model matrix + budgets, then built by `harness init run-args`, which writes it to .shapeup/<slug>/run-args.json AND prints it — tech-lead passes that printed value to the harness run launch as one JSON literal, never a second, hand-typed copy of it. The workflow cannot ask follow-ups and cannot read config files itself, so everything a run will ever need travels in this one record. A workflow script validates its own subset of this shape in code (no runtime schema check at the C1 boundary itself); this entry is the central-registry definition the workflow script, `harness init run-args` and the tech-lead skill all read as the one true shape.",
2708
2714
  "x-tier": "EMBEDDED",
2709
- "x-location": ".shapeup/<slug>/run-args.json — written fresh by tech-lead on every launch and relaunch; the workflow receives it as its args and never reads other config",
2710
- "x-writer": "tech-lead (GATE L0, on every launch AND every relaunch after a paused gate)",
2715
+ "x-location": ".shapeup/<slug>/run-args.json — written fresh on every launch and relaunch by `harness init run-args`; the workflow receives it as its args and never reads other config",
2716
+ "x-writer": "harness init run-args (kernel), invoked by tech-lead at GATE L0 on every launch AND every relaunch after a paused gate",
2711
2717
  "x-readers": "the Workflow runtime (shapeup-run, and shapeup-run's own inner round dispatch)",
2712
2718
  "x-not-here": "Run config the LEDGER already carries does NOT get a second home in RunArgs — eval_dimensions, lens, spec_folder, stack, run_cmd, app_url are read off harness-run.md frontmatter by resume-state on every launch AND every relaunch, so a copy here would be a second source that can disagree with the first. RunArgs carries what a workflow cannot derive from disk (identity, budgets, the model matrix, pluginRoot, startedAt) plus noEval, which no frontmatter line holds.",
2713
2719
  "type": "object",
@@ -2759,16 +2765,13 @@
2759
2765
  },
2760
2766
  "budgets": {
2761
2767
  "type": "object",
2762
- "description": "The three-level circuit breaker's own limits (AGENTS.md) — outer round_budget, inner attempt_budget, and the opt-in wall-clock breaker.",
2768
+ "description": "The two RunArgs-level circuit breakers (AGENTS.md) — outer round_budget, inner attempt_budget. The third, opt-in DEADLINE breaker (the wall-clock budget) is NOT a RunArgs field: it is typed once, at `harness init run --wall-clock-budget`, and lands in the run receipt's `wall_clock_budget_s` — `harness verify budget` reads that receipt field directly and never sees this launch's RunArgs at all, so it has no member here to declare.",
2763
2769
  "properties": {
2764
2770
  "maxRounds": {
2765
2771
  "type": "integer"
2766
2772
  },
2767
2773
  "attemptBudget": {
2768
2774
  "type": "integer"
2769
- },
2770
- "wallClockS": {
2771
- "type": "integer"
2772
2775
  }
2773
2776
  }
2774
2777
  },
@@ -12,7 +12,7 @@
12
12
  // #/$defs/Name — a definition in the SAME schema document
13
13
  // domain.schema.json#/$defs/Name — a definition in a SIBLING file (the central domain
14
14
  // registry; resolved against the schema's own dir,
15
- // falling back to skills/tech-lead/schemas/)
15
+ // falling back to kernel/schemas/)
16
16
  //
17
17
  // Usage (CLI): node kernel/harness.mjs verify envelope <envelope.json> <schema.json>
18
18
  // exit 0 = valid, 1 = invalid (errors printed one per line)
@@ -28,7 +28,7 @@ import { runArgs } from "../lib/argv.mjs";
28
28
  import { runHook, readStdin, settle } from "../../hooks/lib/decision.mjs";
29
29
 
30
30
  const HERE = dirname(fileURLToPath(import.meta.url));
31
- export const SCHEMAS_DIR = resolve(HERE, "../../skills/tech-lead/schemas");
31
+ export const SCHEMAS_DIR = resolve(HERE, "../schemas");
32
32
 
33
33
  /**
34
34
  * Validate a value against the JSON-Schema subset the envelope schemas use (type, required,
@@ -45,7 +45,7 @@ export const PLUGIN_ROOT = resolve(HERE, "../..");
45
45
  * its own domain registry has a broken installation, which is the very thing being checked.
46
46
  */
47
47
  export function roster(root = PLUGIN_ROOT) {
48
- const schemaPath = join(root, "skills/tech-lead/schemas/domain.schema.json");
48
+ const schemaPath = join(root, "kernel/schemas/domain.schema.json");
49
49
  const schema = JSON.parse(readFileSync(schemaPath, "utf8"));
50
50
  const names = schema?.$defs?.WorkerName?.enum;
51
51
  if (!Array.isArray(names) || !names.length) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "shapeup-sdlc",
3
- "version": "3.5.0",
3
+ "version": "3.6.0",
4
4
  "description": "Shape Up for coding agents \u2014 with gates the agent can't talk its way past. Harness for Claude Code.",
5
5
  "bin": {
6
6
  "shapeup-sdlc": "bin/init.mjs"
@@ -93,7 +93,7 @@ never lands in any worker's KB.
93
93
  Orchestrated, this skill is dispatched like every worker: a **WorkOrder** in (`--order <path>`,
94
94
  operation `coach` or `scan`), a **WorkResult** out. Standalone, the raw feedback is passed
95
95
  directly; it maps onto the one payload field registered for this worker in the central domain
96
- registry (`skills/tech-lead/schemas/domain.schema.json`, `x-payload-by-worker`):
96
+ registry (`kernel/schemas/domain.schema.json`, `x-payload-by-worker`):
97
97
 
98
98
  | Payload field | Standalone form | Meaning |
99
99
  |---|---|---|
@@ -117,7 +117,13 @@ fields nobody used" → "Prefer the minimum DTO that satisfies the AC; don't add
117
117
  fields"). Keep the originating why — a rule without its reason gets ignored or misapplied.
118
118
 
119
119
  ### Step 2 — ⏸ GATE COACH-1: Categorize (ASK, never assume)
120
- This is the load-bearing gate. **Do not infer which skill a rule belongs to** — a
120
+ This is the load-bearing gate. **Resolve it first** — `node
121
+ "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" gate --resolve COACH-1 --slug <slug>
122
+ [--file <path>|--preset <name>]` — so the ledger carries a row for the decision this gate makes,
123
+ same as every other gate in the run. Exit 0 (`decision=skip`) — an unattended lane with no live PO;
124
+ record nothing and stop here, the same outcome the CI preset's own note already documents. Exit 4
125
+ (`ask`) — proceed with the categorization below, which IS the PO conversation this decision opens.
126
+ **Do not infer which skill a rule belongs to** — a
121
127
  miscategorized rule lands in a file the wrong worker reads (or no worker reads). Present every
122
128
  candidate rule and ask the PO to assign each one. Emit this block, then stop and wait:
123
129
 
@@ -10,7 +10,7 @@ re-runs a computation that would erase true history.**
10
10
 
11
11
  You are not a worker: no WorkOrder, no WorkResult, invoked directly by the user (or by `/hill`)
12
12
  exactly like `shapeup` is. There is nothing to declare in
13
- `skills/tech-lead/schemas/domain.schema.json` and nothing to teach `harness compile` or
13
+ `kernel/schemas/domain.schema.json` and nothing to teach `harness compile` or
14
14
  `harness reduce ingest` — those steps exist only for dispatched workers.
15
15
 
16
16
  ## What you read
@@ -71,9 +71,8 @@ Build one `{ scope_id, phase }` object per file.
71
71
  ## Rendering — the injection contract
72
72
 
73
73
  The engine ships at `assets/dashboard.template.html` — a complete, self-contained HTML page
74
- (inline CSS/JS, no external fetch beyond Google Fonts, no build step — the same convention as
75
- this repo's own `docs/visualize/*.html`). Do not rewrite it from a text description; read it,
76
- fill in real data, and write the result.
74
+ (inline CSS/JS, no external fetch beyond Google Fonts, no build step). Do not rewrite it from a
75
+ text description; read it, fill in real data, and write the result.
77
76
 
78
77
  1. For each discovered slug, build one entry:
79
78
 
@@ -150,7 +150,7 @@ the harness (this is neither the generator nor the evaluator).
150
150
  Orchestrated, this skill is dispatched like every worker: a **WorkOrder** in (`--order <path>`,
151
151
  operation `hammer`), a **WorkResult** out. The standalone flags below map 1:1 onto the payload
152
152
  fields registered for this worker in the central domain registry
153
- (`skills/tech-lead/schemas/domain.schema.json`, `x-payload-by-worker`):
153
+ (`kernel/schemas/domain.schema.json`, `x-payload-by-worker`):
154
154
 
155
155
  | Payload field | Standalone flag | Meaning |
156
156
  |---|---|---|
@@ -60,15 +60,15 @@ check the lane:
60
60
  legacy loop instead — `references/protocol.md` (BUILD(r)/EVAL) + `references/protocol.md`
61
61
  carry the full step-by-step for both the tiny lane and a scope-less BUILD loop, verbatim, non-
62
62
  regression. Stop reading this file here for that run.
63
- - **Otherwise** (the common case — a scoped spec, any auto level): build `RunArgs`
64
- (`domain.schema.json` `$defs/RunArgs` — `{slug, runId, autoLevel, answers, lane,
65
- models:{exec,eval,qa}, budgets:{maxRounds,attemptBudget,wallClockS}, pluginRoot, startedAt}`,
66
- plus every switch the operator typed — `references/gates.md` GATE L0.9 has the flag→field table,
67
- and a flag that stops here is a flag that was accepted and ignored). **Write that exact object to
68
- `.shapeup/<slug>/run-args.json` before launching**, fresh on every launch and relaunch: the flags
69
- reach the workflow as a value in memory, so it is the run's only evidence of what it was launched
70
- with, and a run that cannot state its own configuration cannot have a claim about it checked.
71
- Then launch with the **`Workflow` tool** — naming `init run`'s staged copy, never the install path:
63
+ - **Otherwise** (the common case — a scoped spec, any auto level): resolve every switch the
64
+ operator typed (`references/gates.md` GATE L0.9b has the flag→field table) and run the kernel's
65
+ sole `RunArgs` writer: `node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" init run-args --slug
66
+ <slug> --auto-level <level> --exec-model <n> [--eval-model <n>] [--qa-model <n>] --max-rounds <N>
67
+ --attempts <N> --plugin-root "${CLAUDE_PLUGIN_ROOT}" [--answers <a>] [--lane <l>] [--no-eval]
68
+ [--no-qa] [--adversarial-verify] [--parallel-scopes <N>]`. It writes `.shapeup/<slug>/run-args.json`
69
+ fresh on every launch/relaunch and prints that identical object — **pass it to `Workflow`
70
+ verbatim, never re-type it**. Then launch with the **`Workflow` tool** — naming `init run`'s
71
+ staged copy, never the install path:
72
72
 
73
73
  ```
74
74
  Workflow({
@@ -114,7 +114,7 @@ did not actually receive from the PO — an unattended lane with no answer for a
114
114
 
115
115
  FIRST freeze the evidence — run state is gitignored, so `shapeup/<slug>/REPORT.md` (already
116
116
  written by `shapeup-run.js` via `harness reduce ship`, or write it now on a `gate_h` close) is all a
117
- teammate sees. Then emit:
117
+ teammate sees. Then RESOLVE the gate — `references/gates.md` GATE L4 has the call — and emit:
118
118
 
119
119
  ```
120
120
  ⏸ GATE L4 — Ship Sign-Off
@@ -113,11 +113,13 @@ Collect (explicit — never inferred):
113
113
  gate to be its first execution.
114
114
  ```
115
115
 
116
- **L0.9b — the launch record.** Every switch the operator typed becomes a `RunArgs` field, or it
117
- does nothing at all: the workflow cannot read a config file and cannot ask a follow-up, so a flag
118
- that stops at the skill boundary was accepted and ignored. That is not hypothetical — `--no-qa` was
119
- documented in seven places across the shipped set and inert in all of them, because no line of this
120
- protocol ever put `noQa` into the record.
116
+ **L0.9b — the launch record.** Every switch the operator typed to *this launch* becomes a `RunArgs`
117
+ field, or it does nothing at all: the workflow cannot read a config file and cannot ask a follow-up,
118
+ so a flag that stops at the skill boundary was accepted and ignored. That is not hypothetical —
119
+ `--no-qa` was documented in seven places across the shipped set and inert in all of them, because no
120
+ line of this protocol ever put `noQa` into the record. `--wall-clock-budget` is the one flag below
121
+ that is not a `RunArgs` field at all — it is consumed earlier, at `init run` itself, and never
122
+ needed to reach this launch; see its row for where it actually lands.
121
123
 
122
124
  | Flag | `RunArgs` field |
123
125
  |---|---|
@@ -125,14 +127,19 @@ protocol ever put `noQa` into the record.
125
127
  | `--no-qa` | `noQa: true` |
126
128
  | `--parallel-scopes N` | `maxParallelScopes: N` — how many scopes build at once (default 4; `1` = sequential) |
127
129
  | `--adversarial-verify` | `adversarialVerify: true` |
128
- | `--rounds N` / `--attempts N` / `--wall-clock-budget S` | `budgets.{maxRounds,attemptBudget,wallClockS}` |
130
+ | `--rounds N` / `--attempts N` | `budgets.{maxRounds,attemptBudget}` |
129
131
  | `--gate-answers <set>` | `answers` |
132
+ | `--wall-clock-budget S` | *(not a `RunArgs` field)* — typed once, on the `harness init run` command line itself, not on this launch; it lands straight in the run receipt as `wall_clock_budget_s`, and the deadline breaker reads that receipt field directly — consumed by `kernel/verify/budget.mjs` as `wall_clock_budget_s`. `budgets` declares only `maxRounds`/`attemptBudget` — the schema, `SKILL.md`'s own RunArgs contract line and this script's own header comment all agree there is no third member |
130
133
  | `--orch-model/--exec-model/--eval-model/--qa-model` | `models.{…}` (L0.8) |
131
134
 
132
- The assembled object is written to `.shapeup/<slug>/run-args.json` before the launch, fresh on every
133
- launch and relaunch. It is the only artifact that records what a run was configured with; the ship
134
- report, a resumed session and any later measurement all read it, and none of them can recover a
135
- value that only ever existed as an argument.
135
+ `harness init run-args` (invoked at Step 2 of `SKILL.md`) is the sole writer of the assembled
136
+ object: it takes the resolved values above, writes `.shapeup/<slug>/run-args.json` fresh on every
137
+ launch and relaunch, and prints the same object back so the launch never re-assembles it by hand. It
138
+ is the only artifact that records what a run was configured with; the ship report, a resumed session
139
+ and any later measurement all read it, and none of them can recover a value that only ever existed
140
+ as an argument. Step 2 is not merely advisory: `shapeup-run.js`'s own Preflight refuses to dispatch
141
+ ORIENT (or anything past it) when this file is missing at the run's local root — a launch that
142
+ skipped this step aborts there rather than proceeding on a silent default.
136
143
 
137
144
  **L0.0 — intake precondition (before any other L0 collection):**
138
145
  ```
@@ -161,7 +168,14 @@ Model matrix : orch=[model] exec=[model] eval=[model] qa=[model] digester=[scrip
161
168
  Budgets : round_budget=[N] (outer) attempt_budget=[N] (inner, per scope)
162
169
  Knowledge : [tech-lead.md — N workflow rules, M suggested values (confirmed above) | none — `/retro --scan` or `/retro --research <stack>` seeds it (optional)]
163
170
  ```
164
- Do NOT start ORIENT until confirmed (interactive/auto). Under --unattended, proceed.
171
+ **Resolve it** — this gate is this skill's own (the workflow never sees it), so it is this skill
172
+ that runs the same tool every other gate resolves through, not a paragraph read as a stand-in for
173
+ one: `node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" gate --resolve L0 --slug <slug>
174
+ [--file <path>|--preset <name>]`. Exit 4 (`ask`) is the confirmation this block already asks for —
175
+ put it to the PO and wait, same as the paragraph above always meant. Exit 0 (`decision=proceed`) —
176
+ continue straight to ORIENT, which is what `--unattended`'s pre-answered set resolves to. Exit 5
177
+ (`abort`) — stop; do not launch. Either way, the gate's own ledger row is what lets a later reader
178
+ see the decision that opened the run, not only the decisions that closed it.
165
179
 
166
180
  ---
167
181
 
@@ -541,5 +555,28 @@ On confirm:
541
555
  - If the PO provides substantive feedback (not just 'y' or empty) → automatically delegate via Agent (model: exec — see references/protocol.md "Invocation mechanism"): Skill(shapeup-sdlc-plugin:coach) with the provided feedback for RLHF. The coach runs its own GATE COACH-1 to have the PO categorize each rule, then files it under the responsible skill in `shapeup/knowledge-base/<skill>.md` (committed → team-shared). Coachable: `task-executor`, `ba-pitch-analyzer`, `qa-edge-hunter`, `orient`, `scope-architect`, `solution-architect` (each reads its own file at the top of its next run) and `tech-lead` (workflow guidance, read at the next GATE L0). Guidance never decides a gate: a filed rule may add a question or a check to a gate block, never an answer. The tech lead does not categorize the feedback itself — that is the coach's gate, by design (no assumptions).
542
556
  - Then output → `✅ [slug] [shipped & deployed | built & verified, deploy pending] — [r] rounds, verdict PASS.`
543
557
 
558
+ **Resolve the gate itself before any of the above** — this is the decision that shipped the run,
559
+ and without it the trace holds no record of that decision at all: `node
560
+ "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" gate --resolve L4 --slug <slug>
561
+ [--file <path>|--preset <name>]`. Exit 0 (`decision=ship|hold`) — render the block above and close
562
+ the run: `node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe resume --slug <slug> --close shipped
563
+ --cause "verdict=<verdict> rounds=<r> decision=<ship|hold>"`. Always issue this call — a `gate_h`
564
+ close is the ordinary case where `shapeup-run.js` handed off without closing the run, and the
565
+ GATE H → L4 path (scope-hammer's census, then this gate) is the one this instruction exists for.
566
+ The close itself is a once-only fact IN THE KERNEL (`closeRun`'s own guard reads a `closed_status:`
567
+ line that only `closeRun` ever writes — never the mutable `status:` line every phase rewrites, this
568
+ call included), not a conditional this instruction has to get right: if this run_id was NOT already
569
+ closed, this call performs the close, fresh. If it was already closed `shipped` (or `aborted`) and
570
+ the cause text is byte-identical to what is already on the ledger, this call is a true idempotent
571
+ no-op. If it was already closed with the SAME status but a genuinely different cause — a run closed
572
+ more than once across relaunches, the ordinary shape a `gate_h` hand-off after an earlier abort takes
573
+ — this call SUPERSEDES it: the new cause is written, the prior one is folded into the same
574
+ `close_cause` line rather than lost, and the kernel call itself still exits 0 (only the RunReturn a
575
+ launch's own `withWarnings` wraps carries the resulting `state_warning` — a prose-driven close like
576
+ this one has no RunReturn to attach it to, so read `close_cause` by hand if this branch matters to
577
+ you). If it was already closed with a DIFFERENT status altogether, this call is refused outright —
578
+ cause intact, never silently flipped. Exit 4 (`ask`) — the block above IS that stop; put it to the
579
+ PO and wait, same as always. L4's answer set carries no `abort`.
580
+
544
581
  ---
545
582
 
@@ -630,7 +630,7 @@ trace. See `references/gates.md` — GATE L0.1.
630
630
  ## Central domain registry
631
631
 
632
632
  Every record type and payload field that crosses a skill boundary is defined exactly once in
633
- `skills/tech-lead/schemas/domain.schema.json` — the envelope schemas (`work-order.schema.json`,
633
+ `kernel/schemas/domain.schema.json` — the envelope schemas (`work-order.schema.json`,
634
634
  `work-result.schema.json`) only `$ref` it. The registry annotates each entity's tier
635
635
  (SHARED/LOCAL), location, sole writer, and readers, carries the machine-readable ERD (`x-erd`),
636
636
  and maps which payload fields each worker may rely on (`x-payload-by-worker`).
@@ -686,13 +686,15 @@ lens: lite | standard | cross-context
686
686
  eval_dimensions: [spec-conformance] # the set from GATE L0.5 (init-run --dimensions); every EVAL order is compiled from THIS line
687
687
  max_rounds: 3
688
688
  auto_level: interactive | auto | unattended
689
- status: orienting | mapping | building | evaluating | shipped | escalated
689
+ status: orienting | mapping | building | evaluating | shipped | escalated | aborted
690
690
  final_verdict: ~ | pass | fail | not-evaluated
691
691
  rounds_used: [N]
692
692
  discovered_rounds: [N]
693
693
  deploy: ~ | deployed | pending-po
694
694
  started_at: [ISO]
695
695
  closed_at: ~ | [ISO]
696
+ close_cause: ~ | [why the run ended at that terminal status — `probe resume --close` writes this and closed_at together]
697
+ closed_status: ~ | [the terminal status actually closed — written ONLY by `probe resume --close`, never by `--set-status`, so it is immune to `status:` above being rewritten by ordinary phase traffic after the close]
696
698
  ---
697
699
  ```
698
700