shapeup-sdlc 3.5.0 → 3.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/AGENTS.md +12 -2
- package/README.md +1 -1
- package/kernel/compile.mjs +18 -6
- package/kernel/harness.mjs +3 -2
- package/kernel/init/run-args.mjs +206 -0
- package/kernel/init/run.mjs +10 -0
- package/kernel/lib/paths.mjs +10 -0
- package/kernel/probe/concurrency.mjs +31 -6
- package/kernel/probe/digest.mjs +15 -1
- package/kernel/probe/owner.mjs +4 -1
- package/kernel/probe/resume.mjs +188 -5
- package/kernel/probe/rounds.mjs +104 -0
- package/kernel/reduce/ingest.mjs +53 -12
- package/kernel/reduce/ship.mjs +15 -30
- package/kernel/reduce/snapshot.mjs +23 -2
- package/kernel/report/export.mjs +54 -2
- package/kernel/report/facts.mjs +24 -2
- package/{skills/tech-lead → kernel}/schemas/domain.schema.json +15 -12
- package/kernel/verify/envelope.mjs +2 -2
- package/kernel/verify/skills.mjs +1 -1
- package/package.json +1 -1
- package/skills/coach/SKILL.md +8 -2
- package/skills/hill-chart/SKILL.md +3 -4
- package/skills/scope-hammer/SKILL.md +1 -1
- package/skills/tech-lead/SKILL.md +10 -10
- package/skills/tech-lead/references/gates.md +48 -11
- package/skills/tech-lead/references/protocol.md +4 -2
- package/skills/tech-lead/workflows/shapeup-run.js +133 -37
- package/skills/translator/SKILL.md +1 -1
- /package/{skills/tech-lead → kernel}/schemas/gate-answers.schema.json +0 -0
- /package/{skills/tech-lead → kernel}/schemas/work-order.schema.json +0 -0
- /package/{skills/tech-lead → kernel}/schemas/work-result.schema.json +0 -0
package/kernel/report/export.mjs
CHANGED
|
@@ -47,10 +47,11 @@ import { runArgs } from "../lib/argv.mjs";
|
|
|
47
47
|
import { splitFrontmatter } from "../lib/contract.mjs";
|
|
48
48
|
import { runIdFromReceipt, readReceipt } from "../lib/paths.mjs";
|
|
49
49
|
import { TABLES, runRow, dispatchFacts } from "./facts.mjs";
|
|
50
|
+
import { deriveRounds } from "../probe/rounds.mjs";
|
|
50
51
|
import {
|
|
51
52
|
localDir, activeScope, receipt as receiptPath, harnessRun, ordersDir, resultsDir,
|
|
52
53
|
trials as trialsPath, verdictsDir, evaluationDir, decisions as decisionsPath,
|
|
53
|
-
exportsDir, exportRunDir,
|
|
54
|
+
gates as gatesPath, roundBuildDir, exportsDir, exportRunDir,
|
|
54
55
|
} from "../lib/paths.mjs";
|
|
55
56
|
|
|
56
57
|
export const EXPORT_SCHEMA_VERSION = 1;
|
|
@@ -164,6 +165,51 @@ function criterionRows(dir, runId, t) {
|
|
|
164
165
|
return out;
|
|
165
166
|
}
|
|
166
167
|
|
|
168
|
+
/**
|
|
169
|
+
* Flatten one gate-crossing ledger row (`gates.jsonl`, `kernel/gate.mjs`'s sole writer) into a
|
|
170
|
+
* flat `gate_decision` fact row. The row on disk already carries exactly these fields
|
|
171
|
+
* (see `appendGateLedger`), so this is a pass-through with a stamped `run_id` fallback rather than
|
|
172
|
+
* a re-derivation: two readers of "what did this gate decide" must not compute the answer twice.
|
|
173
|
+
* @param {object} g - One parsed line of `gates.jsonl`.
|
|
174
|
+
* @param {(string|null)} runId - Run key for a row written before it carried its own.
|
|
175
|
+
* @returns {object} A flat `gate_decision` row.
|
|
176
|
+
*/
|
|
177
|
+
function gateDecisionRow(g, runId) {
|
|
178
|
+
return {
|
|
179
|
+
run_id: g?.run_id ?? runId ?? null,
|
|
180
|
+
gate: g?.gate ?? null,
|
|
181
|
+
decision: g?.decision ?? null,
|
|
182
|
+
status: g?.status ?? null,
|
|
183
|
+
source: g?.source ?? null,
|
|
184
|
+
round: g?.round ?? null,
|
|
185
|
+
has_note: !!(g?.note && String(g.note).trim()),
|
|
186
|
+
};
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* Flatten one round build-gate artifact (`kernel/verify/build.mjs`'s `writeRoundBuild`) into a flat
|
|
191
|
+
* `build_gate` fact row. The gate ends a round exactly as EVAL does (AGENTS.md's round
|
|
192
|
+
* build gate ⚙), and until now had no fact table at all.
|
|
193
|
+
* @param {object} a - A parsed round-build artifact.
|
|
194
|
+
* @param {(string|null)} runId - Run key for an artifact written before it carried its own.
|
|
195
|
+
* @returns {object} A flat `build_gate` row.
|
|
196
|
+
*/
|
|
197
|
+
function buildGateRow(a, runId) {
|
|
198
|
+
const steps = Array.isArray(a?.steps) ? a.steps : [];
|
|
199
|
+
return {
|
|
200
|
+
run_id: a?.run_id ?? runId ?? null,
|
|
201
|
+
round: a?.round ?? null,
|
|
202
|
+
trial: a?.trial ?? null,
|
|
203
|
+
at: a?.at ?? null,
|
|
204
|
+
overall: a?.overall ?? null,
|
|
205
|
+
archetype: a?.archetype ?? null,
|
|
206
|
+
steps_total: steps.length,
|
|
207
|
+
steps_failed: steps.filter((s) => !s?.skipped && s?.pass === false).length,
|
|
208
|
+
warnings: Array.isArray(a?.warnings) ? a.warnings.length : 0,
|
|
209
|
+
discovered_tasks: Array.isArray(a?.discovered_tasks) ? a.discovered_tasks.length : 0,
|
|
210
|
+
};
|
|
211
|
+
}
|
|
212
|
+
|
|
167
213
|
// ---------------------------------------------------------------------------
|
|
168
214
|
// The export itself
|
|
169
215
|
// ---------------------------------------------------------------------------
|
|
@@ -190,7 +236,10 @@ export function collectRun(cwd, slug) {
|
|
|
190
236
|
const results = readJsonDir(resultsDir(cwd, slug), t);
|
|
191
237
|
|
|
192
238
|
const { dispatch, ac_result, discovery, file_touched } = dispatchFacts({ orders, results, runId });
|
|
193
|
-
|
|
239
|
+
// Computed here, once, from the same trace this whole function reads, and handed to
|
|
240
|
+
// runRow rather than re-derived by it: runRow stays pure (no I/O), this function already has cwd.
|
|
241
|
+
const rounds = deriveRounds(cwd, slug, ledger.rounds_used);
|
|
242
|
+
const run = runRow({ receipt: rec, ledger, runId, rounds });
|
|
194
243
|
|
|
195
244
|
// Hook decisions are checkout-wide, so they are FILTERED to this run rather than read from a
|
|
196
245
|
// per-run file. Rows with a null key belong to no run (a hook that fired outside one) and are
|
|
@@ -208,6 +257,9 @@ export function collectRun(cwd, slug) {
|
|
|
208
257
|
t0_verdict: readJsonDir(verdictsDir(cwd, slug), t).map((a) => t0Row(a, runId)),
|
|
209
258
|
criterion_verdict: criterionRows(evaluationDir(cwd, slug), runId, t),
|
|
210
259
|
hook_decision,
|
|
260
|
+
// The decision that crossed each gate, and the round build gate's own artifact.
|
|
261
|
+
gate_decision: readJsonl(gatesPath(cwd, slug), t).map((g) => gateDecisionRow(g, runId)),
|
|
262
|
+
build_gate: readJsonDir(roundBuildDir(cwd, slug), t).map((a) => buildGateRow(a, runId)),
|
|
211
263
|
},
|
|
212
264
|
defects: { records_skipped: t.skipped },
|
|
213
265
|
};
|
package/kernel/report/facts.mjs
CHANGED
|
@@ -27,6 +27,11 @@
|
|
|
27
27
|
export const TABLES = [
|
|
28
28
|
"run", "dispatch", "ac_result", "discovery", "file_touched",
|
|
29
29
|
"trial", "t0_verdict", "criterion_verdict", "hook_decision",
|
|
30
|
+
// The decision that shipped a run (or any other gate) had a ledger row (`gates.jsonl`)
|
|
31
|
+
// and no table — a reader had to open the LOCAL trace itself, which the export exists so nobody
|
|
32
|
+
// has to. `build_gate` is the round build gate's own artifact (kernel/verify/build.mjs), on the
|
|
33
|
+
// same terms: it ends a round exactly as EVAL does, and had no table either.
|
|
34
|
+
"gate_decision", "build_gate",
|
|
30
35
|
];
|
|
31
36
|
|
|
32
37
|
/** Coerce anything to a finite number, or null. Keeps `0` and rejects `NaN`/`""`/undefined. */
|
|
@@ -76,9 +81,14 @@ export function parseOrderStem(orderId) {
|
|
|
76
81
|
* @param {(object|null)} o.receipt - Parsed `receipt.json`.
|
|
77
82
|
* @param {(object|null)} [o.ledger] - Parsed `harness-run.md` frontmatter (a flat scalar map).
|
|
78
83
|
* @param {(string|null)} [o.runId] - The run key, when already resolved.
|
|
84
|
+
* @param {({rounds_used:*, rounds_judged:(number|null)}|null)} [o.rounds] - The two-number
|
|
85
|
+
* derivation (`probe/rounds.mjs`'s `deriveRounds`), computed by the caller because it needs the
|
|
86
|
+
* filesystem and this function stays pure. Falls back to the ledger's own (unreliable — see
|
|
87
|
+
* `deriveRounds`) `rounds_used` line when the caller has not derived one, so an existing caller
|
|
88
|
+
* is unaffected rather than broken.
|
|
79
89
|
* @returns {(object|null)} The run row, or null when there is no receipt to describe.
|
|
80
90
|
*/
|
|
81
|
-
export function runRow({ receipt, ledger = null, runId = null }) {
|
|
91
|
+
export function runRow({ receipt, ledger = null, runId = null, rounds = null }) {
|
|
82
92
|
if (!receipt) return null;
|
|
83
93
|
const c = receipt.config || {};
|
|
84
94
|
const fm = ledger || {};
|
|
@@ -87,6 +97,9 @@ export function runRow({ receipt, ledger = null, runId = null }) {
|
|
|
87
97
|
slug: receipt.slug ?? null,
|
|
88
98
|
started_at: receipt.started_at ?? null,
|
|
89
99
|
closed_at: fm.closed_at && fm.closed_at !== "~" ? fm.closed_at : null,
|
|
100
|
+
// Why the run ended at a terminal status, written by `probe resume --close` alongside
|
|
101
|
+
// `closed_at` — the two facts a trace needs to tell a live run from a dead one apart.
|
|
102
|
+
close_cause: fm.close_cause && fm.close_cause !== "~" ? fm.close_cause : null,
|
|
90
103
|
intake_sha256: receipt.intake_sha256 ?? null,
|
|
91
104
|
intake_chars: num(receipt.intake_chars),
|
|
92
105
|
intake_lines: num(receipt.intake_lines),
|
|
@@ -101,8 +114,17 @@ export function runRow({ receipt, ledger = null, runId = null }) {
|
|
|
101
114
|
// Copied from the ledger, never re-derived: the run's own status line is the harness's answer,
|
|
102
115
|
// and a read plane that recomputed it would be asserting a second one.
|
|
103
116
|
status: fm.status ?? null,
|
|
117
|
+
// The terminal status `closeRun` alone writes, carried BESIDE `status` rather than instead of
|
|
118
|
+
// it. `status:` is ordinary phase traffic and a later phase may move it, so an export that
|
|
119
|
+
// carried only that line could show a run wearing another close's cause. These two disagreeing
|
|
120
|
+
// is itself the fact worth exporting: it says the ledger was written after the close.
|
|
121
|
+
closed_status: fm.closed_status && fm.closed_status !== "~" ? fm.closed_status : null,
|
|
104
122
|
final_verdict: fm.final_verdict && fm.final_verdict !== "~" ? fm.final_verdict : null,
|
|
105
|
-
|
|
123
|
+
// Two fields, not one: `rounds_used` is the highest round carrying ANY build evidence,
|
|
124
|
+
// `rounds_judged` the highest round EVAL actually returned a verdict for. A caller that has not
|
|
125
|
+
// derived `rounds` falls back to the ledger's own (pre-fix, unreliable) line, non-regression.
|
|
126
|
+
rounds_used: rounds ? num(Number(rounds.rounds_used)) : num(Number(fm.rounds_used)),
|
|
127
|
+
rounds_judged: rounds ? num(rounds.rounds_judged) : null,
|
|
106
128
|
};
|
|
107
129
|
}
|
|
108
130
|
|
|
@@ -668,14 +668,14 @@
|
|
|
668
668
|
"string",
|
|
669
669
|
"null"
|
|
670
670
|
],
|
|
671
|
-
"description": "Source file of the failure; null/absent when the log line
|
|
671
|
+
"description": "Source file of the failure; null/absent when the log line named no file at all (raw triple — never invented)."
|
|
672
672
|
},
|
|
673
673
|
"line": {
|
|
674
674
|
"type": [
|
|
675
675
|
"integer",
|
|
676
676
|
"null"
|
|
677
677
|
],
|
|
678
|
-
"description": "Line of the failure; null/absent when the
|
|
678
|
+
"description": "Line of the failure; null/absent when the diagnostic carried no line number — independently of `file`, since a diagnostic can name a file with no line (a resource-compiler error, for example). Never invented to satisfy a shape."
|
|
679
679
|
},
|
|
680
680
|
"core_message": {
|
|
681
681
|
"type": "string",
|
|
@@ -1891,9 +1891,10 @@
|
|
|
1891
1891
|
"building",
|
|
1892
1892
|
"evaluating",
|
|
1893
1893
|
"shipped",
|
|
1894
|
-
"escalated"
|
|
1894
|
+
"escalated",
|
|
1895
|
+
"aborted"
|
|
1895
1896
|
],
|
|
1896
|
-
"description": "Mirrors harness-run.md frontmatter status."
|
|
1897
|
+
"description": "Mirrors harness-run.md frontmatter status — kernel/probe/resume.mjs's RUN_STATUSES is the source enum this one must not drift from."
|
|
1897
1898
|
},
|
|
1898
1899
|
"round": {
|
|
1899
1900
|
"type": "integer",
|
|
@@ -1904,7 +1905,12 @@
|
|
|
1904
1905
|
"description": "From the latest t0/verdicts/r<N>-a<M>.json filename."
|
|
1905
1906
|
},
|
|
1906
1907
|
"rounds_used": {
|
|
1907
|
-
"type": "integer"
|
|
1908
|
+
"type": "integer",
|
|
1909
|
+
"description": "Highest round carrying any build evidence (an order, a T0 verdict, a round build-gate artifact, or an EVAL result) — kernel/probe/rounds.mjs's deriveRounds(), the same derivation the ship report and the export use. Falls back to harness-run.md's literal frontmatter value only when no such evidence exists on disk."
|
|
1910
|
+
},
|
|
1911
|
+
"rounds_judged": {
|
|
1912
|
+
"type": "integer",
|
|
1913
|
+
"description": "Highest round EVAL actually returned a verdict for (an evaluate-r<N>.json result on disk) — its own field, never folded into rounds_used: a round built is not a round judged. Omitted when no round has been judged yet."
|
|
1908
1914
|
},
|
|
1909
1915
|
"max_rounds": {
|
|
1910
1916
|
"type": "integer"
|
|
@@ -2704,10 +2710,10 @@
|
|
|
2704
2710
|
}
|
|
2705
2711
|
},
|
|
2706
2712
|
"RunArgs": {
|
|
2707
|
-
"description": "C1 — the launch half of the workflow's only conversation.
|
|
2713
|
+
"description": "C1 — the launch half of the workflow's only conversation. Resolved ONCE at GATE L0 from harness init run output + the L0.8 model matrix + budgets, then built by `harness init run-args`, which writes it to .shapeup/<slug>/run-args.json AND prints it — tech-lead passes that printed value to the harness run launch as one JSON literal, never a second, hand-typed copy of it. The workflow cannot ask follow-ups and cannot read config files itself, so everything a run will ever need travels in this one record. A workflow script validates its own subset of this shape in code (no runtime schema check at the C1 boundary itself); this entry is the central-registry definition the workflow script, `harness init run-args` and the tech-lead skill all read as the one true shape.",
|
|
2708
2714
|
"x-tier": "EMBEDDED",
|
|
2709
|
-
"x-location": ".shapeup/<slug>/run-args.json — written fresh
|
|
2710
|
-
"x-writer": "tech-lead
|
|
2715
|
+
"x-location": ".shapeup/<slug>/run-args.json — written fresh on every launch and relaunch by `harness init run-args`; the workflow receives it as its args and never reads other config",
|
|
2716
|
+
"x-writer": "harness init run-args (kernel), invoked by tech-lead at GATE L0 on every launch AND every relaunch after a paused gate",
|
|
2711
2717
|
"x-readers": "the Workflow runtime (shapeup-run, and shapeup-run's own inner round dispatch)",
|
|
2712
2718
|
"x-not-here": "Run config the LEDGER already carries does NOT get a second home in RunArgs — eval_dimensions, lens, spec_folder, stack, run_cmd, app_url are read off harness-run.md frontmatter by resume-state on every launch AND every relaunch, so a copy here would be a second source that can disagree with the first. RunArgs carries what a workflow cannot derive from disk (identity, budgets, the model matrix, pluginRoot, startedAt) plus noEval, which no frontmatter line holds.",
|
|
2713
2719
|
"type": "object",
|
|
@@ -2759,16 +2765,13 @@
|
|
|
2759
2765
|
},
|
|
2760
2766
|
"budgets": {
|
|
2761
2767
|
"type": "object",
|
|
2762
|
-
"description": "The
|
|
2768
|
+
"description": "The two RunArgs-level circuit breakers (AGENTS.md) — outer round_budget, inner attempt_budget. The third, opt-in DEADLINE breaker (the wall-clock budget) is NOT a RunArgs field: it is typed once, at `harness init run --wall-clock-budget`, and lands in the run receipt's `wall_clock_budget_s` — `harness verify budget` reads that receipt field directly and never sees this launch's RunArgs at all, so it has no member here to declare.",
|
|
2763
2769
|
"properties": {
|
|
2764
2770
|
"maxRounds": {
|
|
2765
2771
|
"type": "integer"
|
|
2766
2772
|
},
|
|
2767
2773
|
"attemptBudget": {
|
|
2768
2774
|
"type": "integer"
|
|
2769
|
-
},
|
|
2770
|
-
"wallClockS": {
|
|
2771
|
-
"type": "integer"
|
|
2772
2775
|
}
|
|
2773
2776
|
}
|
|
2774
2777
|
},
|
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
// #/$defs/Name — a definition in the SAME schema document
|
|
13
13
|
// domain.schema.json#/$defs/Name — a definition in a SIBLING file (the central domain
|
|
14
14
|
// registry; resolved against the schema's own dir,
|
|
15
|
-
// falling back to
|
|
15
|
+
// falling back to kernel/schemas/)
|
|
16
16
|
//
|
|
17
17
|
// Usage (CLI): node kernel/harness.mjs verify envelope <envelope.json> <schema.json>
|
|
18
18
|
// exit 0 = valid, 1 = invalid (errors printed one per line)
|
|
@@ -28,7 +28,7 @@ import { runArgs } from "../lib/argv.mjs";
|
|
|
28
28
|
import { runHook, readStdin, settle } from "../../hooks/lib/decision.mjs";
|
|
29
29
|
|
|
30
30
|
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
31
|
-
export const SCHEMAS_DIR = resolve(HERE, "
|
|
31
|
+
export const SCHEMAS_DIR = resolve(HERE, "../schemas");
|
|
32
32
|
|
|
33
33
|
/**
|
|
34
34
|
* Validate a value against the JSON-Schema subset the envelope schemas use (type, required,
|
package/kernel/verify/skills.mjs
CHANGED
|
@@ -45,7 +45,7 @@ export const PLUGIN_ROOT = resolve(HERE, "../..");
|
|
|
45
45
|
* its own domain registry has a broken installation, which is the very thing being checked.
|
|
46
46
|
*/
|
|
47
47
|
export function roster(root = PLUGIN_ROOT) {
|
|
48
|
-
const schemaPath = join(root, "
|
|
48
|
+
const schemaPath = join(root, "kernel/schemas/domain.schema.json");
|
|
49
49
|
const schema = JSON.parse(readFileSync(schemaPath, "utf8"));
|
|
50
50
|
const names = schema?.$defs?.WorkerName?.enum;
|
|
51
51
|
if (!Array.isArray(names) || !names.length) {
|
package/package.json
CHANGED
package/skills/coach/SKILL.md
CHANGED
|
@@ -93,7 +93,7 @@ never lands in any worker's KB.
|
|
|
93
93
|
Orchestrated, this skill is dispatched like every worker: a **WorkOrder** in (`--order <path>`,
|
|
94
94
|
operation `coach` or `scan`), a **WorkResult** out. Standalone, the raw feedback is passed
|
|
95
95
|
directly; it maps onto the one payload field registered for this worker in the central domain
|
|
96
|
-
registry (`
|
|
96
|
+
registry (`kernel/schemas/domain.schema.json`, `x-payload-by-worker`):
|
|
97
97
|
|
|
98
98
|
| Payload field | Standalone form | Meaning |
|
|
99
99
|
|---|---|---|
|
|
@@ -117,7 +117,13 @@ fields nobody used" → "Prefer the minimum DTO that satisfies the AC; don't add
|
|
|
117
117
|
fields"). Keep the originating why — a rule without its reason gets ignored or misapplied.
|
|
118
118
|
|
|
119
119
|
### Step 2 — ⏸ GATE COACH-1: Categorize (ASK, never assume)
|
|
120
|
-
This is the load-bearing gate. **
|
|
120
|
+
This is the load-bearing gate. **Resolve it first** — `node
|
|
121
|
+
"${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" gate --resolve COACH-1 --slug <slug>
|
|
122
|
+
[--file <path>|--preset <name>]` — so the ledger carries a row for the decision this gate makes,
|
|
123
|
+
same as every other gate in the run. Exit 0 (`decision=skip`) — an unattended lane with no live PO;
|
|
124
|
+
record nothing and stop here, the same outcome the CI preset's own note already documents. Exit 4
|
|
125
|
+
(`ask`) — proceed with the categorization below, which IS the PO conversation this decision opens.
|
|
126
|
+
**Do not infer which skill a rule belongs to** — a
|
|
121
127
|
miscategorized rule lands in a file the wrong worker reads (or no worker reads). Present every
|
|
122
128
|
candidate rule and ask the PO to assign each one. Emit this block, then stop and wait:
|
|
123
129
|
|
|
@@ -10,7 +10,7 @@ re-runs a computation that would erase true history.**
|
|
|
10
10
|
|
|
11
11
|
You are not a worker: no WorkOrder, no WorkResult, invoked directly by the user (or by `/hill`)
|
|
12
12
|
exactly like `shapeup` is. There is nothing to declare in
|
|
13
|
-
`
|
|
13
|
+
`kernel/schemas/domain.schema.json` and nothing to teach `harness compile` or
|
|
14
14
|
`harness reduce ingest` — those steps exist only for dispatched workers.
|
|
15
15
|
|
|
16
16
|
## What you read
|
|
@@ -71,9 +71,8 @@ Build one `{ scope_id, phase }` object per file.
|
|
|
71
71
|
## Rendering — the injection contract
|
|
72
72
|
|
|
73
73
|
The engine ships at `assets/dashboard.template.html` — a complete, self-contained HTML page
|
|
74
|
-
(inline CSS/JS, no external fetch beyond Google Fonts, no build step
|
|
75
|
-
|
|
76
|
-
fill in real data, and write the result.
|
|
74
|
+
(inline CSS/JS, no external fetch beyond Google Fonts, no build step). Do not rewrite it from a
|
|
75
|
+
text description; read it, fill in real data, and write the result.
|
|
77
76
|
|
|
78
77
|
1. For each discovered slug, build one entry:
|
|
79
78
|
|
|
@@ -150,7 +150,7 @@ the harness (this is neither the generator nor the evaluator).
|
|
|
150
150
|
Orchestrated, this skill is dispatched like every worker: a **WorkOrder** in (`--order <path>`,
|
|
151
151
|
operation `hammer`), a **WorkResult** out. The standalone flags below map 1:1 onto the payload
|
|
152
152
|
fields registered for this worker in the central domain registry
|
|
153
|
-
(`
|
|
153
|
+
(`kernel/schemas/domain.schema.json`, `x-payload-by-worker`):
|
|
154
154
|
|
|
155
155
|
| Payload field | Standalone flag | Meaning |
|
|
156
156
|
|---|---|---|
|
|
@@ -60,15 +60,15 @@ check the lane:
|
|
|
60
60
|
legacy loop instead — `references/protocol.md` (BUILD(r)/EVAL) + `references/protocol.md`
|
|
61
61
|
carry the full step-by-step for both the tiny lane and a scope-less BUILD loop, verbatim, non-
|
|
62
62
|
regression. Stop reading this file here for that run.
|
|
63
|
-
- **Otherwise** (the common case — a scoped spec, any auto level):
|
|
64
|
-
(`
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
`.shapeup/<slug>/run-args.json`
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
63
|
+
- **Otherwise** (the common case — a scoped spec, any auto level): resolve every switch the
|
|
64
|
+
operator typed (`references/gates.md` GATE L0.9b has the flag→field table) and run the kernel's
|
|
65
|
+
sole `RunArgs` writer: `node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" init run-args --slug
|
|
66
|
+
<slug> --auto-level <level> --exec-model <n> [--eval-model <n>] [--qa-model <n>] --max-rounds <N>
|
|
67
|
+
--attempts <N> --plugin-root "${CLAUDE_PLUGIN_ROOT}" [--answers <a>] [--lane <l>] [--no-eval]
|
|
68
|
+
[--no-qa] [--adversarial-verify] [--parallel-scopes <N>]`. It writes `.shapeup/<slug>/run-args.json`
|
|
69
|
+
fresh on every launch/relaunch and prints that identical object — **pass it to `Workflow`
|
|
70
|
+
verbatim, never re-type it**. Then launch with the **`Workflow` tool** — naming `init run`'s
|
|
71
|
+
staged copy, never the install path:
|
|
72
72
|
|
|
73
73
|
```
|
|
74
74
|
Workflow({
|
|
@@ -114,7 +114,7 @@ did not actually receive from the PO — an unattended lane with no answer for a
|
|
|
114
114
|
|
|
115
115
|
FIRST freeze the evidence — run state is gitignored, so `shapeup/<slug>/REPORT.md` (already
|
|
116
116
|
written by `shapeup-run.js` via `harness reduce ship`, or write it now on a `gate_h` close) is all a
|
|
117
|
-
teammate sees. Then emit:
|
|
117
|
+
teammate sees. Then RESOLVE the gate — `references/gates.md` GATE L4 has the call — and emit:
|
|
118
118
|
|
|
119
119
|
```
|
|
120
120
|
⏸ GATE L4 — Ship Sign-Off
|
|
@@ -113,11 +113,13 @@ Collect (explicit — never inferred):
|
|
|
113
113
|
gate to be its first execution.
|
|
114
114
|
```
|
|
115
115
|
|
|
116
|
-
**L0.9b — the launch record.** Every switch the operator typed becomes a `RunArgs`
|
|
117
|
-
does nothing at all: the workflow cannot read a config file and cannot ask a follow-up,
|
|
118
|
-
that stops at the skill boundary was accepted and ignored. That is not hypothetical —
|
|
119
|
-
documented in seven places across the shipped set and inert in all of them, because no
|
|
120
|
-
protocol ever put `noQa` into the record.
|
|
116
|
+
**L0.9b — the launch record.** Every switch the operator typed to *this launch* becomes a `RunArgs`
|
|
117
|
+
field, or it does nothing at all: the workflow cannot read a config file and cannot ask a follow-up,
|
|
118
|
+
so a flag that stops at the skill boundary was accepted and ignored. That is not hypothetical —
|
|
119
|
+
`--no-qa` was documented in seven places across the shipped set and inert in all of them, because no
|
|
120
|
+
line of this protocol ever put `noQa` into the record. `--wall-clock-budget` is the one flag below
|
|
121
|
+
that is not a `RunArgs` field at all — it is consumed earlier, at `init run` itself, and never
|
|
122
|
+
needed to reach this launch; see its row for where it actually lands.
|
|
121
123
|
|
|
122
124
|
| Flag | `RunArgs` field |
|
|
123
125
|
|---|---|
|
|
@@ -125,14 +127,19 @@ protocol ever put `noQa` into the record.
|
|
|
125
127
|
| `--no-qa` | `noQa: true` |
|
|
126
128
|
| `--parallel-scopes N` | `maxParallelScopes: N` — how many scopes build at once (default 4; `1` = sequential) |
|
|
127
129
|
| `--adversarial-verify` | `adversarialVerify: true` |
|
|
128
|
-
| `--rounds N` / `--attempts N`
|
|
130
|
+
| `--rounds N` / `--attempts N` | `budgets.{maxRounds,attemptBudget}` |
|
|
129
131
|
| `--gate-answers <set>` | `answers` |
|
|
132
|
+
| `--wall-clock-budget S` | *(not a `RunArgs` field)* — typed once, on the `harness init run` command line itself, not on this launch; it lands straight in the run receipt as `wall_clock_budget_s`, and the deadline breaker reads that receipt field directly — consumed by `kernel/verify/budget.mjs` as `wall_clock_budget_s`. `budgets` declares only `maxRounds`/`attemptBudget` — the schema, `SKILL.md`'s own RunArgs contract line and this script's own header comment all agree there is no third member |
|
|
130
133
|
| `--orch-model/--exec-model/--eval-model/--qa-model` | `models.{…}` (L0.8) |
|
|
131
134
|
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
135
|
+
`harness init run-args` (invoked at Step 2 of `SKILL.md`) is the sole writer of the assembled
|
|
136
|
+
object: it takes the resolved values above, writes `.shapeup/<slug>/run-args.json` fresh on every
|
|
137
|
+
launch and relaunch, and prints the same object back so the launch never re-assembles it by hand. It
|
|
138
|
+
is the only artifact that records what a run was configured with; the ship report, a resumed session
|
|
139
|
+
and any later measurement all read it, and none of them can recover a value that only ever existed
|
|
140
|
+
as an argument. Step 2 is not merely advisory: `shapeup-run.js`'s own Preflight refuses to dispatch
|
|
141
|
+
ORIENT (or anything past it) when this file is missing at the run's local root — a launch that
|
|
142
|
+
skipped this step aborts there rather than proceeding on a silent default.
|
|
136
143
|
|
|
137
144
|
**L0.0 — intake precondition (before any other L0 collection):**
|
|
138
145
|
```
|
|
@@ -161,7 +168,14 @@ Model matrix : orch=[model] exec=[model] eval=[model] qa=[model] digester=[scrip
|
|
|
161
168
|
Budgets : round_budget=[N] (outer) attempt_budget=[N] (inner, per scope)
|
|
162
169
|
Knowledge : [tech-lead.md — N workflow rules, M suggested values (confirmed above) | none — `/retro --scan` or `/retro --research <stack>` seeds it (optional)]
|
|
163
170
|
```
|
|
164
|
-
|
|
171
|
+
**Resolve it** — this gate is this skill's own (the workflow never sees it), so it is this skill
|
|
172
|
+
that runs the same tool every other gate resolves through, not a paragraph read as a stand-in for
|
|
173
|
+
one: `node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" gate --resolve L0 --slug <slug>
|
|
174
|
+
[--file <path>|--preset <name>]`. Exit 4 (`ask`) is the confirmation this block already asks for —
|
|
175
|
+
put it to the PO and wait, same as the paragraph above always meant. Exit 0 (`decision=proceed`) —
|
|
176
|
+
continue straight to ORIENT, which is what `--unattended`'s pre-answered set resolves to. Exit 5
|
|
177
|
+
(`abort`) — stop; do not launch. Either way, the gate's own ledger row is what lets a later reader
|
|
178
|
+
see the decision that opened the run, not only the decisions that closed it.
|
|
165
179
|
|
|
166
180
|
---
|
|
167
181
|
|
|
@@ -541,5 +555,28 @@ On confirm:
|
|
|
541
555
|
- If the PO provides substantive feedback (not just 'y' or empty) → automatically delegate via Agent (model: exec — see references/protocol.md "Invocation mechanism"): Skill(shapeup-sdlc-plugin:coach) with the provided feedback for RLHF. The coach runs its own GATE COACH-1 to have the PO categorize each rule, then files it under the responsible skill in `shapeup/knowledge-base/<skill>.md` (committed → team-shared). Coachable: `task-executor`, `ba-pitch-analyzer`, `qa-edge-hunter`, `orient`, `scope-architect`, `solution-architect` (each reads its own file at the top of its next run) and `tech-lead` (workflow guidance, read at the next GATE L0). Guidance never decides a gate: a filed rule may add a question or a check to a gate block, never an answer. The tech lead does not categorize the feedback itself — that is the coach's gate, by design (no assumptions).
|
|
542
556
|
- Then output → `✅ [slug] [shipped & deployed | built & verified, deploy pending] — [r] rounds, verdict PASS.`
|
|
543
557
|
|
|
558
|
+
**Resolve the gate itself before any of the above** — this is the decision that shipped the run,
|
|
559
|
+
and without it the trace holds no record of that decision at all: `node
|
|
560
|
+
"${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" gate --resolve L4 --slug <slug>
|
|
561
|
+
[--file <path>|--preset <name>]`. Exit 0 (`decision=ship|hold`) — render the block above and close
|
|
562
|
+
the run: `node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe resume --slug <slug> --close shipped
|
|
563
|
+
--cause "verdict=<verdict> rounds=<r> decision=<ship|hold>"`. Always issue this call — a `gate_h`
|
|
564
|
+
close is the ordinary case where `shapeup-run.js` handed off without closing the run, and the
|
|
565
|
+
GATE H → L4 path (scope-hammer's census, then this gate) is the one this instruction exists for.
|
|
566
|
+
The close itself is a once-only fact IN THE KERNEL (`closeRun`'s own guard reads a `closed_status:`
|
|
567
|
+
line that only `closeRun` ever writes — never the mutable `status:` line every phase rewrites, this
|
|
568
|
+
call included), not a conditional this instruction has to get right: if this run_id was NOT already
|
|
569
|
+
closed, this call performs the close, fresh. If it was already closed `shipped` (or `aborted`) and
|
|
570
|
+
the cause text is byte-identical to what is already on the ledger, this call is a true idempotent
|
|
571
|
+
no-op. If it was already closed with the SAME status but a genuinely different cause — a run closed
|
|
572
|
+
more than once across relaunches, the ordinary shape a `gate_h` hand-off after an earlier abort takes
|
|
573
|
+
— this call SUPERSEDES it: the new cause is written, the prior one is folded into the same
|
|
574
|
+
`close_cause` line rather than lost, and the kernel call itself still exits 0 (only the RunReturn a
|
|
575
|
+
launch's own `withWarnings` wraps carries the resulting `state_warning` — a prose-driven close like
|
|
576
|
+
this one has no RunReturn to attach it to, so read `close_cause` by hand if this branch matters to
|
|
577
|
+
you). If it was already closed with a DIFFERENT status altogether, this call is refused outright —
|
|
578
|
+
cause intact, never silently flipped. Exit 4 (`ask`) — the block above IS that stop; put it to the
|
|
579
|
+
PO and wait, same as always. L4's answer set carries no `abort`.
|
|
580
|
+
|
|
544
581
|
---
|
|
545
582
|
|
|
@@ -630,7 +630,7 @@ trace. See `references/gates.md` — GATE L0.1.
|
|
|
630
630
|
## Central domain registry
|
|
631
631
|
|
|
632
632
|
Every record type and payload field that crosses a skill boundary is defined exactly once in
|
|
633
|
-
`
|
|
633
|
+
`kernel/schemas/domain.schema.json` — the envelope schemas (`work-order.schema.json`,
|
|
634
634
|
`work-result.schema.json`) only `$ref` it. The registry annotates each entity's tier
|
|
635
635
|
(SHARED/LOCAL), location, sole writer, and readers, carries the machine-readable ERD (`x-erd`),
|
|
636
636
|
and maps which payload fields each worker may rely on (`x-payload-by-worker`).
|
|
@@ -686,13 +686,15 @@ lens: lite | standard | cross-context
|
|
|
686
686
|
eval_dimensions: [spec-conformance] # the set from GATE L0.5 (init-run --dimensions); every EVAL order is compiled from THIS line
|
|
687
687
|
max_rounds: 3
|
|
688
688
|
auto_level: interactive | auto | unattended
|
|
689
|
-
status: orienting | mapping | building | evaluating | shipped | escalated
|
|
689
|
+
status: orienting | mapping | building | evaluating | shipped | escalated | aborted
|
|
690
690
|
final_verdict: ~ | pass | fail | not-evaluated
|
|
691
691
|
rounds_used: [N]
|
|
692
692
|
discovered_rounds: [N]
|
|
693
693
|
deploy: ~ | deployed | pending-po
|
|
694
694
|
started_at: [ISO]
|
|
695
695
|
closed_at: ~ | [ISO]
|
|
696
|
+
close_cause: ~ | [why the run ended at that terminal status — `probe resume --close` writes this and closed_at together]
|
|
697
|
+
closed_status: ~ | [the terminal status actually closed — written ONLY by `probe resume --close`, never by `--set-status`, so it is immune to `status:` above being rewritten by ordinary phase traffic after the close]
|
|
696
698
|
---
|
|
697
699
|
```
|
|
698
700
|
|