shapeup-sdlc 3.5.0 → 3.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/AGENTS.md +15 -4
- package/README.md +1 -1
- package/kernel/compile.mjs +55 -7
- package/kernel/harness.mjs +11 -4
- package/kernel/init/run-args.mjs +206 -0
- package/kernel/init/run.mjs +10 -0
- package/kernel/lib/paths.mjs +10 -0
- package/kernel/probe/attempts.mjs +135 -0
- package/kernel/probe/concurrency.mjs +31 -6
- package/kernel/probe/digest.mjs +15 -1
- package/kernel/probe/owner.mjs +4 -1
- package/kernel/probe/resume.mjs +314 -5
- package/kernel/probe/rounds.mjs +104 -0
- package/kernel/reduce/ingest.mjs +53 -12
- package/kernel/reduce/ship.mjs +15 -30
- package/kernel/reduce/snapshot.mjs +23 -2
- package/kernel/report/export.mjs +54 -2
- package/kernel/report/facts.mjs +24 -2
- package/{skills/tech-lead → kernel}/schemas/domain.schema.json +15 -12
- package/kernel/verify/envelope.mjs +2 -2
- package/kernel/verify/skills.mjs +1 -1
- package/package.json +1 -1
- package/skills/ba-pitch-analyzer/SKILL.md +1 -1
- package/skills/ba-pitch-analyzer/references/doc-schemas.md +6 -0
- package/skills/coach/SKILL.md +8 -2
- package/skills/hill-chart/SKILL.md +3 -4
- package/skills/scope-hammer/SKILL.md +11 -3
- package/skills/tech-lead/SKILL.md +10 -10
- package/skills/tech-lead/references/gates.md +48 -11
- package/skills/tech-lead/references/protocol.md +4 -2
- package/skills/tech-lead/workflows/shapeup-run.js +164 -40
- package/skills/translator/SKILL.md +1 -1
- /package/{skills/tech-lead → kernel}/schemas/gate-answers.schema.json +0 -0
- /package/{skills/tech-lead → kernel}/schemas/work-order.schema.json +0 -0
- /package/{skills/tech-lead → kernel}/schemas/work-result.schema.json +0 -0
package/kernel/reduce/ingest.mjs
CHANGED
|
@@ -8,6 +8,13 @@
|
|
|
8
8
|
// task_results[] → tick AC boxes, flip task frontmatter status, append Execution Log,
|
|
9
9
|
// update tasks/_index.md row, propagate unblocks (old P3.1–P3.6)
|
|
10
10
|
// discoveries[] → append to .shapeup/<slug>/discovery/ledger.md (old P3.7 / QA H.3)
|
|
11
|
+
// deviations[] (when → append to the SAME discovery ledger, tagged distinctly. A worker's
|
|
12
|
+
// status:"escalated") protocol names deviations[] as the only channel a blocked worker has —
|
|
13
|
+
// there is no escalates[] field — and until now nothing routed it anywhere:
|
|
14
|
+
// a worker that correctly stopped rather than guessed produced a number and
|
|
15
|
+
// silence. Landing here reaches a reader that already exists — the census
|
|
16
|
+
// reads this ledger's open entries, and the ship report's own "Discovered,
|
|
17
|
+
// not built" section is generated from it.
|
|
11
18
|
// verdict.criteria[] → append evaluation/.verdicts-<target>.jsonl (old evaluator B.0), every
|
|
12
19
|
// row keyed by run_id and carrying the judge's traces_to[] anchor back to
|
|
13
20
|
// the requirement — see step 4 for why neither may be dropped here
|
|
@@ -33,7 +40,7 @@ import { tasksDir, localRoot, dispatchReceipts, legLedger, readRunId } from "../
|
|
|
33
40
|
import { citationProblem } from "../probe/eval.mjs";
|
|
34
41
|
|
|
35
42
|
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
36
|
-
const RESULT_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "
|
|
43
|
+
const RESULT_SCHEMA = JSON.parse(readFileSync(resolve(HERE, "../schemas/work-result.schema.json"), "utf8"));
|
|
37
44
|
|
|
38
45
|
/**
|
|
39
46
|
* @returns {string} Today's date as an ISO `YYYY-MM-DD` string (UTC), for log/frontmatter stamps.
|
|
@@ -195,7 +202,7 @@ export function updateBoardRow(indexBody, taskId, done) {
|
|
|
195
202
|
* verdict{criteria[],refuted[]}).
|
|
196
203
|
* @param {{cwd:string}} opts - cwd: working-directory root every LOCAL path resolves against.
|
|
197
204
|
* @returns {{slug:string, tasks_updated:string[], acs_ticked:number, unblocked:string[],
|
|
198
|
-
* discoveries_appended:number, refuted_unticked:number, verdict_lines:number}} A summary of every write performed.
|
|
205
|
+
* discoveries_appended:number, escalations_appended:number, refuted_unticked:number, verdict_lines:number}} A summary of every write performed.
|
|
199
206
|
* @throws {Error} If a task/board/ledger file it must write is not writable (fs error propagates).
|
|
200
207
|
* `evaluation/.verdicts-*.jsonl` under `.shapeup/<slug>/`.
|
|
201
208
|
*/
|
|
@@ -213,7 +220,7 @@ export function applyResult(result, { cwd }) {
|
|
|
213
220
|
*/
|
|
214
221
|
function applyResultLocked(result, { cwd, slug }) {
|
|
215
222
|
const local = localRoot(cwd, slug);
|
|
216
|
-
const summary = { slug, tasks_updated: [], acs_ticked: 0, unblocked: [], discoveries_appended: 0, refuted_unticked: 0, verdict_lines: 0 };
|
|
223
|
+
const summary = { slug, tasks_updated: [], acs_ticked: 0, unblocked: [], discoveries_appended: 0, escalations_appended: 0, refuted_unticked: 0, verdict_lines: 0 };
|
|
217
224
|
|
|
218
225
|
// 1. Task results → task files + board (old task-executor P3.1/P3.2/P3.6).
|
|
219
226
|
const boardIndex = join(local, "tasks", "_index.md");
|
|
@@ -273,18 +280,52 @@ function applyResultLocked(result, { cwd, slug }) {
|
|
|
273
280
|
}
|
|
274
281
|
}
|
|
275
282
|
|
|
276
|
-
// 3. Discoveries → the ledger (old P3.7 / QA H.3)
|
|
277
|
-
|
|
283
|
+
// 3. Discoveries → the ledger (old P3.7 / QA H.3), plus an ESCALATE's deviations, tagged the same
|
|
284
|
+
// way and landed under the same heading. Single writer: this script. A worker's protocol names
|
|
285
|
+
// `deviations[]` as the only channel a blocked worker has — there is no `escalates[]` field —
|
|
286
|
+
// and routing it into THIS ledger is what makes it reach a reader that already exists: the
|
|
287
|
+
// ship report's own "Discovered, not built" section, and the census a scope's own worker reads
|
|
288
|
+
// before proposing a cut, both read this file's unresolved `+`/`~` entries.
|
|
289
|
+
const escalations = result.status === "escalated" ? (result.deviations || []) : [];
|
|
290
|
+
if (result.discoveries?.length || escalations.length) {
|
|
278
291
|
const ledgerDir = join(local, "discovery");
|
|
279
292
|
mkdirSync(ledgerDir, { recursive: true });
|
|
280
293
|
const ledger = join(ledgerDir, "ledger.md");
|
|
281
294
|
if (!existsSync(ledger)) writeFileSync(ledger, `---\nfeature: ${slug}\n---\n# Discovery Ledger — ${slug}\n`);
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
295
|
+
// IDEMPOTENT ON THE ORDER, not merely append-only. `reduce ingest` is the single writer, but
|
|
296
|
+
// nothing stops the SAME order/result pair from being applied twice — a replayed ingest over an
|
|
297
|
+
// already-applied result, never a fresh attempt (a real re-attempt earns its own order_id,
|
|
298
|
+
// `r<N>-a<N+1>`). Measured: replaying one identical escalated WorkResult doubled its
|
|
299
|
+
// `[ESCALATE]` line in this ledger, and both GATE H's census and the ship report's "Discovered,
|
|
300
|
+
// not built" section count this file's open entries — so a replay silently inflated the count
|
|
301
|
+
// for a WorkResult that ran exactly once. A block heading names its order verbatim, so a second
|
|
302
|
+
// ingest of the same order recognises its own prior write and skips the append rather than
|
|
303
|
+
// duplicating it.
|
|
304
|
+
const existingLedger = readFileSync(ledger, "utf8");
|
|
305
|
+
/**
|
|
306
|
+
* The ledger heading one order's block is filed under — the idempotency key: a second ingest
|
|
307
|
+
* of the SAME order recognises its own prior write by this string, verbatim.
|
|
308
|
+
* @param {string} oid - `result.order_id` this block belongs to.
|
|
309
|
+
* @returns {string} The heading prefix (open-ended — the date suffix varies, the order id does not).
|
|
310
|
+
*/
|
|
311
|
+
const headingFor = (oid) => `## Discovered — ${oid} (`;
|
|
312
|
+
const alreadyLogged = existingLedger.includes(headingFor(result.order_id));
|
|
313
|
+
if (alreadyLogged) {
|
|
314
|
+
summary.discoveries_appended = 0;
|
|
315
|
+
summary.escalations_appended = 0;
|
|
316
|
+
} else {
|
|
317
|
+
const discoveryLines = (result.discoveries || []).map((d) => {
|
|
318
|
+
const tags = [d.lens ? `[lens:${d.lens}]` : "", d.severity_hint ? `severity-hint: ${d.severity_hint}` : "", d.test_gap ? `test-gap: ${d.test_gap}` : "", d.contradicts ? `contradicts: ${d.contradicts}` : "", d.traces_to?.length ? `traces_to: ${d.traces_to.join(", ")}` : ""].filter(Boolean);
|
|
319
|
+
return `${d.marker} ${d.lens ? tags[0] + " " : ""}${d.line}${d.repro ? `\n repro: ${d.repro}` : ""}${tags.slice(d.lens ? 1 : 0).map((t) => `\n ${t}`).join("")}`;
|
|
320
|
+
});
|
|
321
|
+
// `+` (candidate work), matching the schema's own reading of that marker — an ESCALATE is
|
|
322
|
+
// exactly that: work a worker could not safely do without a decision only the census can make.
|
|
323
|
+
const escalateLines = escalations.map((d) => `+ [ESCALATE] ${d}`);
|
|
324
|
+
const lines = [...discoveryLines, ...escalateLines].join("\n");
|
|
325
|
+
appendFileSync(ledger, `\n${headingFor(result.order_id)}${today()})\n${lines}\n`);
|
|
326
|
+
summary.discoveries_appended = (result.discoveries || []).length;
|
|
327
|
+
summary.escalations_appended = escalations.length;
|
|
328
|
+
}
|
|
288
329
|
}
|
|
289
330
|
|
|
290
331
|
// 4. Verdict bookkeeping (old evaluator B.0/B.2/B.2b) — judge returns data, ingest writes.
|
|
@@ -657,5 +698,5 @@ export async function cli(rawArgv) {
|
|
|
657
698
|
}
|
|
658
699
|
}
|
|
659
700
|
|
|
660
|
-
console.log(`✅ ingested ${result.order_id} — tasks: [${s.tasks_updated.join(", ")}] · ACs ticked: ${s.acs_ticked} · unblocked: [${s.unblocked.join(", ")}] · discoveries: ${s.discoveries_appended} · verdict lines: ${s.verdict_lines} · refuted un-ticked: ${s.refuted_unticked}`);
|
|
701
|
+
console.log(`✅ ingested ${result.order_id} — tasks: [${s.tasks_updated.join(", ")}] · ACs ticked: ${s.acs_ticked} · unblocked: [${s.unblocked.join(", ")}] · discoveries: ${s.discoveries_appended} · escalations: ${s.escalations_appended} · verdict lines: ${s.verdict_lines} · refuted un-ticked: ${s.refuted_unticked}`);
|
|
661
702
|
}
|
package/kernel/reduce/ship.mjs
CHANGED
|
@@ -31,12 +31,13 @@ import { join, dirname } from "node:path";
|
|
|
31
31
|
import { runArgs } from "../lib/argv.mjs";
|
|
32
32
|
import {
|
|
33
33
|
report as reportPath, tasksDir, verdictsDir, trials, evaluationDir, qaDir,
|
|
34
|
-
roundLedger, discoveryLedger, receipt as receiptPath, harnessRun, relShared,
|
|
34
|
+
roundLedger, discoveryLedger, receipt as receiptPath, harnessRun, relShared,
|
|
35
35
|
activeOrder,
|
|
36
36
|
} from "../lib/paths.mjs";
|
|
37
37
|
import { readTrials } from "../verify/t0.mjs";
|
|
38
38
|
import { ratchetReport } from "../probe/stats.mjs";
|
|
39
39
|
import { projectRequirements, summaryLine } from "../probe/requirements.mjs";
|
|
40
|
+
import { deriveRounds } from "../probe/rounds.mjs";
|
|
40
41
|
import { collectDiff, scanDiff, summarize } from "./leftovers.mjs";
|
|
41
42
|
|
|
42
43
|
/** @returns {string} Today as `YYYY-MM-DD` (UTC). */
|
|
@@ -167,13 +168,13 @@ export function section(md, heading) {
|
|
|
167
168
|
*/
|
|
168
169
|
export function buildReport(facts) {
|
|
169
170
|
const {
|
|
170
|
-
slug, at, verdict, qa, rounds, board, t0, artifacts, ratchet,
|
|
171
|
+
slug, at, verdict, qa, rounds, roundsJudged, board, t0, artifacts, ratchet,
|
|
171
172
|
evalCriteria, evalBugs, qaFindings, decisions, discovered, intakeSha, leftovers, requirements,
|
|
172
173
|
} = facts;
|
|
173
174
|
|
|
174
175
|
const L = [];
|
|
175
176
|
L.push("---", "type: ship-report", `feature: ${slug}`, `date: ${at}`,
|
|
176
|
-
`verdict: ${verdict}`, `rounds_used: ${rounds ?? "~"}`, `qa: ${qa}`,
|
|
177
|
+
`verdict: ${verdict}`, `rounds_used: ${rounds ?? "~"}`, `rounds_judged: ${roundsJudged ?? "~"}`, `qa: ${qa}`,
|
|
177
178
|
`intake_sha256: ${intakeSha ?? "~"}`, "---", "");
|
|
178
179
|
L.push(`# ${slug} — ship report`, "");
|
|
179
180
|
L.push("Frozen at GATE L4. Every figure below is derived from run artifacts on disk — the trial",
|
|
@@ -183,6 +184,10 @@ export function buildReport(facts) {
|
|
|
183
184
|
L.push("| | |", "|---|---|");
|
|
184
185
|
L.push(`| Verdict | **${verdict}** |`);
|
|
185
186
|
L.push(`| Rounds used | ${rounds ?? "—"} |`);
|
|
187
|
+
// A round built and a round judged are different facts — a round can die before EVAL ever sees
|
|
188
|
+
// it, so this row is its own line rather than folded into "Rounds used" above. Omitted when EVAL
|
|
189
|
+
// never ran at all, the same way the sections below it are.
|
|
190
|
+
if (roundsJudged != null) L.push(`| Rounds judged | ${roundsJudged} |`);
|
|
186
191
|
L.push(`| Board | ${board.done}/${board.total} tasks done |`);
|
|
187
192
|
L.push(`| T0 artifacts | ${artifacts} |`);
|
|
188
193
|
L.push(`| QA | ${qa} |`);
|
|
@@ -291,32 +296,6 @@ export function buildReport(facts) {
|
|
|
291
296
|
return L.join("\n");
|
|
292
297
|
}
|
|
293
298
|
|
|
294
|
-
/**
|
|
295
|
-
* How many BUILD/EVAL rounds this run actually completed.
|
|
296
|
-
*
|
|
297
|
-
* `harness-run.md`'s `rounds_used` frontmatter field is written ONCE, at GATE L0.1 (`init run`),
|
|
298
|
-
* as `0` — nothing in the round loop ever rewrites it as rounds complete. The orchestrator's own
|
|
299
|
-
* `RunReturn` carries the real count (`shapeup-run.js`'s `rounds_used: round`), but that value
|
|
300
|
-
* never reaches `reduce ship`, so every real run's report printed "Rounds used | 0" beside its own
|
|
301
|
-
* `results/evaluate-r1.json` — a claim the frontmatter makes about the run, contradicted by the
|
|
302
|
-
* artifact sitting next to it. Derived instead, the same way `probe resume`'s `eval_rounds_done`
|
|
303
|
-
* already does: the highest `evaluate-r<N>.json` result on disk. Falls back to the frontmatter
|
|
304
|
-
* value only when no EVAL round ever ran (the `--tiny` lane has no round concept at all), so a
|
|
305
|
-
* bare or tiny run's reporting is unchanged.
|
|
306
|
-
* @param {string} cwd - Project root.
|
|
307
|
-
* @param {string} slug - Feature slug.
|
|
308
|
-
* @param {(string|undefined)} fallback - `run.rounds_used` from the frontmatter.
|
|
309
|
-
* @returns {(number|string|undefined)} The derived round count, or the fallback.
|
|
310
|
-
*/
|
|
311
|
-
function roundsUsed(cwd, slug, fallback) {
|
|
312
|
-
const dir = resultsDir(cwd, slug);
|
|
313
|
-
const done = (existsSync(dir) ? readdirSync(dir) : [])
|
|
314
|
-
.map((f) => f.match(/^evaluate-r(\d+)\.json$/))
|
|
315
|
-
.filter(Boolean)
|
|
316
|
-
.map((m) => Number(m[1]));
|
|
317
|
-
return done.length ? Math.max(...done) : fallback;
|
|
318
|
-
}
|
|
319
|
-
|
|
320
299
|
/**
|
|
321
300
|
* Gather every fact from disk and render the report.
|
|
322
301
|
* @param {{cwd:string, slug:string, verdict?:string, qa?:string}} opts - Inputs.
|
|
@@ -331,12 +310,18 @@ export function generate({ cwd, slug, verdict, qa }) {
|
|
|
331
310
|
const ledger = readIf(roundLedger(cwd, slug));
|
|
332
311
|
const discovery = readIf(discoveryLedger(cwd, slug));
|
|
333
312
|
|
|
313
|
+
// Two numbers, not one: `rounds` (built — an order, a T0 verdict, a build-gate artifact or an
|
|
314
|
+
// EVAL result) and `roundsJudged` (EVAL actually returned a verdict for), kept separate so a run
|
|
315
|
+
// whose later rounds never reached EVAL still reports the rounds it built.
|
|
316
|
+
const derivedRounds = deriveRounds(cwd, slug, run.rounds_used);
|
|
317
|
+
|
|
334
318
|
const facts = {
|
|
335
319
|
slug,
|
|
336
320
|
at: today(),
|
|
337
321
|
verdict: verdict || run.final_verdict || "not-evaluated",
|
|
338
322
|
qa: qa || (huntReport ? "run" : "skipped"),
|
|
339
|
-
rounds:
|
|
323
|
+
rounds: derivedRounds.rounds_used,
|
|
324
|
+
roundsJudged: derivedRounds.rounds_judged,
|
|
340
325
|
intakeSha: receipt.intake_sha256,
|
|
341
326
|
board: boardCensus(cwd, slug),
|
|
342
327
|
t0: t0Summary(cwd, slug),
|
|
@@ -25,6 +25,7 @@ import { resolve, join } from "node:path";
|
|
|
25
25
|
import { validate } from "../verify/envelope.mjs";
|
|
26
26
|
import { runArgs } from "../lib/argv.mjs";
|
|
27
27
|
import { localDir, localRoot, relLocal, globLocal, runSnapshot as runSnapshotPath } from "../lib/paths.mjs";
|
|
28
|
+
import { deriveRounds } from "../probe/rounds.mjs";
|
|
28
29
|
|
|
29
30
|
/**
|
|
30
31
|
* Read a JSON file, tolerating absence/parse errors.
|
|
@@ -126,8 +127,28 @@ export function deriveSnapshot(cwd) {
|
|
|
126
127
|
if (existsSync(runPath)) {
|
|
127
128
|
try {
|
|
128
129
|
const fm = frontmatter(readFileSync(runPath, "utf8"));
|
|
129
|
-
|
|
130
|
-
|
|
130
|
+
// The terminal allowlist mirrors kernel/probe/resume.mjs's TERMINAL_STATUSES, not just the two
|
|
131
|
+
// members it used to carry — a schema/allowlist that disagrees with RUN_STATUSES is silent on
|
|
132
|
+
// both sides here (findRun only ever surfaces a MID_RUN run today, so this whole branch is
|
|
133
|
+
// unreachable in practice), but is exactly the divergence class this file's own imports exist
|
|
134
|
+
// to close elsewhere (deriveRounds, above). "aborted" is a real RUN_STATUSES member.
|
|
135
|
+
if (MID_RUN.has(fm.status) || ["shipped", "escalated", "aborted"].includes(fm.status)) snapshot.status = fm.status;
|
|
136
|
+
// MIGRATED to the same mechanical derivation `reduce ship` and `report export` already use
|
|
137
|
+
// (kernel/probe/rounds.mjs), rather than reading `harness-run.md`'s `rounds_used` literally.
|
|
138
|
+
// That frontmatter line is written ONCE, as 0, by `init run`, and nothing in the round loop
|
|
139
|
+
// ever rewrites it — so the third mechanical reader of "how many rounds did this run build"
|
|
140
|
+
// was the one still reporting the pre-fix number. Measured on a two-round, no-EVAL fixture:
|
|
141
|
+
// this line alone reported `rounds_used: 0` beside its own `round: 2` a few lines below,
|
|
142
|
+
// the exact disagreement `deriveRounds` exists to close. `fm.rounds_used` still travels in as
|
|
143
|
+
// the fallback for a run neither this fix nor deriveRounds can see evidence for (a `--tiny`
|
|
144
|
+
// lane, or a run from before any of these artifacts existed).
|
|
145
|
+
const derivedRounds = deriveRounds(cwd, run.slug, fm.rounds_used);
|
|
146
|
+
if (Number.isFinite(Number(derivedRounds.rounds_used))) snapshot.rounds_used = Number(derivedRounds.rounds_used);
|
|
147
|
+
// Kept as its OWN field, never folded into rounds_used — a round built is not a round judged,
|
|
148
|
+
// and collapsing the two into one number is exactly the ambiguity this migration removes.
|
|
149
|
+
// null (no EVAL result on disk yet) is a real answer and is not written at all, the same
|
|
150
|
+
// optional-field discipline every other snapshot field here follows.
|
|
151
|
+
if (Number.isFinite(derivedRounds.rounds_judged)) snapshot.rounds_judged = derivedRounds.rounds_judged;
|
|
131
152
|
if (/^\d+$/.test(fm.max_rounds || "")) snapshot.max_rounds = Number(fm.max_rounds);
|
|
132
153
|
if (fm.auto_level) snapshot.auto_level = fm.auto_level;
|
|
133
154
|
if (fm.spec_folder) snapshot.spec_folder = fm.spec_folder;
|
package/kernel/report/export.mjs
CHANGED
|
@@ -47,10 +47,11 @@ import { runArgs } from "../lib/argv.mjs";
|
|
|
47
47
|
import { splitFrontmatter } from "../lib/contract.mjs";
|
|
48
48
|
import { runIdFromReceipt, readReceipt } from "../lib/paths.mjs";
|
|
49
49
|
import { TABLES, runRow, dispatchFacts } from "./facts.mjs";
|
|
50
|
+
import { deriveRounds } from "../probe/rounds.mjs";
|
|
50
51
|
import {
|
|
51
52
|
localDir, activeScope, receipt as receiptPath, harnessRun, ordersDir, resultsDir,
|
|
52
53
|
trials as trialsPath, verdictsDir, evaluationDir, decisions as decisionsPath,
|
|
53
|
-
exportsDir, exportRunDir,
|
|
54
|
+
gates as gatesPath, roundBuildDir, exportsDir, exportRunDir,
|
|
54
55
|
} from "../lib/paths.mjs";
|
|
55
56
|
|
|
56
57
|
export const EXPORT_SCHEMA_VERSION = 1;
|
|
@@ -164,6 +165,51 @@ function criterionRows(dir, runId, t) {
|
|
|
164
165
|
return out;
|
|
165
166
|
}
|
|
166
167
|
|
|
168
|
+
/**
|
|
169
|
+
* Flatten one gate-crossing ledger row (`gates.jsonl`, `kernel/gate.mjs`'s sole writer) into a
|
|
170
|
+
* flat `gate_decision` fact row. The row on disk already carries exactly these fields
|
|
171
|
+
* (see `appendGateLedger`), so this is a pass-through with a stamped `run_id` fallback rather than
|
|
172
|
+
* a re-derivation: two readers of "what did this gate decide" must not compute the answer twice.
|
|
173
|
+
* @param {object} g - One parsed line of `gates.jsonl`.
|
|
174
|
+
* @param {(string|null)} runId - Run key for a row written before it carried its own.
|
|
175
|
+
* @returns {object} A flat `gate_decision` row.
|
|
176
|
+
*/
|
|
177
|
+
function gateDecisionRow(g, runId) {
|
|
178
|
+
return {
|
|
179
|
+
run_id: g?.run_id ?? runId ?? null,
|
|
180
|
+
gate: g?.gate ?? null,
|
|
181
|
+
decision: g?.decision ?? null,
|
|
182
|
+
status: g?.status ?? null,
|
|
183
|
+
source: g?.source ?? null,
|
|
184
|
+
round: g?.round ?? null,
|
|
185
|
+
has_note: !!(g?.note && String(g.note).trim()),
|
|
186
|
+
};
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* Flatten one round build-gate artifact (`kernel/verify/build.mjs`'s `writeRoundBuild`) into a flat
|
|
191
|
+
* `build_gate` fact row. The gate ends a round exactly as EVAL does (AGENTS.md's round
|
|
192
|
+
* build gate ⚙), and until now had no fact table at all.
|
|
193
|
+
* @param {object} a - A parsed round-build artifact.
|
|
194
|
+
* @param {(string|null)} runId - Run key for an artifact written before it carried its own.
|
|
195
|
+
* @returns {object} A flat `build_gate` row.
|
|
196
|
+
*/
|
|
197
|
+
function buildGateRow(a, runId) {
|
|
198
|
+
const steps = Array.isArray(a?.steps) ? a.steps : [];
|
|
199
|
+
return {
|
|
200
|
+
run_id: a?.run_id ?? runId ?? null,
|
|
201
|
+
round: a?.round ?? null,
|
|
202
|
+
trial: a?.trial ?? null,
|
|
203
|
+
at: a?.at ?? null,
|
|
204
|
+
overall: a?.overall ?? null,
|
|
205
|
+
archetype: a?.archetype ?? null,
|
|
206
|
+
steps_total: steps.length,
|
|
207
|
+
steps_failed: steps.filter((s) => !s?.skipped && s?.pass === false).length,
|
|
208
|
+
warnings: Array.isArray(a?.warnings) ? a.warnings.length : 0,
|
|
209
|
+
discovered_tasks: Array.isArray(a?.discovered_tasks) ? a.discovered_tasks.length : 0,
|
|
210
|
+
};
|
|
211
|
+
}
|
|
212
|
+
|
|
167
213
|
// ---------------------------------------------------------------------------
|
|
168
214
|
// The export itself
|
|
169
215
|
// ---------------------------------------------------------------------------
|
|
@@ -190,7 +236,10 @@ export function collectRun(cwd, slug) {
|
|
|
190
236
|
const results = readJsonDir(resultsDir(cwd, slug), t);
|
|
191
237
|
|
|
192
238
|
const { dispatch, ac_result, discovery, file_touched } = dispatchFacts({ orders, results, runId });
|
|
193
|
-
|
|
239
|
+
// Computed here, once, from the same trace this whole function reads, and handed to
|
|
240
|
+
// runRow rather than re-derived by it: runRow stays pure (no I/O), this function already has cwd.
|
|
241
|
+
const rounds = deriveRounds(cwd, slug, ledger.rounds_used);
|
|
242
|
+
const run = runRow({ receipt: rec, ledger, runId, rounds });
|
|
194
243
|
|
|
195
244
|
// Hook decisions are checkout-wide, so they are FILTERED to this run rather than read from a
|
|
196
245
|
// per-run file. Rows with a null key belong to no run (a hook that fired outside one) and are
|
|
@@ -208,6 +257,9 @@ export function collectRun(cwd, slug) {
|
|
|
208
257
|
t0_verdict: readJsonDir(verdictsDir(cwd, slug), t).map((a) => t0Row(a, runId)),
|
|
209
258
|
criterion_verdict: criterionRows(evaluationDir(cwd, slug), runId, t),
|
|
210
259
|
hook_decision,
|
|
260
|
+
// The decision that crossed each gate, and the round build gate's own artifact.
|
|
261
|
+
gate_decision: readJsonl(gatesPath(cwd, slug), t).map((g) => gateDecisionRow(g, runId)),
|
|
262
|
+
build_gate: readJsonDir(roundBuildDir(cwd, slug), t).map((a) => buildGateRow(a, runId)),
|
|
211
263
|
},
|
|
212
264
|
defects: { records_skipped: t.skipped },
|
|
213
265
|
};
|
package/kernel/report/facts.mjs
CHANGED
|
@@ -27,6 +27,11 @@
|
|
|
27
27
|
export const TABLES = [
|
|
28
28
|
"run", "dispatch", "ac_result", "discovery", "file_touched",
|
|
29
29
|
"trial", "t0_verdict", "criterion_verdict", "hook_decision",
|
|
30
|
+
// The decision that shipped a run (or any other gate) had a ledger row (`gates.jsonl`)
|
|
31
|
+
// and no table — a reader had to open the LOCAL trace itself, which the export exists so nobody
|
|
32
|
+
// has to. `build_gate` is the round build gate's own artifact (kernel/verify/build.mjs), on the
|
|
33
|
+
// same terms: it ends a round exactly as EVAL does, and had no table either.
|
|
34
|
+
"gate_decision", "build_gate",
|
|
30
35
|
];
|
|
31
36
|
|
|
32
37
|
/** Coerce anything to a finite number, or null. Keeps `0` and rejects `NaN`/`""`/undefined. */
|
|
@@ -76,9 +81,14 @@ export function parseOrderStem(orderId) {
|
|
|
76
81
|
* @param {(object|null)} o.receipt - Parsed `receipt.json`.
|
|
77
82
|
* @param {(object|null)} [o.ledger] - Parsed `harness-run.md` frontmatter (a flat scalar map).
|
|
78
83
|
* @param {(string|null)} [o.runId] - The run key, when already resolved.
|
|
84
|
+
* @param {({rounds_used:*, rounds_judged:(number|null)}|null)} [o.rounds] - The two-number
|
|
85
|
+
* derivation (`probe/rounds.mjs`'s `deriveRounds`), computed by the caller because it needs the
|
|
86
|
+
* filesystem and this function stays pure. Falls back to the ledger's own (unreliable — see
|
|
87
|
+
* `deriveRounds`) `rounds_used` line when the caller has not derived one, so an existing caller
|
|
88
|
+
* is unaffected rather than broken.
|
|
79
89
|
* @returns {(object|null)} The run row, or null when there is no receipt to describe.
|
|
80
90
|
*/
|
|
81
|
-
export function runRow({ receipt, ledger = null, runId = null }) {
|
|
91
|
+
export function runRow({ receipt, ledger = null, runId = null, rounds = null }) {
|
|
82
92
|
if (!receipt) return null;
|
|
83
93
|
const c = receipt.config || {};
|
|
84
94
|
const fm = ledger || {};
|
|
@@ -87,6 +97,9 @@ export function runRow({ receipt, ledger = null, runId = null }) {
|
|
|
87
97
|
slug: receipt.slug ?? null,
|
|
88
98
|
started_at: receipt.started_at ?? null,
|
|
89
99
|
closed_at: fm.closed_at && fm.closed_at !== "~" ? fm.closed_at : null,
|
|
100
|
+
// Why the run ended at a terminal status, written by `probe resume --close` alongside
|
|
101
|
+
// `closed_at` — the two facts a trace needs to tell a live run from a dead one apart.
|
|
102
|
+
close_cause: fm.close_cause && fm.close_cause !== "~" ? fm.close_cause : null,
|
|
90
103
|
intake_sha256: receipt.intake_sha256 ?? null,
|
|
91
104
|
intake_chars: num(receipt.intake_chars),
|
|
92
105
|
intake_lines: num(receipt.intake_lines),
|
|
@@ -101,8 +114,17 @@ export function runRow({ receipt, ledger = null, runId = null }) {
|
|
|
101
114
|
// Copied from the ledger, never re-derived: the run's own status line is the harness's answer,
|
|
102
115
|
// and a read plane that recomputed it would be asserting a second one.
|
|
103
116
|
status: fm.status ?? null,
|
|
117
|
+
// The terminal status `closeRun` alone writes, carried BESIDE `status` rather than instead of
|
|
118
|
+
// it. `status:` is ordinary phase traffic and a later phase may move it, so an export that
|
|
119
|
+
// carried only that line could show a run wearing another close's cause. These two disagreeing
|
|
120
|
+
// is itself the fact worth exporting: it says the ledger was written after the close.
|
|
121
|
+
closed_status: fm.closed_status && fm.closed_status !== "~" ? fm.closed_status : null,
|
|
104
122
|
final_verdict: fm.final_verdict && fm.final_verdict !== "~" ? fm.final_verdict : null,
|
|
105
|
-
|
|
123
|
+
// Two fields, not one: `rounds_used` is the highest round carrying ANY build evidence,
|
|
124
|
+
// `rounds_judged` the highest round EVAL actually returned a verdict for. A caller that has not
|
|
125
|
+
// derived `rounds` falls back to the ledger's own (pre-fix, unreliable) line, non-regression.
|
|
126
|
+
rounds_used: rounds ? num(Number(rounds.rounds_used)) : num(Number(fm.rounds_used)),
|
|
127
|
+
rounds_judged: rounds ? num(rounds.rounds_judged) : null,
|
|
106
128
|
};
|
|
107
129
|
}
|
|
108
130
|
|
|
@@ -668,14 +668,14 @@
|
|
|
668
668
|
"string",
|
|
669
669
|
"null"
|
|
670
670
|
],
|
|
671
|
-
"description": "Source file of the failure; null/absent when the log line
|
|
671
|
+
"description": "Source file of the failure; null/absent when the log line named no file at all (raw triple — never invented)."
|
|
672
672
|
},
|
|
673
673
|
"line": {
|
|
674
674
|
"type": [
|
|
675
675
|
"integer",
|
|
676
676
|
"null"
|
|
677
677
|
],
|
|
678
|
-
"description": "Line of the failure; null/absent when the
|
|
678
|
+
"description": "Line of the failure; null/absent when the diagnostic carried no line number — independently of `file`, since a diagnostic can name a file with no line (a resource-compiler error, for example). Never invented to satisfy a shape."
|
|
679
679
|
},
|
|
680
680
|
"core_message": {
|
|
681
681
|
"type": "string",
|
|
@@ -1891,9 +1891,10 @@
|
|
|
1891
1891
|
"building",
|
|
1892
1892
|
"evaluating",
|
|
1893
1893
|
"shipped",
|
|
1894
|
-
"escalated"
|
|
1894
|
+
"escalated",
|
|
1895
|
+
"aborted"
|
|
1895
1896
|
],
|
|
1896
|
-
"description": "Mirrors harness-run.md frontmatter status."
|
|
1897
|
+
"description": "Mirrors harness-run.md frontmatter status — kernel/probe/resume.mjs's RUN_STATUSES is the source enum this one must not drift from."
|
|
1897
1898
|
},
|
|
1898
1899
|
"round": {
|
|
1899
1900
|
"type": "integer",
|
|
@@ -1904,7 +1905,12 @@
|
|
|
1904
1905
|
"description": "From the latest t0/verdicts/r<N>-a<M>.json filename."
|
|
1905
1906
|
},
|
|
1906
1907
|
"rounds_used": {
|
|
1907
|
-
"type": "integer"
|
|
1908
|
+
"type": "integer",
|
|
1909
|
+
"description": "Highest round carrying any build evidence (an order, a T0 verdict, a round build-gate artifact, or an EVAL result) — kernel/probe/rounds.mjs's deriveRounds(), the same derivation the ship report and the export use. Falls back to harness-run.md's literal frontmatter value only when no such evidence exists on disk."
|
|
1910
|
+
},
|
|
1911
|
+
"rounds_judged": {
|
|
1912
|
+
"type": "integer",
|
|
1913
|
+
"description": "Highest round EVAL actually returned a verdict for (an evaluate-r<N>.json result on disk) — its own field, never folded into rounds_used: a round built is not a round judged. Omitted when no round has been judged yet."
|
|
1908
1914
|
},
|
|
1909
1915
|
"max_rounds": {
|
|
1910
1916
|
"type": "integer"
|
|
@@ -2704,10 +2710,10 @@
|
|
|
2704
2710
|
}
|
|
2705
2711
|
},
|
|
2706
2712
|
"RunArgs": {
|
|
2707
|
-
"description": "C1 — the launch half of the workflow's only conversation.
|
|
2713
|
+
"description": "C1 — the launch half of the workflow's only conversation. Resolved ONCE at GATE L0 from harness init run output + the L0.8 model matrix + budgets, then built by `harness init run-args`, which writes it to .shapeup/<slug>/run-args.json AND prints it — tech-lead passes that printed value to the harness run launch as one JSON literal, never a second, hand-typed copy of it. The workflow cannot ask follow-ups and cannot read config files itself, so everything a run will ever need travels in this one record. A workflow script validates its own subset of this shape in code (no runtime schema check at the C1 boundary itself); this entry is the central-registry definition the workflow script, `harness init run-args` and the tech-lead skill all read as the one true shape.",
|
|
2708
2714
|
"x-tier": "EMBEDDED",
|
|
2709
|
-
"x-location": ".shapeup/<slug>/run-args.json — written fresh
|
|
2710
|
-
"x-writer": "tech-lead
|
|
2715
|
+
"x-location": ".shapeup/<slug>/run-args.json — written fresh on every launch and relaunch by `harness init run-args`; the workflow receives it as its args and never reads other config",
|
|
2716
|
+
"x-writer": "harness init run-args (kernel), invoked by tech-lead at GATE L0 on every launch AND every relaunch after a paused gate",
|
|
2711
2717
|
"x-readers": "the Workflow runtime (shapeup-run, and shapeup-run's own inner round dispatch)",
|
|
2712
2718
|
"x-not-here": "Run config the LEDGER already carries does NOT get a second home in RunArgs — eval_dimensions, lens, spec_folder, stack, run_cmd, app_url are read off harness-run.md frontmatter by resume-state on every launch AND every relaunch, so a copy here would be a second source that can disagree with the first. RunArgs carries what a workflow cannot derive from disk (identity, budgets, the model matrix, pluginRoot, startedAt) plus noEval, which no frontmatter line holds.",
|
|
2713
2719
|
"type": "object",
|
|
@@ -2759,16 +2765,13 @@
|
|
|
2759
2765
|
},
|
|
2760
2766
|
"budgets": {
|
|
2761
2767
|
"type": "object",
|
|
2762
|
-
"description": "The
|
|
2768
|
+
"description": "The two RunArgs-level circuit breakers (AGENTS.md) — outer round_budget, inner attempt_budget. The third, opt-in DEADLINE breaker (the wall-clock budget) is NOT a RunArgs field: it is typed once, at `harness init run --wall-clock-budget`, and lands in the run receipt's `wall_clock_budget_s` — `harness verify budget` reads that receipt field directly and never sees this launch's RunArgs at all, so it has no member here to declare.",
|
|
2763
2769
|
"properties": {
|
|
2764
2770
|
"maxRounds": {
|
|
2765
2771
|
"type": "integer"
|
|
2766
2772
|
},
|
|
2767
2773
|
"attemptBudget": {
|
|
2768
2774
|
"type": "integer"
|
|
2769
|
-
},
|
|
2770
|
-
"wallClockS": {
|
|
2771
|
-
"type": "integer"
|
|
2772
2775
|
}
|
|
2773
2776
|
}
|
|
2774
2777
|
},
|
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
// #/$defs/Name — a definition in the SAME schema document
|
|
13
13
|
// domain.schema.json#/$defs/Name — a definition in a SIBLING file (the central domain
|
|
14
14
|
// registry; resolved against the schema's own dir,
|
|
15
|
-
// falling back to
|
|
15
|
+
// falling back to kernel/schemas/)
|
|
16
16
|
//
|
|
17
17
|
// Usage (CLI): node kernel/harness.mjs verify envelope <envelope.json> <schema.json>
|
|
18
18
|
// exit 0 = valid, 1 = invalid (errors printed one per line)
|
|
@@ -28,7 +28,7 @@ import { runArgs } from "../lib/argv.mjs";
|
|
|
28
28
|
import { runHook, readStdin, settle } from "../../hooks/lib/decision.mjs";
|
|
29
29
|
|
|
30
30
|
const HERE = dirname(fileURLToPath(import.meta.url));
|
|
31
|
-
export const SCHEMAS_DIR = resolve(HERE, "
|
|
31
|
+
export const SCHEMAS_DIR = resolve(HERE, "../schemas");
|
|
32
32
|
|
|
33
33
|
/**
|
|
34
34
|
* Validate a value against the JSON-Schema subset the envelope schemas use (type, required,
|
package/kernel/verify/skills.mjs
CHANGED
|
@@ -45,7 +45,7 @@ export const PLUGIN_ROOT = resolve(HERE, "../..");
|
|
|
45
45
|
* its own domain registry has a broken installation, which is the very thing being checked.
|
|
46
46
|
*/
|
|
47
47
|
export function roster(root = PLUGIN_ROOT) {
|
|
48
|
-
const schemaPath = join(root, "
|
|
48
|
+
const schemaPath = join(root, "kernel/schemas/domain.schema.json");
|
|
49
49
|
const schema = JSON.parse(readFileSync(schemaPath, "utf8"));
|
|
50
50
|
const names = schema?.$defs?.WorkerName?.enum;
|
|
51
51
|
if (!Array.isArray(names) || !names.length) {
|
package/package.json
CHANGED
|
@@ -121,7 +121,7 @@ covered AC is a requirement the run can be measured against.
|
|
|
121
121
|
|---|---|---|
|
|
122
122
|
| `reconcile` | Verify `ledger.feature == payload.feature` (mismatch → STOP). Map each `[+]` Keep item → its owning UC; new task continues numbering (never renumber); `~`/Cut → synthesis "Hammered Out" row, no file. A Keep item asserting a new invariant → APPEND `[INV-NN]` + TS-INV row to that UC (append-only sections in your substrate). A new actor/action with no UC → `status: "escalated"` + a `deviations[]` spec-ambiguity entry: spawning a UC mid-cycle is silent re-shaping, the PO decides. Finish with board-derive (appetite overflow → report) + spec-lint | re-run phases 1–5; edit UC Steps; resolve the appetite HAMMER yourself |
|
|
123
123
|
| `retrofit-surface` | Append `## Test Surface` (derived rows only, after Error Cases) to each UC of a pre-surface spec; an all-sources-empty UC gets the explicit empty-sources line | touch anything else — append-only substrate |
|
|
124
|
-
| `coverage` | Extract **atomic** customer requirement clauses from `payload.requirements` (default: the pitch) and write the SHARED `shapeup/<slug>/requirements.md` registry: one `\| REQ-id \| clause (verbatim) \| source \| status \| note \|` row per clause. Split compound sentences into one testable clause each — a clause lost *inside* a bigger sentence is a requirement nothing can be traced to. **Assign REQ-ids ONCE and freeze them** (they behave like scope_id, never TASK-NNN — every `covers:` link rots otherwise): re-running, append new clauses with fresh ids, mark a removed clause `CUT (PO-approved)`, never renumber or delete. Status starts `covered` (a live requirement); only the PO sets `CUT`. The REQ source itself is frozen — the registry is a separate derived file. **Numbering.** A source clause already carrying an `R<n>` keeps its number — `R12` → `REQ-12` — and its `source` cell records where it came from verbatim (`shaping.md R12`), because that cell is the only thing that survives a re-run. A clause with no R-id takes the next free number ABOVE the highest `R<n>` in the source, so it can never collide with one added later. Splitting a compound clause keeps `REQ-12` for the first atomic part and records `shaping.md R12 (split 2/3)` for the rest — a requirement graded in parts is why splitting matters at all. On a re-run, match an existing id by its frozen `source` cell and clause text, **never** by re-deriving the number from the source's current order | edit the REQ source; renumber existing REQ-ids; delete a dropped clause instead of marking it CUT; invent a requirement not in the source; re-point an existing REQ-id because the source's R-numbers shifted |
|
|
124
|
+
| `coverage` | Extract **atomic** customer requirement clauses from `payload.requirements` (default: the pitch) and write the SHARED `shapeup/<slug>/requirements.md` registry: one `\| REQ-id \| clause (verbatim) \| source \| status \| note \|` row per clause. **The `source` cell, and any other prose in this file, cites the committed pitch or shaping doc — never the gitignored run-tier path** (`.shapeup/<slug>/intake.md`, or any other `.shapeup/` path): spec-lint's `TIER-DIRECTION` rule reds a committed file naming that path in ANY form, a bare path in a sentence exactly as much as a `[[tasks/...]]` wikilink, because it dangles on every other clone. When the intake has no committed original to name, describe the run tier without a path (`"the pitch staged for this run"`) rather than citing where it actually lives. Split compound sentences into one testable clause each — a clause lost *inside* a bigger sentence is a requirement nothing can be traced to. **Assign REQ-ids ONCE and freeze them** (they behave like scope_id, never TASK-NNN — every `covers:` link rots otherwise): re-running, append new clauses with fresh ids, mark a removed clause `CUT (PO-approved)`, never renumber or delete. Status starts `covered` (a live requirement); only the PO sets `CUT`. The REQ source itself is frozen — the registry is a separate derived file. **Numbering.** A source clause already carrying an `R<n>` keeps its number — `R12` → `REQ-12` — and its `source` cell records where it came from verbatim (`shaping.md R12`), because that cell is the only thing that survives a re-run. A clause with no R-id takes the next free number ABOVE the highest `R<n>` in the source, so it can never collide with one added later. Splitting a compound clause keeps `REQ-12` for the first atomic part and records `shaping.md R12 (split 2/3)` for the rest — a requirement graded in parts is why splitting matters at all. On a re-run, match an existing id by its frozen `source` cell and clause text, **never** by re-deriving the number from the source's current order | edit the REQ source; renumber existing REQ-ids; delete a dropped clause instead of marking it CUT; invent a requirement not in the source; re-point an existing REQ-id because the source's R-numbers shifted |
|
|
125
125
|
---
|
|
126
126
|
|
|
127
127
|
## Anti-rationalization table
|
|
@@ -275,6 +275,12 @@ Always use wikilinks (double brackets), never relative paths like `../domain-mod
|
|
|
275
275
|
`.shapeup/` is gitignored, so a committed task link dangles on every fresh clone.
|
|
276
276
|
spec-lint flags it as a red `TIER-DIRECTION` finding. Coverage views (synthesis
|
|
277
277
|
traceability) record derived counts/status, not task ids.
|
|
278
|
+
- **The rule is not only about wikilinks.** `TIER-DIRECTION` reds *any* line in a SHARED
|
|
279
|
+
doc that names a `.shapeup/` path — a bare path cited in a sentence or a table cell,
|
|
280
|
+
not only a `[[tasks/...]]` link. A provenance sentence that names its real source
|
|
281
|
+
(`"extracted from .shapeup/<slug>/intake.md"`) reds for the same reason a task
|
|
282
|
+
wikilink does: the path dangles on every other clone. Cite the committed pitch or
|
|
283
|
+
shaping doc instead, or describe the run tier without a path.
|
|
278
284
|
- `[[tasks/...]]` wikilinks are valid only inside LOCAL documents (task files, the board,
|
|
279
285
|
EVAL reports), where they resolve against the LOCAL root (`.shapeup/<slug>/`);
|
|
280
286
|
every wikilink in a SHARED doc stays `spec_folder`-relative.
|
package/skills/coach/SKILL.md
CHANGED
|
@@ -93,7 +93,7 @@ never lands in any worker's KB.
|
|
|
93
93
|
Orchestrated, this skill is dispatched like every worker: a **WorkOrder** in (`--order <path>`,
|
|
94
94
|
operation `coach` or `scan`), a **WorkResult** out. Standalone, the raw feedback is passed
|
|
95
95
|
directly; it maps onto the one payload field registered for this worker in the central domain
|
|
96
|
-
registry (`
|
|
96
|
+
registry (`kernel/schemas/domain.schema.json`, `x-payload-by-worker`):
|
|
97
97
|
|
|
98
98
|
| Payload field | Standalone form | Meaning |
|
|
99
99
|
|---|---|---|
|
|
@@ -117,7 +117,13 @@ fields nobody used" → "Prefer the minimum DTO that satisfies the AC; don't add
|
|
|
117
117
|
fields"). Keep the originating why — a rule without its reason gets ignored or misapplied.
|
|
118
118
|
|
|
119
119
|
### Step 2 — ⏸ GATE COACH-1: Categorize (ASK, never assume)
|
|
120
|
-
This is the load-bearing gate. **
|
|
120
|
+
This is the load-bearing gate. **Resolve it first** — `node
|
|
121
|
+
"${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" gate --resolve COACH-1 --slug <slug>
|
|
122
|
+
[--file <path>|--preset <name>]` — so the ledger carries a row for the decision this gate makes,
|
|
123
|
+
same as every other gate in the run. Exit 0 (`decision=skip`) — an unattended lane with no live PO;
|
|
124
|
+
record nothing and stop here, the same outcome the CI preset's own note already documents. Exit 4
|
|
125
|
+
(`ask`) — proceed with the categorization below, which IS the PO conversation this decision opens.
|
|
126
|
+
**Do not infer which skill a rule belongs to** — a
|
|
121
127
|
miscategorized rule lands in a file the wrong worker reads (or no worker reads). Present every
|
|
122
128
|
candidate rule and ask the PO to assign each one. Emit this block, then stop and wait:
|
|
123
129
|
|
|
@@ -10,7 +10,7 @@ re-runs a computation that would erase true history.**
|
|
|
10
10
|
|
|
11
11
|
You are not a worker: no WorkOrder, no WorkResult, invoked directly by the user (or by `/hill`)
|
|
12
12
|
exactly like `shapeup` is. There is nothing to declare in
|
|
13
|
-
`
|
|
13
|
+
`kernel/schemas/domain.schema.json` and nothing to teach `harness compile` or
|
|
14
14
|
`harness reduce ingest` — those steps exist only for dispatched workers.
|
|
15
15
|
|
|
16
16
|
## What you read
|
|
@@ -71,9 +71,8 @@ Build one `{ scope_id, phase }` object per file.
|
|
|
71
71
|
## Rendering — the injection contract
|
|
72
72
|
|
|
73
73
|
The engine ships at `assets/dashboard.template.html` — a complete, self-contained HTML page
|
|
74
|
-
(inline CSS/JS, no external fetch beyond Google Fonts, no build step
|
|
75
|
-
|
|
76
|
-
fill in real data, and write the result.
|
|
74
|
+
(inline CSS/JS, no external fetch beyond Google Fonts, no build step). Do not rewrite it from a
|
|
75
|
+
text description; read it, fill in real data, and write the result.
|
|
77
76
|
|
|
78
77
|
1. For each discovered slug, build one entry:
|
|
79
78
|
|
|
@@ -67,8 +67,16 @@ H0.0 Ownership is DERIVED, never stated. Before the census says "no scope owns
|
|
|
67
67
|
H0.1 Unresolved scopes (breaker cases only):
|
|
68
68
|
- uphill/downhill scopes when round_budget hit 0 → CARRY candidates (their own hill
|
|
69
69
|
phase + open unknowns, from hill/<scope-id>.yml)
|
|
70
|
-
- scopes with hammer_proposals (attempt_budget exhausted) → CARRY candidates
|
|
71
|
-
|
|
70
|
+
- scopes with hammer_proposals (attempt_budget exhausted) → CARRY candidates. Exhaustion
|
|
71
|
+
is DERIVED, never read off `t0/verdicts/*.json` directly — a compiled order or a T0
|
|
72
|
+
verdict is writable by the very scope being judged and proves nothing on its own. Run
|
|
73
|
+
node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" probe attempts --slug <slug> \
|
|
74
|
+
--scope <scope-id> --round <n> --attempt-budget <n>
|
|
75
|
+
and cite its `spent`/`tripped` fields (exit 1 = tripped) — an attempt counts only when a
|
|
76
|
+
dispatch receipt AND either a leg-completion row or a WorkResult attest it, so a leg still
|
|
77
|
+
in flight holds it open rather than reading as exhausted. This is the SAME derivation the
|
|
78
|
+
round loop's own inner breaker reads, so the census and the breaker cannot disagree about
|
|
79
|
+
the same exhaustion the way a live run once measured.
|
|
72
80
|
H0.2 QA findings (qa-edge-hunter's hunt-report.md, when present) — all `~` by default.
|
|
73
81
|
H0.3 Discovered-task ledger entries still open (discovery/ledger.md, `[+]`/`~` unresolved).
|
|
74
82
|
H0.4 Attempt-budget hammer proposals (scopes that exhausted their T0 attempts during BUILD).
|
|
@@ -150,7 +158,7 @@ the harness (this is neither the generator nor the evaluator).
|
|
|
150
158
|
Orchestrated, this skill is dispatched like every worker: a **WorkOrder** in (`--order <path>`,
|
|
151
159
|
operation `hammer`), a **WorkResult** out. The standalone flags below map 1:1 onto the payload
|
|
152
160
|
fields registered for this worker in the central domain registry
|
|
153
|
-
(`
|
|
161
|
+
(`kernel/schemas/domain.schema.json`, `x-payload-by-worker`):
|
|
154
162
|
|
|
155
163
|
| Payload field | Standalone flag | Meaning |
|
|
156
164
|
|---|---|---|
|