shapeup-sdlc 3.7.1 → 3.7.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/AGENTS.md +7 -6
- package/README.md +18 -8
- package/SECURITY.md +2 -2
- package/hooks/lib/decision.mjs +10 -2
- package/hooks/sandbox-guard.mjs +78 -11
- package/kernel/compile.mjs +27 -1
- package/kernel/gate.mjs +1 -1
- package/kernel/lib/paths.mjs +38 -3
- package/kernel/probe/eval.mjs +79 -7
- package/kernel/probe/leg.mjs +80 -3
- package/kernel/probe/resume.mjs +10 -5
- package/kernel/reduce/graph.mjs +23 -4
- package/kernel/reduce/hill.mjs +200 -25
- package/kernel/reduce/ship.mjs +43 -1
- package/kernel/report/export.mjs +6 -1
- package/kernel/report/facts.mjs +3 -0
- package/kernel/schemas/domain.schema.json +7 -2
- package/package.json +1 -1
- package/skills/hill-chart/SKILL.md +10 -6
- package/skills/tech-lead/references/gates.md +1 -1
- package/skills/tech-lead/workflows/shapeup-run.js +133 -19
|
@@ -41,7 +41,8 @@
|
|
|
41
41
|
// { status: "shipped", verdict, rounds_used, dims_not_evaluated, qa_findings, report }
|
|
42
42
|
// { status: "paused", paused_at, block, valid_decisions, context }
|
|
43
43
|
// { status: "aborted", aborted_at, reason }
|
|
44
|
-
// { status: "gate_h", breaker: "outer"|"
|
|
44
|
+
// { status: "gate_h", breaker: "outer"|"attempt_budget"|"none"|"deadline", hammer_proposals, green_scopes,
|
|
45
|
+
// tripped_scopes?, unapplied_results? }
|
|
45
46
|
|
|
46
47
|
// meta must be a PURE LITERAL — the runtime parses it statically, before the body ever runs, and
|
|
47
48
|
// rejects the whole script on anything it has to evaluate. A `+`-joined description is a
|
|
@@ -471,6 +472,32 @@ const T0CHECK = {
|
|
|
471
472
|
required: ["green"],
|
|
472
473
|
};
|
|
473
474
|
|
|
475
|
+
/** `probe leg --order` — did one named order's result reach the single writer? Any phase. */
|
|
476
|
+
const ORDERLEG = {
|
|
477
|
+
type: "object",
|
|
478
|
+
properties: {
|
|
479
|
+
closed: { type: "boolean" },
|
|
480
|
+
found: { type: "boolean" },
|
|
481
|
+
order: nullable("string"),
|
|
482
|
+
has_result: { type: "boolean" },
|
|
483
|
+
applied: { type: "boolean" },
|
|
484
|
+
},
|
|
485
|
+
required: ["closed", "found", "has_result", "applied"],
|
|
486
|
+
};
|
|
487
|
+
|
|
488
|
+
/** `probe attempts` — the attested census: what was spent, and whether the breaker really tripped. */
|
|
489
|
+
const ATTEMPTS = {
|
|
490
|
+
type: "object",
|
|
491
|
+
properties: {
|
|
492
|
+
scope_id: { type: "string" },
|
|
493
|
+
spent: { type: "integer" },
|
|
494
|
+
in_flight: { type: "integer" },
|
|
495
|
+
green: { type: "boolean" },
|
|
496
|
+
tripped: { type: "boolean" },
|
|
497
|
+
},
|
|
498
|
+
required: ["scope_id", "spent", "tripped"],
|
|
499
|
+
};
|
|
500
|
+
|
|
474
501
|
/** `probe leg` — did this scope's result reach the board, or is it finished work nothing applied? */
|
|
475
502
|
const LEGCHECK = {
|
|
476
503
|
type: "object",
|
|
@@ -817,6 +844,42 @@ async function crossGate(gateId, phaseName, validDecisions, ctx) {
|
|
|
817
844
|
const attest = (phaseKey, phaseName, label) =>
|
|
818
845
|
cmd(`probe resume --slug ${slug} --require ${phaseKey}`, phaseName, label);
|
|
819
846
|
|
|
847
|
+
/**
|
|
848
|
+
* The other half of a phase post-condition: the single writer ran.
|
|
849
|
+
*
|
|
850
|
+
* The artifact check asks whether the worker wrote its product. It cannot ask whether the
|
|
851
|
+
* WorkResult was applied — the leg row, the discoveries, the board — because the product is written
|
|
852
|
+
* by the worker directly and the envelope is applied by `reduce ingest`, a separate act the leg's
|
|
853
|
+
* script names as its last step. Measured on a live run: a planning phase landed a result naming
|
|
854
|
+
* seventeen artifacts and no leg row, the artifact check passed, and the run walked on with the
|
|
855
|
+
* result's discoveries never reaching the ledger. So this asks the leg ledger by order name. A
|
|
856
|
+
* result on disk that nothing applied is ingested here — the same repair the build round makes,
|
|
857
|
+
* for the same reason: the single writer is the invariant, not which step invokes it — and if it
|
|
858
|
+
* still is not applied after that, the run stops with a cause naming the writer that did not run,
|
|
859
|
+
* rather than folding the gap into a phase that "completed".
|
|
860
|
+
*
|
|
861
|
+
* @param {string} gate - The gate name to report the abort under.
|
|
862
|
+
* @param {string} phaseKey - The phase, which is also its order's file stem.
|
|
863
|
+
* @param {string} phaseName - Progress group.
|
|
864
|
+
* @returns {Promise<(object|null)>} An aborted RunReturn, or null when the leg closed (or the
|
|
865
|
+
* question could not be asked — a probe that did not run proves nothing, and is logged as such).
|
|
866
|
+
*/
|
|
867
|
+
async function requireLeg(gate, phaseKey, phaseName) {
|
|
868
|
+
const ask = () => query(`probe leg --slug ${slug} --order "${phaseKey}"`, ORDERLEG, phaseName, `legcheck:${phaseKey}`);
|
|
869
|
+
let leg = await ask();
|
|
870
|
+
if (!leg || !leg.found) { log(`${gate} — could not ask the leg ledger about "${phaseKey}" (probe returned ${leg ? "no order" : "nothing"}); proceeding on the artifact alone.`); return null; }
|
|
871
|
+
if (!leg.has_result || leg.applied) return null;
|
|
872
|
+
log(`${gate} — "${phaseKey}" came back with a result nothing applied (no leg row). Ingesting it here: ${leg.order}.`);
|
|
873
|
+
await advisory(`reduce ingest --order "${leg.order}"`, phaseName, `late-ingest:${phaseKey}`);
|
|
874
|
+
leg = await ask();
|
|
875
|
+
if (leg?.applied) return null;
|
|
876
|
+
return aborted(gate,
|
|
877
|
+
`${gate}: the single writer did not run for "${phaseKey}" — its WorkResult is on disk and no leg row ` +
|
|
878
|
+
`records it being applied, and a late \`reduce ingest\` did not take either. The phase's product exists; ` +
|
|
879
|
+
`what it discovered and reported never reached the ledger or the board. Read the result, run ` +
|
|
880
|
+
`\`reduce ingest --order "${leg?.order ?? "<its order>"}"\` by hand to see why it refuses, then relaunch.`);
|
|
881
|
+
}
|
|
882
|
+
|
|
820
883
|
/**
|
|
821
884
|
* The phase post-condition: the artifact is on disk, or the run stops here.
|
|
822
885
|
*
|
|
@@ -830,7 +893,7 @@ const attest = (phaseKey, phaseName, label) =>
|
|
|
830
893
|
*/
|
|
831
894
|
async function requirePhase(gate, phaseKey, phaseName) {
|
|
832
895
|
const r = await attest(phaseKey, phaseName, `require:${phaseKey}`);
|
|
833
|
-
if (r.exit_code === 0) return
|
|
896
|
+
if (r.exit_code === 0) return await requireLeg(gate, phaseKey, phaseName);
|
|
834
897
|
// Exit 6 is `probe resume --require`'s OWN documented code for "the artifact really is not on
|
|
835
898
|
// disk" (kernel/probe/resume.mjs banner). Any other value — including -1, the courier's sentinel
|
|
836
899
|
// for a tool call that never ran — is not that predicate answering "no"; it is the predicate never
|
|
@@ -1001,7 +1064,11 @@ async function closeIfTerminal(ret) {
|
|
|
1001
1064
|
const cause = ret.status === "aborted"
|
|
1002
1065
|
? `${ret.aborted_at || "?"}: ${ret.reason || "no reason recorded"}`
|
|
1003
1066
|
: ret.status === "gate_h"
|
|
1004
|
-
?
|
|
1067
|
+
? (ret.stalled ? `stalled=${ret.stalled} ` : "")
|
|
1068
|
+
+ `breaker=${ret.breaker ?? "?"} green_scopes=${Array.isArray(ret.green_scopes) ? ret.green_scopes.length : "?"} hammer_proposals=${Array.isArray(ret.hammer_proposals) ? ret.hammer_proposals.length : "?"}`
|
|
1069
|
+
+ (Array.isArray(ret.tripped_scopes) ? ` tripped_scopes=${ret.tripped_scopes.length}` : "")
|
|
1070
|
+
// A result the single writer never applied is named at the close, not folded into "not green".
|
|
1071
|
+
+ (Array.isArray(ret.unapplied_results) && ret.unapplied_results.length ? ` unapplied_results=${ret.unapplied_results.length}` : "")
|
|
1005
1072
|
: `verdict=${ret.verdict ?? "?"} rounds=${ret.rounds_used ?? "?"} qa_findings=${ret.qa_findings ?? "?"}`;
|
|
1006
1073
|
// `--close-arm` hands the kernel the arm itself (not a status this file decided was terminal) —
|
|
1007
1074
|
// a non-terminal arm (`paused`, `ok`) still exits 0 with no "decision" key, so the branches below
|
|
@@ -1368,6 +1435,7 @@ let verdict = null;
|
|
|
1368
1435
|
// other two. The cure is the same each time and it is not a bigger variable: re-derive the fact.
|
|
1369
1436
|
const allGreen = [];
|
|
1370
1437
|
const allHammer = [];
|
|
1438
|
+
const allUnapplied = []; // results on disk the single writer never applied, run-wide
|
|
1371
1439
|
// OUTSIDE the loop, because its whole purpose is to cross a round boundary: round r's verdict is
|
|
1372
1440
|
// what round r+1 has to act on. Declared inside, it was in the temporal dead zone at the BUILD that
|
|
1373
1441
|
// needed it — a runtime error no static check can see, since nothing but a real second round ever
|
|
@@ -1408,7 +1476,7 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1408
1476
|
const budget = await cmd(`verify budget --slug ${slug} --strict`, "Build", `budget:r${round}`);
|
|
1409
1477
|
if (budget.exit_code === 6) {
|
|
1410
1478
|
await advisory(`reduce hill --slug ${slug}`, "Build", "hill-derive");
|
|
1411
|
-
return await withWarnings({ status: "gate_h", breaker: "deadline", hammer_proposals: allHammer, green_scopes: allGreen });
|
|
1479
|
+
return await withWarnings({ status: "gate_h", breaker: "deadline", unapplied_results: allUnapplied, hammer_proposals: allHammer, green_scopes: allGreen });
|
|
1412
1480
|
}
|
|
1413
1481
|
|
|
1414
1482
|
log(`BUILD round ${round} — ${scopes.length} scope(s), up to ${maxParallelScopes} at once, attempt budget ${attemptBudget}`);
|
|
@@ -1463,8 +1531,22 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1463
1531
|
: { scope_id: s.scope_id, pending: true }), // not green yet → stage 2 builds it
|
|
1464
1532
|
async (pre, s) => (pre?.pending ? buildScope(s, round) : pre),
|
|
1465
1533
|
async (res, s) => {
|
|
1466
|
-
if (!res
|
|
1467
|
-
|
|
1534
|
+
if (!res) return res;
|
|
1535
|
+
// THE SINGLE WRITER IS ASKED FIRST, WHATEVER THE LEG SAID. This question used to sit behind
|
|
1536
|
+
// the green checks below, and the state it exists to catch — a result on disk that nothing
|
|
1537
|
+
// applied — was reachable only for a scope that was already fully green. Measured live: a
|
|
1538
|
+
// leg reported green with no T0 verdict at all, the T0 re-read correctly returned "not
|
|
1539
|
+
// green", the round returned before this line, and the result — six tasks, two discoveries
|
|
1540
|
+
// — was never read by anyone. A dead leg (`__failed`) can have left a result too. So every
|
|
1541
|
+
// settled scope is asked, and what it answers travels on the result to the close, where an
|
|
1542
|
+
// unapplied result is named rather than folded into "not green". Application stays gated
|
|
1543
|
+
// on the green checks: ingest ticks acceptance boxes, and a result T0 never measured must
|
|
1544
|
+
// not mark work green — the repair below is for a scope that IS green.
|
|
1545
|
+
const applied = await query(`probe leg --slug ${slug} --scope ${s.scope_id} --round ${round}`,
|
|
1546
|
+
LEGCHECK, "Build", `legcheck:${s.scope_id}-r${round}`);
|
|
1547
|
+
const unapplied = applied?.closed ? [] : (applied?.unapplied || []);
|
|
1548
|
+
if (res.__failed) return unapplied.length ? { ...res, unapplied } : res;
|
|
1549
|
+
if (!res.green) return unapplied.length ? { ...res, unapplied } : res;
|
|
1468
1550
|
// THE T0 RE-READ IS SKIPPED FOR A RESUMED SCOPE; THE LEG CHECK BELOW IS NOT.
|
|
1469
1551
|
//
|
|
1470
1552
|
// `resumed` means the graph already reported this scope green for this round, so re-reading
|
|
@@ -1481,7 +1563,7 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1481
1563
|
if (!confirmed?.green) {
|
|
1482
1564
|
log(`BUILD r${round} — ${s.scope_id} reported green but no T0 verdict is on disk for this ` +
|
|
1483
1565
|
`round; treating it as not green (the evaluator cites that artifact, and it is not there).`);
|
|
1484
|
-
return { ...res, green: false, reason: "reported green with no T0 verdict artifact on disk" };
|
|
1566
|
+
return { ...res, green: false, reason: "reported green with no T0 verdict artifact on disk", ...(unapplied.length ? { unapplied } : {}) };
|
|
1485
1567
|
}
|
|
1486
1568
|
}
|
|
1487
1569
|
// AND ITS RESULT HAS TO HAVE REACHED THE BOARD. A green T0 says the worker's fixtures ran and
|
|
@@ -1494,9 +1576,7 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1494
1576
|
//
|
|
1495
1577
|
// The evidence is the leg-completion row, because `reduce ingest` writes it: its presence
|
|
1496
1578
|
// proves the writer ran, and it is not something the leg can assert about itself.
|
|
1497
|
-
|
|
1498
|
-
LEGCHECK, "Build", `legcheck:${s.scope_id}-r${round}`);
|
|
1499
|
-
for (const orderPath of (applied?.closed ? [] : applied?.unapplied || [])) {
|
|
1579
|
+
for (const orderPath of unapplied) {
|
|
1500
1580
|
// INGESTED HERE RATHER THAN FAILED. The result is on disk and valid — re-running the leg
|
|
1501
1581
|
// would pay a whole attempt again for work already done. Only the single writer writes
|
|
1502
1582
|
// shared state, and that writer is this command; which step invokes it is not the invariant.
|
|
@@ -1524,6 +1604,10 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1524
1604
|
|
|
1525
1605
|
for (const [i, res] of settled.entries()) {
|
|
1526
1606
|
const scopeId = buildOrder[i].scope_id;
|
|
1607
|
+
for (const p of (res?.unapplied || [])) {
|
|
1608
|
+
if (!allUnapplied.includes(p)) allUnapplied.push(p);
|
|
1609
|
+
log(`BUILD r${round} — ${scopeId} has a result on disk nothing applied: ${p}`);
|
|
1610
|
+
}
|
|
1527
1611
|
// A dead builder is a SPENT ATTEMPT, not a dead run: the scope goes to GATE H's census and
|
|
1528
1612
|
// the round continues. Killing the run here would discard every other scope's green work.
|
|
1529
1613
|
if (!res || res.__failed) {
|
|
@@ -1541,10 +1625,26 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1541
1625
|
for (const sid of roundGreen) { const i = allHammer.indexOf(sid); if (i !== -1) allHammer.splice(i, 1); }
|
|
1542
1626
|
for (const sid of roundHammer) if (!allHammer.includes(sid) && !allGreen.includes(sid)) allHammer.push(sid);
|
|
1543
1627
|
|
|
1544
|
-
//
|
|
1628
|
+
// NOTHING GREEN AND SOMETHING QUEUED → GATE H. This used to return the literal `inner` as its breaker, and the
|
|
1629
|
+
// protocol's INNER breaker is the per-scope attempt budget, which "queues a proposal, never blocks
|
|
1630
|
+
// the round" — so the close named a breaker the attested census flatly denied (one attempt spent
|
|
1631
|
+
// of five, `tripped: false`), and the operator was told a scope had exhausted its attempts after
|
|
1632
|
+
// it used one. The word is now earned: `probe attempts` — the one derivation built so the census
|
|
1633
|
+
// and the breaker cannot disagree — is asked for every queued scope, and the return names
|
|
1634
|
+
// `attempt_budget` only for the scopes it says tripped, `none` when the round simply stalled.
|
|
1545
1635
|
if (roundGreen.length === 0 && roundHammer.length > 0) {
|
|
1636
|
+
const tripped = [];
|
|
1637
|
+
for (const sid of roundHammer) {
|
|
1638
|
+
const census = await query(`probe attempts --slug ${slug} --scope "${sid}" --round ${round} --attempt-budget ${attemptBudget}`,
|
|
1639
|
+
ATTEMPTS, "Build", `census:${sid}-r${round}`);
|
|
1640
|
+
if (census?.tripped) tripped.push(sid);
|
|
1641
|
+
}
|
|
1546
1642
|
await advisory(`reduce hill --slug ${slug}`, "Build", "hill-derive");
|
|
1547
|
-
return await withWarnings({
|
|
1643
|
+
return await withWarnings({
|
|
1644
|
+
status: "gate_h", breaker: tripped.length ? "attempt_budget" : "none", stalled: "no_green",
|
|
1645
|
+
tripped_scopes: tripped, unapplied_results: allUnapplied,
|
|
1646
|
+
hammer_proposals: allHammer, green_scopes: allGreen,
|
|
1647
|
+
});
|
|
1548
1648
|
}
|
|
1549
1649
|
|
|
1550
1650
|
// ---- ROUND BUILD GATE — the feature builds and launches, measured before anyone is asked --------
|
|
@@ -1596,7 +1696,11 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1596
1696
|
`feature; this one does not build or launch. Round ${round + 1} fixes the gate's failing step.`);
|
|
1597
1697
|
} else if (args.noEval) {
|
|
1598
1698
|
log("EVAL — skipped (--no-eval)");
|
|
1599
|
-
|
|
1699
|
+
// references/protocol.md's own words for this, twice: a run --no-eval ships records
|
|
1700
|
+
// `not-evaluated` — "recorded plainly — never silently upgraded [to `pass`]". Nothing verified
|
|
1701
|
+
// the feature beyond task-executor's own per-AC evidence checks, and the report this run
|
|
1702
|
+
// freezes at GATE L4 has to say that as plainly as the gate block already does.
|
|
1703
|
+
verdict = "not-evaluated";
|
|
1600
1704
|
} else {
|
|
1601
1705
|
const e = await worker({
|
|
1602
1706
|
skill: "spec-evaluator", operation: "evaluate", schema: EVAL, phase: "Eval", label: `eval:r${round}`,
|
|
@@ -1647,15 +1751,17 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1647
1751
|
const g3 = await crossGate("L3", "Eval", ["loop", "stop", "ask"], { round, verdict, build_gate: buildGate });
|
|
1648
1752
|
if (g3.stop) return await withWarnings(g3.stop);
|
|
1649
1753
|
|
|
1650
|
-
|
|
1754
|
+
// "not-evaluated" (--no-eval) ships exactly like "pass" — protocol.md: the run "goes straight to
|
|
1755
|
+
// SHIP" once EVAL is skipped, never spends another round waiting on a verdict nobody is producing.
|
|
1756
|
+
if (verdict === "pass" || verdict === "not-evaluated") break; // → QA → GATE H → ship
|
|
1651
1757
|
if (g3.decision === "stop" || round >= maxRounds) {
|
|
1652
|
-
return await withWarnings({ status: "gate_h", breaker: "outer", hammer_proposals: allHammer, green_scopes: allGreen });
|
|
1758
|
+
return await withWarnings({ status: "gate_h", breaker: "outer", unapplied_results: allUnapplied, hammer_proposals: allHammer, green_scopes: allGreen });
|
|
1653
1759
|
}
|
|
1654
1760
|
round += 1;
|
|
1655
1761
|
}
|
|
1656
1762
|
|
|
1657
|
-
if (verdict !== "pass") {
|
|
1658
|
-
return await withWarnings({ status: "gate_h", breaker: "outer", hammer_proposals: allHammer, green_scopes: allGreen });
|
|
1763
|
+
if (verdict !== "pass" && verdict !== "not-evaluated") {
|
|
1764
|
+
return await withWarnings({ status: "gate_h", breaker: "outer", unapplied_results: allUnapplied, hammer_proposals: allHammer, green_scopes: allGreen });
|
|
1659
1765
|
}
|
|
1660
1766
|
|
|
1661
1767
|
// ---- QA (post-PASS, pre-ship) — a level-up, never a gate. `--no-qa` answers it "skip". --------
|
|
@@ -1692,7 +1798,12 @@ if (h.verdict === "cannot-ship") {
|
|
|
1692
1798
|
if (g.stop) return await withWarnings(g.stop);
|
|
1693
1799
|
}
|
|
1694
1800
|
|
|
1695
|
-
|
|
1801
|
+
// The frozen report's own verdict line has to say what actually happened — a `--no-eval` run
|
|
1802
|
+
// verified nothing beyond task-executor's per-AC checks, and `reduce ship` already accepts
|
|
1803
|
+
// "not-evaluated" as a real verdict (its own usage string, and `generate()`'s no-artifact
|
|
1804
|
+
// default). Hardcoding PASS here is exactly the silent upgrade protocol.md's Rules forbid.
|
|
1805
|
+
const shipVerdict = verdict === "not-evaluated" ? "not-evaluated" : "PASS";
|
|
1806
|
+
const ship = await cmd(`reduce ship --slug ${slug} --verdict ${shipVerdict} --qa ${qaRan ? "run" : "skipped"}`, "Ship", "ship-report");
|
|
1696
1807
|
await advisory(`report export --slug ${slug}`, "Ship", "export-run");
|
|
1697
1808
|
// The run's own concurrency, printed once where the records are complete and before the next run
|
|
1698
1809
|
// supersedes the trace. It is a projection over `receipts/dispatch.jsonl` and `legs.jsonl`, so it
|
|
@@ -1707,7 +1818,10 @@ const ALL_DIMS = ["spec-conformance", "tdd-surface", "integration", "completenes
|
|
|
1707
1818
|
|
|
1708
1819
|
return await withWarnings({
|
|
1709
1820
|
status: "shipped",
|
|
1710
|
-
verdict
|
|
1821
|
+
// The real verdict this run reached — "pass" or, over a --no-eval run, "not-evaluated". GATE L4's
|
|
1822
|
+
// own sign-off block reads this field verbatim (SKILL.md Step 4); hardcoding "pass" here told a
|
|
1823
|
+
// human answering that gate the run was graded when it never was.
|
|
1824
|
+
verdict,
|
|
1711
1825
|
rounds_used: round,
|
|
1712
1826
|
dims_not_evaluated: ALL_DIMS.filter((d) => !evalDims.includes(d)),
|
|
1713
1827
|
qa_findings: qaFindings,
|