shapeup-sdlc 3.7.0 → 3.7.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +1 -1
- package/AGENTS.md +3 -2
- package/README.md +22 -8
- package/SECURITY.md +3 -2
- package/hooks/hooks.json +10 -0
- package/hooks/tier-guard.mjs +162 -0
- package/kernel/compile.mjs +42 -2
- package/kernel/gate.mjs +68 -3
- package/kernel/probe/attempts.mjs +22 -5
- package/kernel/probe/eval.mjs +79 -7
- package/kernel/probe/resume.mjs +45 -9
- package/kernel/reduce/hill.mjs +200 -25
- package/kernel/reduce/ship.mjs +82 -4
- package/kernel/schemas/domain.schema.json +18 -4
- package/kernel/verify/spec.mjs +52 -20
- package/kernel/verify/t0.mjs +67 -2
- package/package.json +1 -1
- package/skills/ba-pitch-analyzer/references/doc-schemas.md +4 -0
- package/skills/hill-chart/SKILL.md +10 -6
- package/skills/tech-lead/references/gates.md +1 -1
- package/skills/tech-lead/references/tiny-lane.md +2 -1
- package/skills/tech-lead/workflows/shapeup-run.js +19 -5
|
@@ -1215,7 +1215,7 @@
|
|
|
1215
1215
|
}
|
|
1216
1216
|
},
|
|
1217
1217
|
"CommandResult": {
|
|
1218
|
-
"description": "One executed command's outcome inside a T0 artifact — produced by actually running the command; no agent can fabricate it.",
|
|
1218
|
+
"description": "One executed command's outcome inside a T0 artifact — produced by actually running the command; no agent can fabricate it. It carries the EVIDENCE, not only the score: `exit` maps a command that never started onto the same 1 a real failure returns, so without `error` a refused or timed-out command and a broken build are the same record everywhere downstream, and without the output a verdict asserting `exit 1` is an assertion nobody can check.",
|
|
1219
1219
|
"x-tier": "EMBEDDED",
|
|
1220
1220
|
"type": "object",
|
|
1221
1221
|
"properties": {
|
|
@@ -1223,10 +1223,23 @@
|
|
|
1223
1223
|
"type": "string"
|
|
1224
1224
|
},
|
|
1225
1225
|
"exit": {
|
|
1226
|
-
"type": "integer"
|
|
1226
|
+
"type": "integer",
|
|
1227
|
+
"description": "The process exit code, or 1 when the command never produced one. Not a discriminator on its own — read `error` to tell a crash from a failure."
|
|
1227
1228
|
},
|
|
1228
1229
|
"pass": {
|
|
1229
1230
|
"type": "boolean"
|
|
1231
|
+
},
|
|
1232
|
+
"error": {
|
|
1233
|
+
"type": "string",
|
|
1234
|
+
"description": "Present ONLY when the command did not run to completion — a spawn failure, a maxBuffer overflow, or the 10-minute timeout. Its presence is the fact the ratchet grades as `crash` (tree restored, never counted as a reverted attempt); its absence means the command ran and the exit code is its own."
|
|
1235
|
+
},
|
|
1236
|
+
"stdout_tail": {
|
|
1237
|
+
"type": "string",
|
|
1238
|
+
"description": "The last 4000 characters of stdout, omitted when nothing was printed. Kept for passing commands too: a fixture that exits 0 having run zero tests is the false green this layer exists to catch. A truncated tail carries a leading `[truncated: kept the last N of M characters]` line, so a partial stream never reads as a complete one."
|
|
1239
|
+
},
|
|
1240
|
+
"stderr_tail": {
|
|
1241
|
+
"type": "string",
|
|
1242
|
+
"description": "The last 4000 characters of stderr, same bound and same truncation marker as `stdout_tail`."
|
|
1230
1243
|
}
|
|
1231
1244
|
}
|
|
1232
1245
|
},
|
|
@@ -2833,9 +2846,10 @@
|
|
|
2833
2846
|
"type": "string",
|
|
2834
2847
|
"enum": [
|
|
2835
2848
|
"pass",
|
|
2836
|
-
"fail"
|
|
2849
|
+
"fail",
|
|
2850
|
+
"not-evaluated"
|
|
2837
2851
|
],
|
|
2838
|
-
"description": "shipped | ok: the round or run's EVAL verdict, lowercased from spec-evaluator's PASS|FAIL."
|
|
2852
|
+
"description": "shipped | ok: the round or run's EVAL verdict, lowercased from spec-evaluator's PASS|FAIL — or `not-evaluated` when a `shipped` return closed a `--no-eval` run, which reaches SHIP without ever dispatching the judge. Recorded plainly, never upgraded to `pass` (references/protocol.md)."
|
|
2839
2853
|
},
|
|
2840
2854
|
"rounds_used": {
|
|
2841
2855
|
"type": "integer"
|
package/kernel/verify/spec.mjs
CHANGED
|
@@ -235,11 +235,56 @@ export function lintScopes(scopes, repoFiles) {
|
|
|
235
235
|
}
|
|
236
236
|
|
|
237
237
|
/** Text forms worth scanning; anything else in a committed tree is not a reference carrier. */
|
|
238
|
-
const SCANNED = /\.(md|markdown|yml|yaml|json|txt)$/i;
|
|
238
|
+
export const SCANNED = /\.(md|markdown|yml|yaml|json|txt)$/i;
|
|
239
239
|
|
|
240
240
|
/** A machine-local board id. Strict on purpose: a committed tree has no reason to carry one at all. */
|
|
241
241
|
const TASK_ID = /\bTASK-[A-Za-z0-9][\w.-]*/;
|
|
242
242
|
|
|
243
|
+
/**
|
|
244
|
+
* The tier-direction violations one piece of text carries — the whole rule, on a string.
|
|
245
|
+
*
|
|
246
|
+
* EXPORTED BECAUSE IT HAS TWO ENFORCEMENT POINTS NOW, and they must not be allowed to drift.
|
|
247
|
+
* `lintCommittedTier` below walks files at GATE L1b; `hooks/tier-guard.mjs` refuses the same text
|
|
248
|
+
* at the moment a tool writes it, minutes earlier, while the writer still holds the context needed
|
|
249
|
+
* to rephrase. Four producers wrote committed files this rule reds and none of them learned from
|
|
250
|
+
* the lint, because by the time it speaks the dispatch that wrote the line is over. A second
|
|
251
|
+
* enforcement point is only worth having if it enforces the SAME predicate, so both call this.
|
|
252
|
+
*
|
|
253
|
+
* The detail strings are the message the writer reads, so they carry the remedy, not just the
|
|
254
|
+
* verdict: a board id resolves on the machine that wrote it and nowhere else, and a path into the
|
|
255
|
+
* gitignored tier dangles on every clone.
|
|
256
|
+
*
|
|
257
|
+
* @param {string} text - The file body, or the fragment a tool is about to write.
|
|
258
|
+
* @returns {Array<{kind:("board-id"|"local-path"), token:string, line:number, detail:string}>}
|
|
259
|
+
* One entry per offending line and form, in file order; [] when clean.
|
|
260
|
+
*/
|
|
261
|
+
export function tierLeaks(text) {
|
|
262
|
+
// Built from the LOCAL constant, never a literal — the storage roots have exactly one home.
|
|
263
|
+
const esc = LOCAL.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
264
|
+
// `\S+` where the walk's own rule is `\S`: the same LINES match either way (a path with one
|
|
265
|
+
// non-space character after the slash has at least one), and the longer form yields the token to
|
|
266
|
+
// quote back at the writer. Trailing punctuation the prose wrapped it in is trimmed off the
|
|
267
|
+
// quote only — never off the test.
|
|
268
|
+
const localPath = new RegExp(`${esc}/\\S+`);
|
|
269
|
+
const leaks = [];
|
|
270
|
+
String(text ?? "").split(/\r?\n/).forEach((line, i) => {
|
|
271
|
+
const task = line.match(TASK_ID);
|
|
272
|
+
if (task) {
|
|
273
|
+
leaks.push({ kind: "board-id", token: task[0], line: i + 1, detail:
|
|
274
|
+
`names ${task[0]} — a committed file cannot carry a board id. Boards live in ${LOCAL}/ ` +
|
|
275
|
+
"(gitignored) and renumber on every regeneration, so this resolves on the machine that wrote it " +
|
|
276
|
+
"and nowhere else. Cite the use case or the scope_id, which are stable." });
|
|
277
|
+
}
|
|
278
|
+
const path = line.match(localPath);
|
|
279
|
+
if (path) {
|
|
280
|
+
leaks.push({ kind: "local-path", token: path[0].replace(/[`)\]},.;:'"]+$/, ""), line: i + 1, detail:
|
|
281
|
+
`points into ${LOCAL}/ — a committed file cannot reference the gitignored tier; the path ` +
|
|
282
|
+
"dangles on every other clone. Name the committed artifact, or describe the tier without a path." });
|
|
283
|
+
}
|
|
284
|
+
});
|
|
285
|
+
return leaks;
|
|
286
|
+
}
|
|
287
|
+
|
|
243
288
|
/**
|
|
244
289
|
* Lint the WHOLE committed tree for references into the gitignored tier.
|
|
245
290
|
*
|
|
@@ -265,28 +310,15 @@ const TASK_ID = /\bTASK-[A-Za-z0-9][\w.-]*/;
|
|
|
265
310
|
export function lintCommittedTier({ cwd, slug }) {
|
|
266
311
|
const root = sharedRoot(cwd, slug);
|
|
267
312
|
if (!existsSync(root)) return [];
|
|
268
|
-
// Built from the LOCAL constant, never a literal — the storage roots have exactly one home.
|
|
269
|
-
const localPath = new RegExp(`${LOCAL.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}/\\S`);
|
|
270
313
|
const findings = [];
|
|
271
314
|
for (const rel of walkFiles(root)) {
|
|
272
315
|
if (!SCANNED.test(rel)) continue;
|
|
273
|
-
let
|
|
274
|
-
try {
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
279
|
-
findings.push({ rule: "TIER-DIRECTION", level: "red", detail:
|
|
280
|
-
`${at} names ${task[0]} — a committed file cannot carry a board id. Boards live in ${LOCAL}/ ` +
|
|
281
|
-
"(gitignored) and renumber on every regeneration, so this resolves on the machine that wrote it " +
|
|
282
|
-
"and nowhere else. Cite the use case or the scope_id, which are stable." });
|
|
283
|
-
}
|
|
284
|
-
if (localPath.test(line)) {
|
|
285
|
-
findings.push({ rule: "TIER-DIRECTION", level: "red", detail:
|
|
286
|
-
`${at} points into ${LOCAL}/ — a committed file cannot reference the gitignored tier; the path ` +
|
|
287
|
-
"dangles on every other clone. Name the committed artifact, or describe the tier without a path." });
|
|
288
|
-
}
|
|
289
|
-
});
|
|
316
|
+
let text;
|
|
317
|
+
try { text = readFileSync(join(root, rel), "utf8"); } catch { continue; }
|
|
318
|
+
for (const leak of tierLeaks(text)) {
|
|
319
|
+
findings.push({ rule: "TIER-DIRECTION", level: "red",
|
|
320
|
+
detail: `${relative(cwd, join(root, rel))}:${leak.line} ${leak.detail}` });
|
|
321
|
+
}
|
|
290
322
|
}
|
|
291
323
|
return findings;
|
|
292
324
|
}
|
package/kernel/verify/t0.mjs
CHANGED
|
@@ -79,6 +79,69 @@ function runCommand(cmd, cwd) {
|
|
|
79
79
|
return { cmd, exit: r.status ?? 1, pass: r.status === 0, stdout, stderr, ...(error ? { error } : {}) };
|
|
80
80
|
}
|
|
81
81
|
|
|
82
|
+
/**
|
|
83
|
+
* How much of each stream the persisted record keeps, per command.
|
|
84
|
+
*
|
|
85
|
+
* A bound rather than the whole stream, because one chatty fixture would otherwise make every
|
|
86
|
+
* reader of the run trace pay for it — and a bound at the END rather than the start, because that
|
|
87
|
+
* is where a stack trace, an assertion diff and a test summary all land.
|
|
88
|
+
*/
|
|
89
|
+
export const EVIDENCE_TAIL_CHARS = 4000;
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* The last `limit` characters of a stream, marked when anything was dropped.
|
|
93
|
+
*
|
|
94
|
+
* THE MARKER IS NOT DECORATION. A truncated tail that does not say so is a partial stream that
|
|
95
|
+
* reads as a complete one, which is the same class of defect as the one this whole change fixes:
|
|
96
|
+
* a record that overstates what it actually holds.
|
|
97
|
+
*
|
|
98
|
+
* @param {(string|null|undefined)} text - The captured stream.
|
|
99
|
+
* @param {number} [limit] - Characters to keep.
|
|
100
|
+
* @returns {(string|null)} The kept tail, or null when the stream was empty (the field is then
|
|
101
|
+
* omitted rather than stored as "", so "nothing was printed" stays visibly different from
|
|
102
|
+
* "nothing was kept").
|
|
103
|
+
*/
|
|
104
|
+
export function boundedTail(text, limit = EVIDENCE_TAIL_CHARS) {
|
|
105
|
+
const s = String(text ?? "");
|
|
106
|
+
if (!s) return null;
|
|
107
|
+
if (s.length <= limit) return s;
|
|
108
|
+
return `[truncated: kept the last ${limit} of ${s.length} characters]\n${s.slice(-limit)}`;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/**
|
|
112
|
+
* One command's outcome as the verdict artifact stores it — the evidence, not just the score.
|
|
113
|
+
*
|
|
114
|
+
* WHAT THIS FIXES. The artifact used to keep `{cmd, exit, pass}`, and `runCommand` maps a command
|
|
115
|
+
* that never started onto `exit: 1` — the same number a genuine failure returns. So a refused or
|
|
116
|
+
* timed-out command and a broken build were the SAME RECORD everywhere downstream: the digest, the
|
|
117
|
+
* hill, the report, the evaluator's citation and anyone reading the trace back afterwards. The
|
|
118
|
+
* kernel computed the difference (`error`, which the ratchet grades as `crash`) and discarded it
|
|
119
|
+
* one line later.
|
|
120
|
+
*
|
|
121
|
+
* `exit` is deliberately left as it is. Mapping a crash to some other number would change what the
|
|
122
|
+
* ratchet compares and what every existing reader parses; the distinction travels in `error`, which
|
|
123
|
+
* is the field that actually means "this never ran", and which the crash branch already reads.
|
|
124
|
+
*
|
|
125
|
+
* Output is kept for PASSING commands too, and that direction is not an afterthought: a fixture
|
|
126
|
+
* that exits 0 having run zero tests is the false green this evidence layer exists to catch, and
|
|
127
|
+
* its stdout is the only place that shows.
|
|
128
|
+
*
|
|
129
|
+
* @param {({cmd:string, exit:number, pass:boolean, stdout?:string, stderr?:string, error?:string}|null)} r
|
|
130
|
+
* A `runCommand` result, or null when no command was declared.
|
|
131
|
+
* @returns {(object|null)} The record to persist; null passes through unchanged.
|
|
132
|
+
*/
|
|
133
|
+
export function commandEvidence(r) {
|
|
134
|
+
if (!r) return null;
|
|
135
|
+
const stdout = boundedTail(r.stdout);
|
|
136
|
+
const stderr = boundedTail(r.stderr);
|
|
137
|
+
return {
|
|
138
|
+
cmd: r.cmd, exit: r.exit, pass: r.pass,
|
|
139
|
+
...(r.error ? { error: r.error } : {}),
|
|
140
|
+
...(stdout ? { stdout_tail: stdout } : {}),
|
|
141
|
+
...(stderr ? { stderr_tail: stderr } : {}),
|
|
142
|
+
};
|
|
143
|
+
}
|
|
144
|
+
|
|
82
145
|
/**
|
|
83
146
|
* Run every e2e fixture command for a scope.
|
|
84
147
|
* @param {string[]} fixtures - Fixture command lines (null/empty → no commands).
|
|
@@ -513,8 +576,10 @@ export async function cli(rawArgv) {
|
|
|
513
576
|
const { path, sha256: hash, trial } = writeArtifact(outDir, round, attempt, {
|
|
514
577
|
...(runId ? { run_id: runId } : {}),
|
|
515
578
|
scope_id: contract.scope_id,
|
|
516
|
-
|
|
517
|
-
|
|
579
|
+
// The evidence, not just the score — see `commandEvidence` for what the three-field record
|
|
580
|
+
// could not tell apart, and why `exit` still reads the way it always did.
|
|
581
|
+
fixtures: fixtures.results.map((r) => commandEvidence(r)),
|
|
582
|
+
db_probe: commandEvidence(dbProbe),
|
|
518
583
|
seesaw,
|
|
519
584
|
...verdict,
|
|
520
585
|
score: s,
|
package/package.json
CHANGED
|
@@ -281,6 +281,10 @@ Always use wikilinks (double brackets), never relative paths like `../domain-mod
|
|
|
281
281
|
(`"extracted from .shapeup/<slug>/intake.md"`) reds for the same reason a task
|
|
282
282
|
wikilink does: the path dangles on every other clone. Cite the committed pitch or
|
|
283
283
|
shaping doc instead, or describe the run tier without a path.
|
|
284
|
+
- **You will be stopped at the write, not at the gate.** The same rule is enforced as a
|
|
285
|
+
refusal on the Write/Edit itself, quoting the offending token back. There is nothing to
|
|
286
|
+
appeal: rephrase the line and write it again. Waiting for spec-lint to tell you at Board
|
|
287
|
+
Review costs the whole phase, which is what it used to cost every time.
|
|
284
288
|
- `[[tasks/...]]` wikilinks are valid only inside LOCAL documents (task files, the board,
|
|
285
289
|
EVAL reports), where they resolve against the LOCAL root (`.shapeup/<slug>/`);
|
|
286
290
|
every wikilink in a SHARED doc stays `spec_folder`-relative.
|
|
@@ -48,12 +48,16 @@ node "${CLAUDE_PLUGIN_ROOT}/kernel/harness.mjs" reduce graph --slug <slug>
|
|
|
48
48
|
```
|
|
49
49
|
|
|
50
50
|
**For a committed-only slug (`hasCommitted && !hasLocal`), NEVER call `reduce hill`.**
|
|
51
|
-
`deriveHill()` folds whatever T0 verdicts
|
|
52
|
-
none
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
51
|
+
`deriveHill()` folds whatever T0 verdicts, evaluation results and ledger rows exist in the LOCAL
|
|
52
|
+
run trace; a committed-only pitch has none, because that trace was cleaned up after shipping.
|
|
53
|
+
|
|
54
|
+
The runtime no longer takes that silence for an answer: a derivation whose run trace is absent
|
|
55
|
+
declines to write anything and reports every scope as underived (`derived: false`), so the
|
|
56
|
+
committed shards survive a call made in error. Treat that as a backstop, not a licence — an
|
|
57
|
+
underived report is not a refresh, and rendering it as one would show a pitch's true `FINISHED`
|
|
58
|
+
as freshly confirmed when nothing confirmed it. Read `shapeup/<slug>/hill/*.yml` for those slugs
|
|
59
|
+
exactly as committed, and render them with the archived state the template already implements
|
|
60
|
+
(see below). Same reasoning applies to `reduce graph` — there is no local trace to append from.
|
|
57
61
|
|
|
58
62
|
## Reading the hill shards
|
|
59
63
|
|
|
@@ -389,7 +389,7 @@ feature) and the failing step is compiled into round r+1's orders as `payload.bu
|
|
|
389
389
|
means L0 pinned no run command and the profile names no probe — the round proceeds over an
|
|
390
390
|
unproven build, and the block says so.
|
|
391
391
|
|
|
392
|
-
Under `--interactive` / `--auto`, the
|
|
392
|
+
Under `--interactive` / `--auto`, the gate block carries the board facts — `green_scopes`, `hammer_proposals` — and requires explicit PO approval to proceed. Under `--unattended` the answer set resolves it. **There is no hook behind this gate**: the `PreToolUse` warning it used to carry was retired into the block in v2.0, so what an unfinished board costs you here is a human reading the numbers, not a machine refusing the call.
|
|
393
393
|
|
|
394
394
|
---
|
|
395
395
|
|
|
@@ -26,7 +26,8 @@ scales down, its *verification floor* does not.
|
|
|
26
26
|
ingest-result dispatch. Tiny never means "just edit the file inline".
|
|
27
27
|
- **T0 verification.** A tiny change still proves itself by running — never by claim. If there
|
|
28
28
|
is no runnable check at all, that is a fit-check failure, not a reason to skip T0.
|
|
29
|
-
- **The safety
|
|
29
|
+
- **The machine guards — safety spine, substrate sandbox, tier guard.** They do not scale down,
|
|
30
|
+
and a tiny lane is where a committed file is most likely to be written by hand.
|
|
30
31
|
- **The discovery ledger.** `lane: tiny` is recorded, so a later reader knows exactly what was
|
|
31
32
|
NOT checked (no EVAL verdict, no QA charter, no wiring assertion).
|
|
32
33
|
|
|
@@ -1596,7 +1596,11 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1596
1596
|
`feature; this one does not build or launch. Round ${round + 1} fixes the gate's failing step.`);
|
|
1597
1597
|
} else if (args.noEval) {
|
|
1598
1598
|
log("EVAL — skipped (--no-eval)");
|
|
1599
|
-
|
|
1599
|
+
// references/protocol.md's own words for this, twice: a run --no-eval ships records
|
|
1600
|
+
// `not-evaluated` — "recorded plainly — never silently upgraded [to `pass`]". Nothing verified
|
|
1601
|
+
// the feature beyond task-executor's own per-AC evidence checks, and the report this run
|
|
1602
|
+
// freezes at GATE L4 has to say that as plainly as the gate block already does.
|
|
1603
|
+
verdict = "not-evaluated";
|
|
1600
1604
|
} else {
|
|
1601
1605
|
const e = await worker({
|
|
1602
1606
|
skill: "spec-evaluator", operation: "evaluate", schema: EVAL, phase: "Eval", label: `eval:r${round}`,
|
|
@@ -1647,14 +1651,16 @@ while (verdict !== "pass" && round <= maxRounds) {
|
|
|
1647
1651
|
const g3 = await crossGate("L3", "Eval", ["loop", "stop", "ask"], { round, verdict, build_gate: buildGate });
|
|
1648
1652
|
if (g3.stop) return await withWarnings(g3.stop);
|
|
1649
1653
|
|
|
1650
|
-
|
|
1654
|
+
// "not-evaluated" (--no-eval) ships exactly like "pass" — protocol.md: the run "goes straight to
|
|
1655
|
+
// SHIP" once EVAL is skipped, never spends another round waiting on a verdict nobody is producing.
|
|
1656
|
+
if (verdict === "pass" || verdict === "not-evaluated") break; // → QA → GATE H → ship
|
|
1651
1657
|
if (g3.decision === "stop" || round >= maxRounds) {
|
|
1652
1658
|
return await withWarnings({ status: "gate_h", breaker: "outer", hammer_proposals: allHammer, green_scopes: allGreen });
|
|
1653
1659
|
}
|
|
1654
1660
|
round += 1;
|
|
1655
1661
|
}
|
|
1656
1662
|
|
|
1657
|
-
if (verdict !== "pass") {
|
|
1663
|
+
if (verdict !== "pass" && verdict !== "not-evaluated") {
|
|
1658
1664
|
return await withWarnings({ status: "gate_h", breaker: "outer", hammer_proposals: allHammer, green_scopes: allGreen });
|
|
1659
1665
|
}
|
|
1660
1666
|
|
|
@@ -1692,7 +1698,12 @@ if (h.verdict === "cannot-ship") {
|
|
|
1692
1698
|
if (g.stop) return await withWarnings(g.stop);
|
|
1693
1699
|
}
|
|
1694
1700
|
|
|
1695
|
-
|
|
1701
|
+
// The frozen report's own verdict line has to say what actually happened — a `--no-eval` run
|
|
1702
|
+
// verified nothing beyond task-executor's per-AC checks, and `reduce ship` already accepts
|
|
1703
|
+
// "not-evaluated" as a real verdict (its own usage string, and `generate()`'s no-artifact
|
|
1704
|
+
// default). Hardcoding PASS here is exactly the silent upgrade protocol.md's Rules forbid.
|
|
1705
|
+
const shipVerdict = verdict === "not-evaluated" ? "not-evaluated" : "PASS";
|
|
1706
|
+
const ship = await cmd(`reduce ship --slug ${slug} --verdict ${shipVerdict} --qa ${qaRan ? "run" : "skipped"}`, "Ship", "ship-report");
|
|
1696
1707
|
await advisory(`report export --slug ${slug}`, "Ship", "export-run");
|
|
1697
1708
|
// The run's own concurrency, printed once where the records are complete and before the next run
|
|
1698
1709
|
// supersedes the trace. It is a projection over `receipts/dispatch.jsonl` and `legs.jsonl`, so it
|
|
@@ -1707,7 +1718,10 @@ const ALL_DIMS = ["spec-conformance", "tdd-surface", "integration", "completenes
|
|
|
1707
1718
|
|
|
1708
1719
|
return await withWarnings({
|
|
1709
1720
|
status: "shipped",
|
|
1710
|
-
verdict
|
|
1721
|
+
// The real verdict this run reached — "pass" or, over a --no-eval run, "not-evaluated". GATE L4's
|
|
1722
|
+
// own sign-off block reads this field verbatim (SKILL.md Step 4); hardcoding "pass" here told a
|
|
1723
|
+
// human answering that gate the run was graded when it never was.
|
|
1724
|
+
verdict,
|
|
1711
1725
|
rounds_used: round,
|
|
1712
1726
|
dims_not_evaluated: ALL_DIMS.filter((d) => !evalDims.includes(d)),
|
|
1713
1727
|
qa_findings: qaFindings,
|