@gobing-ai/spur 0.3.69 → 0.3.71
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/config/corpus-baseline.json +0 -18
- package/config/plugin-scripts.json +8 -0
- package/config/workflow-composition-baseline.json +123 -377
- package/config/workflows/idea-pipeline.yaml +4 -3
- package/config/workflows/task-pipeline.yaml +58 -9
- package/config/workflows/wrapup-pipeline.yaml +10 -4
- package/package.json +9 -9
- package/plugins/sp/commands/dev-fixall.md +4 -3
- package/plugins/sp/plugin.json +1 -1
- package/plugins/sp/scripts/task-evidence-precheck.ts +181 -0
- package/plugins/sp/scripts/verify-answer-lint.ts +395 -0
- package/plugins/sp/skills/code-implementation/SKILL.md +1 -1
- package/plugins/sp/skills/code-verification/SKILL.md +11 -10
- package/plugins/sp/skills/code-verification/references/verdict-schema.md +10 -0
- package/plugins/sp/skills/spec-decomposition/SKILL.md +1 -1
- package/plugins/sp/skills/spur-cli/references/history.md +8 -6
- package/plugins/sp/skills/spur-dev/references/dev-operations.md +2 -2
- package/plugins/sp/skills/spur-dev/references/execution-workflow.md +8 -1
- package/plugins/sp/skills/spur-dev/references/gate-checklists.md +17 -1
- package/plugins/sp/skills/spur-dev/references/inline-pipeline-driver.md +25 -0
- package/spur.js +1385 -291
- package/web/_astro/BoardApp.BnjsI80-.js +178 -0
- package/web/_astro/BoardApp.SJcrHBZp.js +1 -0
- package/web/_astro/{TaskDetail.Dl2Eaj1w.js → TaskDetail.BvkKvo57.js} +1 -1
- package/web/_astro/{arc.uG14rp8A.js → arc.DuEIzPMi.js} +1 -1
- package/web/_astro/{architectureDiagram-3BPJPVTR.Dye6uD_x.js → architectureDiagram-3BPJPVTR.D0dUxCjA.js} +1 -1
- package/web/_astro/{blockDiagram-GPEHLZMM.B9Pkh7Hb.js → blockDiagram-GPEHLZMM.Cv60iOpM.js} +1 -1
- package/web/_astro/{c4Diagram-AAUBKEIU.C2x7SC_X.js → c4Diagram-AAUBKEIU.Ddx7WGhb.js} +1 -1
- package/web/_astro/channel.tWfETQvX.js +1 -0
- package/web/_astro/{chunk-2J33WTMH.D2p4-nWk.js → chunk-2J33WTMH.BGKzU_1t.js} +1 -1
- package/web/_astro/{chunk-4BX2VUAB.S-6jf33o.js → chunk-4BX2VUAB.FlApjIIH.js} +1 -1
- package/web/_astro/{chunk-55IACEB6.DVXj4Fdh.js → chunk-55IACEB6.CDfmvDeW.js} +1 -1
- package/web/_astro/{chunk-727SXJPM.Dz689FMN.js → chunk-727SXJPM.T9bc-xir.js} +1 -1
- package/web/_astro/{chunk-AQP2D5EJ.KxYj5TnI.js → chunk-AQP2D5EJ.BiTc-MQo.js} +1 -1
- package/web/_astro/{chunk-FMBD7UC4.itTQyHQB.js → chunk-FMBD7UC4.he4KrHni.js} +1 -1
- package/web/_astro/{chunk-ND2GUHAM.euSrbJf5.js → chunk-ND2GUHAM.D05XuUuJ.js} +1 -1
- package/web/_astro/{chunk-QZHKN3VN.OWASJRQy.js → chunk-QZHKN3VN.CA2NThIE.js} +1 -1
- package/web/_astro/{classDiagram-4FO5ZUOK.BLvrlpNO.js → classDiagram-4FO5ZUOK.Couj-zYZ.js} +1 -1
- package/web/_astro/{classDiagram-v2-Q7XG4LA2.BLvrlpNO.js → classDiagram-v2-Q7XG4LA2.Couj-zYZ.js} +1 -1
- package/web/_astro/{cose-bilkent-S5V4N54A.XBF-rmyD.js → cose-bilkent-S5V4N54A.B8YYW7NG.js} +1 -1
- package/web/_astro/{cynefin-OW5HDTMX.DlCx762Z.js → cynefin-OW5HDTMX.BExFdiin.js} +1 -1
- package/web/_astro/{dagre-BM42HDAG.D17Rshxv.js → dagre-BM42HDAG.BybKbz3q.js} +1 -1
- package/web/_astro/{diagram-2AECGRRQ.AhBIVJC8.js → diagram-2AECGRRQ.UyRTSl9n.js} +1 -1
- package/web/_astro/{diagram-5GNKFQAL.C9ximjyC.js → diagram-5GNKFQAL.BrxCucBf.js} +1 -1
- package/web/_astro/{diagram-KO2AKTUF.CZb7Ru_9.js → diagram-KO2AKTUF.CdE5oy5J.js} +1 -1
- package/web/_astro/{diagram-LMA3HP47.BW7LwqoS.js → diagram-LMA3HP47.BJLgdosK.js} +1 -1
- package/web/_astro/{diagram-OG6HWLK6.XC025W0V.js → diagram-OG6HWLK6.CUynieTU.js} +1 -1
- package/web/_astro/{erDiagram-TEJ5UH35.CpMXmBDP.js → erDiagram-TEJ5UH35.C0vS6DJv.js} +1 -1
- package/web/_astro/{flowDiagram-I6XJVG4X.D2ednJWg.js → flowDiagram-I6XJVG4X.T3QLi_en.js} +1 -1
- package/web/_astro/{ganttDiagram-6RSMTGT7.BjL9FGKO.js → ganttDiagram-6RSMTGT7.BNsk3w9Z.js} +1 -1
- package/web/_astro/{gitGraphDiagram-PVQCEYII.B90g1VGk.js → gitGraphDiagram-PVQCEYII.Mwe2I4V6.js} +1 -1
- package/web/_astro/index.9npdrEIr.css +1 -0
- package/web/_astro/{infoDiagram-5YYISTIA.RqgycKtQ.js → infoDiagram-5YYISTIA.BlcjLmtc.js} +1 -1
- package/web/_astro/{ishikawaDiagram-YF4QCWOH.Ctn-zt6a.js → ishikawaDiagram-YF4QCWOH.js8qeS0h.js} +1 -1
- package/web/_astro/{journeyDiagram-JHISSGLW.DJhT8Ctp.js → journeyDiagram-JHISSGLW.CM6UK0a4.js} +1 -1
- package/web/_astro/{kanban-definition-UN3LZRKU.BY1QdejI.js → kanban-definition-UN3LZRKU.a6ihOzMd.js} +1 -1
- package/web/_astro/{linear.Di7YObSt.js → linear.CsIB2jFu.js} +1 -1
- package/web/_astro/{mermaid.core.CbxtJS3Q.js → mermaid.core.Br2Fo22q.js} +4 -4
- package/web/_astro/{mindmap-definition-RKZ34NQL.CxvR4g_J.js → mindmap-definition-RKZ34NQL.29inC1Mk.js} +1 -1
- package/web/_astro/{pieDiagram-4H26LBE5.jNWqnBHH.js → pieDiagram-4H26LBE5.C9CxG_Kf.js} +1 -1
- package/web/_astro/{quadrantDiagram-W4KKPZXB.BCp12MbA.js → quadrantDiagram-W4KKPZXB.6qo9MOJM.js} +1 -1
- package/web/_astro/{requirementDiagram-4Y6WPE33.Dxhm4TyR.js → requirementDiagram-4Y6WPE33.DN07zrP5.js} +1 -1
- package/web/_astro/{sankeyDiagram-5OEKKPKP.BtQXp4J9.js → sankeyDiagram-5OEKKPKP.BSw5o173.js} +1 -1
- package/web/_astro/{sequenceDiagram-3UESZ5HK.BWEM1R_Q.js → sequenceDiagram-3UESZ5HK.LJPzySKw.js} +1 -1
- package/web/_astro/{stateDiagram-AJRCARHV.BG3wUkWB.js → stateDiagram-AJRCARHV.ClNjEiKV.js} +1 -1
- package/web/_astro/{stateDiagram-v2-BHNVJYJU.BLtMeFVP.js → stateDiagram-v2-BHNVJYJU.c6Z-_WfX.js} +1 -1
- package/web/_astro/{timeline-definition-PNZ67QCA.D5fHo0az.js → timeline-definition-PNZ67QCA.CIMR-87j.js} +1 -1
- package/web/_astro/{vennDiagram-CIIHVFJN.0DcuMluU.js → vennDiagram-CIIHVFJN.BnRSRI9I.js} +1 -1
- package/web/_astro/{wardleyDiagram-YWT4CUSO.BZ-dxgHm.js → wardleyDiagram-YWT4CUSO.uN08C3gv.js} +1 -1
- package/web/_astro/{xychartDiagram-2RQKCTM6.Bg-XWF7z.js → xychartDiagram-2RQKCTM6.BjbEvfuq.js} +1 -1
- package/web/index.html +2 -2
- package/web/_astro/BoardApp.BQFbkeqq.js +0 -178
- package/web/_astro/BoardApp.CTkqrhWd.js +0 -1
- package/web/_astro/channel.Dsvulp7W.js +0 -1
- package/web/_astro/index.BVXdIsZV.css +0 -1
|
@@ -0,0 +1,395 @@
|
|
|
1
|
+
#!/usr/bin/env bun
|
|
2
|
+
/**
|
|
3
|
+
* verify-answer-lint — deterministic pre-verdict answer lint (0726 R3).
|
|
4
|
+
*
|
|
5
|
+
* Runs AFTER the verify agent exits and BEFORE `spur task verdict --from-answer`.
|
|
6
|
+
* The verifier owns the answer file (`.spur/run/<wbs>-verify-answer.txt`): it creates
|
|
7
|
+
* it with `Verdict: PARTIAL`, appends one complete requirement/AC row at a time, and
|
|
8
|
+
* only replaces the first verdict line once every row is certified. Because the file
|
|
9
|
+
* is now append-progress instead of a single captured blob, malformed rows can reach
|
|
10
|
+
* the verdict step — this lint rejects each invalid class with a row-level message:
|
|
11
|
+
*
|
|
12
|
+
* - missing, duplicate, or unknown requirement IDs (vs the task's Requirements)
|
|
13
|
+
* - AC IDs that do not exactly match the task's AC checklist label or a linked
|
|
14
|
+
* feature scenario title
|
|
15
|
+
* - status / evidence-type values the verdict parser would drop
|
|
16
|
+
* - empty evidence
|
|
17
|
+
*
|
|
18
|
+
* Compound evidence types (`test + command`) stay valid — normalization mirrors
|
|
19
|
+
* `packages/app/src/services/task-verdict.ts` exactly, so anything this lint accepts
|
|
20
|
+
* is also accepted by `spur task verdict --from-answer` (and vice versa).
|
|
21
|
+
*
|
|
22
|
+
* Exits non-zero on any finding, with bounded diagnostics (first 10). Writes nothing.
|
|
23
|
+
*
|
|
24
|
+
* Ships with the plugin to arbitrary projects; node-builtin only — no workspace imports.
|
|
25
|
+
*
|
|
26
|
+
* Usage:
|
|
27
|
+
* bun plugins/sp/scripts/verify-answer-lint.ts <wbs> --answer <path> [--spur-bin <path>]
|
|
28
|
+
*
|
|
29
|
+
* Env: SPUR_BIN
|
|
30
|
+
*/
|
|
31
|
+
|
|
32
|
+
import { execFileSync } from 'node:child_process';
|
|
33
|
+
import { existsSync, readFileSync } from 'node:fs';
|
|
34
|
+
import { fileURLToPath } from 'node:url';
|
|
35
|
+
|
|
36
|
+
// ─── CLI (same spur-bin chain as task-evidence-precheck.ts) ─────────────────
|
|
37
|
+
|
|
38
|
+
function usage(): never {
|
|
39
|
+
console.error('Usage: bun plugins/sp/scripts/verify-answer-lint.ts <wbs> --answer <path> [--spur-bin <path>]');
|
|
40
|
+
process.exit(1);
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
function defaultSpurBin(): string {
|
|
44
|
+
if (process.env.SPUR_BIN) return process.env.SPUR_BIN;
|
|
45
|
+
const local = fileURLToPath(new URL('../../../apps/cli/src/index.ts', import.meta.url));
|
|
46
|
+
if (existsSync(local)) return `bun ${local}`;
|
|
47
|
+
return 'spur';
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
function parseArgs(argv: string[]): { wbs: string; answer: string; spurBin: string } {
|
|
51
|
+
let spurBin = defaultSpurBin();
|
|
52
|
+
let wbs = '';
|
|
53
|
+
let answer = '';
|
|
54
|
+
let i = 0;
|
|
55
|
+
while (i < argv.length) {
|
|
56
|
+
const arg = argv[i];
|
|
57
|
+
if (arg === '--spur-bin') {
|
|
58
|
+
spurBin = argv[i + 1] ?? defaultSpurBin();
|
|
59
|
+
i += 2;
|
|
60
|
+
} else if (arg === '--answer') {
|
|
61
|
+
answer = argv[i + 1] ?? '';
|
|
62
|
+
i += 2;
|
|
63
|
+
} else if (!arg.startsWith('--')) {
|
|
64
|
+
wbs = arg;
|
|
65
|
+
i++;
|
|
66
|
+
} else {
|
|
67
|
+
i++;
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
if (!wbs || !answer) usage();
|
|
71
|
+
return { wbs, answer, spurBin };
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
function runSpur(spurBin: string, args: string[]): string {
|
|
75
|
+
const [file = 'spur', ...lead] = spurBin.split(/\s+/).filter(Boolean);
|
|
76
|
+
return execFileSync(file, [...lead, ...args], {
|
|
77
|
+
encoding: 'utf-8',
|
|
78
|
+
stdio: ['pipe', 'pipe', 'pipe'],
|
|
79
|
+
});
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// ─── Answer parsing — mirrors packages/app/src/services/task-verdict.ts ─────
|
|
83
|
+
|
|
84
|
+
interface ReqRow {
|
|
85
|
+
id: string;
|
|
86
|
+
status: string;
|
|
87
|
+
evidence: string;
|
|
88
|
+
line: number;
|
|
89
|
+
}
|
|
90
|
+
interface AcRow {
|
|
91
|
+
id: string;
|
|
92
|
+
status: string;
|
|
93
|
+
evidenceType: string;
|
|
94
|
+
evidence: string;
|
|
95
|
+
line: number;
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
function splitTableCells(line: string): string[] {
|
|
99
|
+
return line
|
|
100
|
+
.split(/(?<!\\)\|/)
|
|
101
|
+
.map((c) => c.replace(/\\\|/g, '|').trim())
|
|
102
|
+
.filter(Boolean);
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
function normalizeReqStatus(raw: string): string | null {
|
|
106
|
+
if (/\bMET\b/.test(raw)) return 'MET';
|
|
107
|
+
if (/\bPARTIAL\b/.test(raw)) return 'PARTIAL';
|
|
108
|
+
if (/\bUNMET\b/.test(raw)) return 'UNMET';
|
|
109
|
+
return null;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
function normalizeAcStatus(raw: string): string | null {
|
|
113
|
+
if (/\bMET\b/.test(raw)) return 'MET';
|
|
114
|
+
if (/\bPARTIAL\b/.test(raw)) return 'PARTIAL';
|
|
115
|
+
if (/\bUNMET\b/.test(raw)) return 'UNMET';
|
|
116
|
+
if (/\bN\/A\b/.test(raw) || /\bNA\b/.test(raw)) return 'N/A';
|
|
117
|
+
return null;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
function normalizeEvidenceTypeToken(normalized: string): string | null {
|
|
121
|
+
if (normalized === 'test') return 'test';
|
|
122
|
+
if (normalized === 'command') return 'command';
|
|
123
|
+
if (
|
|
124
|
+
normalized === 'static-ref' ||
|
|
125
|
+
normalized === 'static' ||
|
|
126
|
+
normalized === 'doc' ||
|
|
127
|
+
normalized === 'docs' ||
|
|
128
|
+
normalized === 'documentation'
|
|
129
|
+
) {
|
|
130
|
+
return 'static-ref';
|
|
131
|
+
}
|
|
132
|
+
if (normalized === 'manual-review' || normalized === 'manual') return 'manual-review';
|
|
133
|
+
if (normalized === 'llm-judge' || normalized === 'judge') return 'llm-judge';
|
|
134
|
+
if (normalized === 'n/a' || normalized === 'na') return 'n/a';
|
|
135
|
+
return null;
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
const EVIDENCE_TYPE_PRECEDENCE = ['test', 'command', 'static-ref', 'manual-review', 'llm-judge', 'n/a'] as const;
|
|
139
|
+
|
|
140
|
+
function normalizeEvidenceType(raw: string): string | null {
|
|
141
|
+
const normalized = raw.toLowerCase().trim();
|
|
142
|
+
const single = normalizeEvidenceTypeToken(normalized);
|
|
143
|
+
if (single !== null) return single;
|
|
144
|
+
const parts = normalized.split(/[+,/]/).filter((p) => p.trim());
|
|
145
|
+
const tokens = parts.map((part) => normalizeEvidenceTypeToken(part.trim())).filter((t) => t !== null);
|
|
146
|
+
if (tokens.length < 2 || tokens.length !== parts.length) return null;
|
|
147
|
+
return EVIDENCE_TYPE_PRECEDENCE.find((candidate) => tokens.includes(candidate)) ?? null;
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
interface AnswerTables {
|
|
151
|
+
verdict: { value: string; line: number } | null;
|
|
152
|
+
reqs: ReqRow[];
|
|
153
|
+
acs: AcRow[];
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
function parseAnswer(text: string): AnswerTables {
|
|
157
|
+
const out: AnswerTables = { verdict: null, reqs: [], acs: [] };
|
|
158
|
+
const lines = text.split('\n');
|
|
159
|
+
let reqTable = false;
|
|
160
|
+
let acTable = false;
|
|
161
|
+
|
|
162
|
+
for (let i = 0; i < lines.length; i++) {
|
|
163
|
+
const trimmed = lines[i]?.trim() ?? '';
|
|
164
|
+
const lineNo = i + 1;
|
|
165
|
+
|
|
166
|
+
// A markdown heading closes whichever table is open (mirrors the verdict parser).
|
|
167
|
+
if ((reqTable || acTable) && /^#{1,6}\s/.test(trimmed)) {
|
|
168
|
+
reqTable = false;
|
|
169
|
+
acTable = false;
|
|
170
|
+
continue;
|
|
171
|
+
}
|
|
172
|
+
if (!trimmed.startsWith('|')) continue;
|
|
173
|
+
const cells = splitTableCells(trimmed);
|
|
174
|
+
if (/^[-:]+$/.test(cells[0] ?? '')) continue;
|
|
175
|
+
|
|
176
|
+
const h0 = (cells[0] ?? '').toLowerCase();
|
|
177
|
+
const h1 = (cells[1] ?? '').toLowerCase();
|
|
178
|
+
|
|
179
|
+
// Requirement header: `| Req | Status | Evidence |` (id-like first cell + status column).
|
|
180
|
+
if (!reqTable && !acTable && cells.length >= 2) {
|
|
181
|
+
const idLike = h0.includes('req') || h0 === 'requirement' || h0 === 'r#' || h0 === 'r' || /^r\d+$/.test(h0);
|
|
182
|
+
if (idLike && (h1.includes('status') || h1 === 'verdict')) {
|
|
183
|
+
reqTable = true;
|
|
184
|
+
continue;
|
|
185
|
+
}
|
|
186
|
+
if (
|
|
187
|
+
(h0 === 'ac' || h0.includes('acceptance')) &&
|
|
188
|
+
h1.includes('status') &&
|
|
189
|
+
(cells[2] ?? '').toLowerCase().includes('evidence')
|
|
190
|
+
) {
|
|
191
|
+
acTable = true;
|
|
192
|
+
continue;
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
if (reqTable && cells.length >= 2) {
|
|
197
|
+
// An AC header following the requirement table closes it (mirrors the parser).
|
|
198
|
+
if ((h0 === 'ac' || h0.includes('acceptance')) && h1.includes('status')) {
|
|
199
|
+
reqTable = false;
|
|
200
|
+
acTable = true;
|
|
201
|
+
continue;
|
|
202
|
+
}
|
|
203
|
+
out.reqs.push({
|
|
204
|
+
id: cells[0] ?? '',
|
|
205
|
+
status: (cells[1] ?? '').toUpperCase(),
|
|
206
|
+
evidence: cells[2] ?? '',
|
|
207
|
+
line: lineNo,
|
|
208
|
+
});
|
|
209
|
+
continue;
|
|
210
|
+
}
|
|
211
|
+
if (acTable && cells.length >= 3) {
|
|
212
|
+
out.acs.push({
|
|
213
|
+
id: cells[0] ?? '',
|
|
214
|
+
status: cells[1] ?? '',
|
|
215
|
+
evidenceType: cells[2] ?? '',
|
|
216
|
+
evidence: cells[3] ?? '',
|
|
217
|
+
line: lineNo,
|
|
218
|
+
});
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
const verdictMatches = [...text.matchAll(/^\s*Verdict:\s*(\S+)\s*$/gim)];
|
|
223
|
+
if (verdictMatches.length === 1) {
|
|
224
|
+
out.verdict = {
|
|
225
|
+
value: verdictMatches[0]?.[1] ?? '',
|
|
226
|
+
line: text.slice(0, verdictMatches[0]?.index ?? 0).split('\n').length,
|
|
227
|
+
};
|
|
228
|
+
}
|
|
229
|
+
return out;
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
// ─── Task-side identity extraction ───────────────────────────────────────────
|
|
233
|
+
|
|
234
|
+
function sectionBetween(text: string, heading: string): string {
|
|
235
|
+
const marker = new RegExp(`^#{1,6}\\s+${heading}\\s*$`, 'im');
|
|
236
|
+
const match = marker.exec(text);
|
|
237
|
+
if (!match || match.index === undefined) return '';
|
|
238
|
+
const rest = text.slice(match.index);
|
|
239
|
+
const lineEnd = rest.indexOf('\n');
|
|
240
|
+
const afterHeading = lineEnd === -1 ? '' : rest.slice(lineEnd + 1);
|
|
241
|
+
const next = afterHeading.search(/^#{1,6}\s/m);
|
|
242
|
+
return next === -1 ? afterHeading : afterHeading.slice(0, next);
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
function extractRequirementIds(taskContent: string): string[] {
|
|
246
|
+
const section = sectionBetween(taskContent, 'Requirements');
|
|
247
|
+
const ids = new Set<string>();
|
|
248
|
+
// Corpus forms: bold-wrapped (`**R1. Title.**`, bare `**R1**`); right after a list
|
|
249
|
+
// marker with optional checkbox (`- [ ] R1.` — the dominant corpus form, `- R1. Title.`,
|
|
250
|
+
// `- R1:`) or a bare checkbox with the marker omitted (`[x] R1.`); line-start (`R1:`).
|
|
251
|
+
// The marker and the checkbox are never both optional — that would match bare prose
|
|
252
|
+
// (`R1 is …`) and fabricate declarations. Sub-IDs (`R1.1`) match in every form.
|
|
253
|
+
for (const m of section.matchAll(/\*\*(R\d+(?:\.\d+)*)\b/g)) ids.add(m[1] ?? '');
|
|
254
|
+
for (const m of section.matchAll(/^(?:[-*]\s+(?:\[[ xX]\]\s+)?|\[[ xX]\]\s+)(R\d+(?:\.\d+)*)/gm))
|
|
255
|
+
ids.add(m[1] ?? '');
|
|
256
|
+
for (const m of section.matchAll(/^(R\d+(?:\.\d+)*)\s*[.:]/gm)) ids.add(m[1] ?? '');
|
|
257
|
+
return [...ids];
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
function extractAcIdentities(taskContent: string, featureContent: string | null): string[] {
|
|
261
|
+
const identities = new Set<string>();
|
|
262
|
+
const section = sectionBetween(taskContent, 'Acceptance Criteria');
|
|
263
|
+
// Checkbox labels (`- [x] AC1 (R1): …`, 0726) and plain bullets (`- AC1: Given …`,
|
|
264
|
+
// 0713/0727) both yield the label text up to `:` plus its leading token.
|
|
265
|
+
for (const m of section.matchAll(/^[-*]\s+(?:\[[ xX]\]\s+)?(.+?)\s*(?::|$)/gm)) {
|
|
266
|
+
const label = (m[1] ?? '').trim();
|
|
267
|
+
if (!label) continue;
|
|
268
|
+
identities.add(label);
|
|
269
|
+
const leading = label.split(/\s+/)[0] ?? '';
|
|
270
|
+
if (leading && leading !== label) identities.add(leading);
|
|
271
|
+
}
|
|
272
|
+
if (featureContent !== null) {
|
|
273
|
+
for (const m of featureContent.matchAll(/^[ \t]*Scenario:\s*(.+)\s*$/gm)) {
|
|
274
|
+
const title = (m[1] ?? '').trim();
|
|
275
|
+
if (title) identities.add(title);
|
|
276
|
+
}
|
|
277
|
+
}
|
|
278
|
+
return [...identities];
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
// ─── Main ────────────────────────────────────────────────────────────────────
|
|
282
|
+
|
|
283
|
+
function main(): void {
|
|
284
|
+
const { wbs, answer, spurBin } = parseArgs(process.argv.slice(2));
|
|
285
|
+
const findings: string[] = [];
|
|
286
|
+
const add = (msg: string): void => {
|
|
287
|
+
if (findings.length < 10) findings.push(msg);
|
|
288
|
+
};
|
|
289
|
+
|
|
290
|
+
if (!existsSync(answer)) {
|
|
291
|
+
console.error(`verify-answer-lint: FAIL — answer file not found: ${answer}`);
|
|
292
|
+
process.exit(1);
|
|
293
|
+
}
|
|
294
|
+
const raw = readFileSync(answer, 'utf8');
|
|
295
|
+
if (!raw.trim()) {
|
|
296
|
+
console.error(`verify-answer-lint: FAIL — answer file is empty: ${answer}`);
|
|
297
|
+
process.exit(1);
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
const tables = parseAnswer(raw);
|
|
301
|
+
if (tables.verdict === null) {
|
|
302
|
+
add('no `Verdict:` line (expected exactly one `Verdict: PASS|PARTIAL|FAIL` line)');
|
|
303
|
+
} else if (!/^(PASS|PARTIAL|FAIL)$/i.test(tables.verdict.value)) {
|
|
304
|
+
add(`line ${tables.verdict.line}: invalid Verdict value "${tables.verdict.value}" (PASS | PARTIAL | FAIL)`);
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
let taskContent = '';
|
|
308
|
+
let featureId = '';
|
|
309
|
+
try {
|
|
310
|
+
const task = JSON.parse(runSpur(spurBin, ['task', 'show', wbs, '--json'])) as {
|
|
311
|
+
content?: string;
|
|
312
|
+
body?: string;
|
|
313
|
+
feature_id?: string;
|
|
314
|
+
frontmatter?: { feature_id?: string };
|
|
315
|
+
};
|
|
316
|
+
taskContent = task.content ?? task.body ?? '';
|
|
317
|
+
featureId = task.feature_id ?? task.frontmatter?.feature_id ?? '';
|
|
318
|
+
} catch {
|
|
319
|
+
console.error(`verify-answer-lint: FAIL — could not fetch task ${wbs} via ${spurBin}`);
|
|
320
|
+
process.exit(1);
|
|
321
|
+
}
|
|
322
|
+
if (!taskContent) {
|
|
323
|
+
console.error(`verify-answer-lint: FAIL — task ${wbs} returned no content via ${spurBin}`);
|
|
324
|
+
process.exit(1);
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
let featureContent: string | null = null;
|
|
328
|
+
if (featureId) {
|
|
329
|
+
try {
|
|
330
|
+
const feature = JSON.parse(runSpur(spurBin, ['feature', 'show', featureId, '--json'])) as {
|
|
331
|
+
content?: string;
|
|
332
|
+
};
|
|
333
|
+
featureContent = feature.content ?? '';
|
|
334
|
+
} catch {
|
|
335
|
+
featureContent = null; // checklist labels still apply; scenario titles unavailable
|
|
336
|
+
}
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
const reqIds = extractRequirementIds(taskContent);
|
|
340
|
+
const acIdentities = extractAcIdentities(taskContent, featureContent);
|
|
341
|
+
|
|
342
|
+
// Requirement rows: completeness, no unknowns, no duplicates, valid status, non-empty evidence.
|
|
343
|
+
const seenReq = new Set<string>();
|
|
344
|
+
for (const row of tables.reqs) {
|
|
345
|
+
if (!reqIds.includes(row.id))
|
|
346
|
+
add(`line ${row.line}: unknown requirement ID "${row.id}" (task declares: ${reqIds.join(', ') || 'none'})`);
|
|
347
|
+
else if (seenReq.has(row.id)) add(`line ${row.line}: duplicate requirement row "${row.id}"`);
|
|
348
|
+
seenReq.add(row.id);
|
|
349
|
+
if (normalizeReqStatus(row.status) === null)
|
|
350
|
+
add(`line ${row.line}: "${row.id}" invalid status "${row.status}" (MET | PARTIAL | UNMET)`);
|
|
351
|
+
if (!row.evidence.trim()) add(`line ${row.line}: "${row.id}" has empty evidence`);
|
|
352
|
+
}
|
|
353
|
+
for (const id of reqIds) {
|
|
354
|
+
if (!seenReq.has(id)) add(`missing requirement row for "${id}"`);
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
// AC rows: identity must exactly match a checklist label/token or a scenario title;
|
|
358
|
+
// status and evidence type must normalize; evidence non-empty. AC completeness is the
|
|
359
|
+
// verifier's authoring contract, not a lint rejection class (0726 R3).
|
|
360
|
+
const seenAc = new Set<string>();
|
|
361
|
+
for (const row of tables.acs) {
|
|
362
|
+
if (!acIdentities.includes(row.id)) {
|
|
363
|
+
add(
|
|
364
|
+
`line ${row.line}: AC ID "${row.id.slice(0, 60)}" matches no task AC checklist label or scenario title`,
|
|
365
|
+
);
|
|
366
|
+
} else if (seenAc.has(row.id)) {
|
|
367
|
+
add(`line ${row.line}: duplicate AC row "${row.id.slice(0, 60)}"`);
|
|
368
|
+
}
|
|
369
|
+
seenAc.add(row.id);
|
|
370
|
+
if (normalizeAcStatus(row.status) === null)
|
|
371
|
+
add(`line ${row.line}: invalid AC status "${row.status}" (MET | PARTIAL | UNMET | N/A)`);
|
|
372
|
+
if (normalizeEvidenceType(row.evidenceType) === null)
|
|
373
|
+
add(
|
|
374
|
+
`line ${row.line}: invalid evidence type "${row.evidenceType}" (test | command | static-ref | manual-review | llm-judge | n/a, or a + compound)`,
|
|
375
|
+
);
|
|
376
|
+
if (!row.evidence.trim()) add(`line ${row.line}: AC "${row.id.slice(0, 40)}" has empty evidence`);
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
if (findings.length > 0) {
|
|
380
|
+
console.error(
|
|
381
|
+
`verify-answer-lint: FAIL — ${findings.length}${findings.length >= 10 ? '+' : ''} finding(s) in ${answer}`,
|
|
382
|
+
);
|
|
383
|
+
for (const f of findings) console.error(` ${f}`);
|
|
384
|
+
process.exit(1);
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
const reqCount = tables.reqs.length;
|
|
388
|
+
const acCount = tables.acs.length;
|
|
389
|
+
console.error(
|
|
390
|
+
`verify-answer-lint: PASS — ${reqCount} requirement row(s), ${acCount} AC row(s), verdict ${tables.verdict?.value ?? '?'}`,
|
|
391
|
+
);
|
|
392
|
+
process.exit(0);
|
|
393
|
+
}
|
|
394
|
+
|
|
395
|
+
main();
|
|
@@ -81,7 +81,7 @@ surfaces is what keeps that gate quiet.
|
|
|
81
81
|
## Implement scope: do not run the project quality gate
|
|
82
82
|
|
|
83
83
|
During implement, the pipeline's `test` hop runs `${vars.qualityGateCmd}` (the full project gate:
|
|
84
|
-
`bun run
|
|
84
|
+
`bun run spur-check`) immediately after this step and is the gate that actually
|
|
85
85
|
decides pass/fail. Running it inside implement is pure redundancy — it cannot change the outcome and
|
|
86
86
|
only burns wall clock and context budget.
|
|
87
87
|
|
|
@@ -247,9 +247,10 @@ completion gate (`PARTIAL`/`FAIL` route the pipeline to `failed`).
|
|
|
247
247
|
### Step 10 — Emit the verdict artifact (the only verify output)
|
|
248
248
|
|
|
249
249
|
Assemble the evidence and **emit the canonical verdict artifact** — verification writes no task
|
|
250
|
-
section (F92 0593 R1). Under the pipeline
|
|
251
|
-
`.spur/run/<wbs>-verify-answer.txt
|
|
252
|
-
`.spur/run/<wbs>-verdict.json`, and
|
|
250
|
+
section (F92 0593 R1). Under the pipeline **you** write
|
|
251
|
+
`.spur/run/<wbs>-verify-answer.txt` (0726 R3: host `expectFile`, never captured/overwritten); a
|
|
252
|
+
deterministic shell step lints it and derives `.spur/run/<wbs>-verdict.json`, and `record`
|
|
253
|
+
transcribes `## Testing` from it.
|
|
253
254
|
|
|
254
255
|
**Standalone** (`/sp:dev-verify` outside the pipeline), write the artifact yourself, then invoke
|
|
255
256
|
the deterministic Testing writer `spur task record` (section authorship never happens here):
|
|
@@ -306,14 +307,14 @@ The per-requirement traceability table MUST use `| Req | Status | Evidence |` (e
|
|
|
306
307
|
The parser is tolerant of these variants (defense-in-depth), but the authoring contract is
|
|
307
308
|
canonical.
|
|
308
309
|
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
|
|
313
|
-
|
|
314
|
-
|
|
310
|
+
Under the pipeline, the verifier owns the answer file `.spur/run/<wbs>-verify-answer.txt` (0726
|
|
311
|
+
R3): `Verdict: PARTIAL` first, append one row at a time, replace the verdict line only when all
|
|
312
|
+
rows are certified — interruptions leave lintable partial rows; retries fill only missing IDs.
|
|
313
|
+
Host: `expectFile` → `verify-answer-lint.ts` → verdict + `spur task check` (R9). Vocabularies,
|
|
314
|
+
rejection classes, and the AC identity rule: `references/verdict-schema.md`. Sections follow the
|
|
315
|
+
Step 10 contract.
|
|
315
316
|
|
|
316
|
-
**Standalone** (
|
|
317
|
+
**Standalone** (outside the pipeline — no host lint/derive step), write the answer file and
|
|
317
318
|
artifact yourself; shape and field-by-field contract in
|
|
318
319
|
[references/verdict-schema.md](references/verdict-schema.md):
|
|
319
320
|
|
|
@@ -106,6 +106,16 @@ For answer files, emit a matching parseable table:
|
|
|
106
106
|
| Scenario: CLI emits JSON | MET | test | `apps/cli/tests/foo.test.ts:42` |
|
|
107
107
|
```
|
|
108
108
|
|
|
109
|
+
**Authoring contract under the pipeline (0726 R3).** The verifier owns the answer file: write
|
|
110
|
+
`Verdict: PARTIAL` first, append one complete row at a time, and replace the first verdict line only
|
|
111
|
+
after every row is certified. `verify-answer-lint.ts` gates the file before `spur task verdict
|
|
112
|
+
--from-answer` and rejects, with row-level diagnostics: missing/duplicate/unknown requirement IDs,
|
|
113
|
+
AC ids that do not exactly match a task AC checklist label (or its leading token, e.g. `AC1`) or a
|
|
114
|
+
linked feature scenario title, invalid status (`MET | PARTIAL | UNMET` for requirements;
|
|
115
|
+
`N/A` additionally allowed for AC), invalid evidence type (`test | command | static-ref |
|
|
116
|
+
manual-review | llm-judge | n/a`, or a `+` compound), and empty evidence. Interrupted runs keep the
|
|
117
|
+
rows that pass the lint and complete only the missing IDs on retry.
|
|
118
|
+
|
|
109
119
|
## Checks evidence
|
|
110
120
|
|
|
111
121
|
Wave C verification can emit the following additive `checks[]` rows:
|
|
@@ -57,7 +57,7 @@ granularity knobs (min/target/force-split hours), and parent/umbrella-task conve
|
|
|
57
57
|
spur task batch-create --file decomposition.json # bare JSON array; atomic, all-or-nothing
|
|
58
58
|
```
|
|
59
59
|
|
|
60
|
-
Validate locally against `
|
|
60
|
+
Validate locally against `task-batch.schema.json` (runtime SSOT: the Zod
|
|
61
61
|
`taskBatchSchema`) before invoking the CLI — a single violation rejects the entire batch. The gate is
|
|
62
62
|
the only proof the decomposition is well-formed; never hand-write task files to bypass it.
|
|
63
63
|
|
|
@@ -22,15 +22,16 @@ live in `packages/app/src/services/history-service.ts` (`FanOutResult`, `DailyRe
|
|
|
22
22
|
| `analyze` | Aggregate imported rows and write a versioned forensic artifact | `--since <iso>` `--until <iso>` `--source <source>` `--session <id>` `--run <runId>` `--task <wbs>` `--top <n>` `--out <path>` `--json` |
|
|
23
23
|
| `report [path]` | Purely render an existing artifact; default to `latest.json` | `--mode <name>` `--task <wbs>` `--top <n>` `--json` |
|
|
24
24
|
| `daily` | Run import-all → analyze → artifact → 90-day report pruning once | `--since <iso>` `--until <iso>` `--root <path>` `--source-timeout <ms>` `--mode <name>` `--json` |
|
|
25
|
+
| `reset` | Destructively wipe every `history_*` table for a clean re-import; refuses without `--yes` | `--yes` `--json` |
|
|
25
26
|
|
|
26
27
|
Every JSON-capable verb also advertises `--json-envelope`; use the facade's machine-output contract.
|
|
27
28
|
|
|
28
29
|
## `import` - isolated fan-out
|
|
29
30
|
|
|
30
31
|
```bash
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
32
|
+
spur history import --source all --dry-run --json
|
|
33
|
+
spur history import --source codex --mode incremental --json
|
|
34
|
+
spur history import --source codex --file session.jsonl --mode force-file --json
|
|
34
35
|
```
|
|
35
36
|
|
|
36
37
|
- `--source all` and a single source use the same per-source fan-out path. A failed/timed-out source
|
|
@@ -38,9 +39,10 @@ bun run apps/cli/src/index.ts history import --source codex --file session.jsonl
|
|
|
38
39
|
- Modes are `incremental`, `full`, and `force-file`. `--file` with the default `all` source is a
|
|
39
40
|
usage error. `--file --mode full` requires `--dry-run`; use `force-file` for a real single-file
|
|
40
41
|
write.
|
|
41
|
-
- JSON contains `entries`, `warnings`, `exitCode`, and CLI/importer `provenance`.
|
|
42
|
-
|
|
43
|
-
`spur` that may be
|
|
42
|
+
- JSON contains `entries`, `warnings`, `exitCode`, and CLI/importer `provenance`. Record that
|
|
43
|
+
provenance for any real-data validation. **When developing Spur itself**, invoke the source-local
|
|
44
|
+
CLI (`bun run apps/cli/src/index.ts history import …`) instead of a global `spur` that may be a
|
|
45
|
+
stale published bundle; in every other project the installed `spur` is the CLI.
|
|
44
46
|
- Exit `0` when every source is clean/empty, `2` for a mixed failure or any degraded source, and `1`
|
|
45
47
|
when all sources fail. CLI usage guards also exit `1` on this noun.
|
|
46
48
|
|
|
@@ -463,8 +463,8 @@ is the procedure. The backing is a combination of git CLI, `spur` CLI, and agent
|
|
|
463
463
|
6. Run `bun run test`. Collect all failures.
|
|
464
464
|
7. If tests are green, done.
|
|
465
465
|
8. **Test fix loop:** for each failure, diagnose (test bug vs implementation bug), apply the fix, re-run the **failing test only** (`bun test <file> --test-name-pattern "<test>"`). Do NOT re-run the full suite per fix — it is the dominant loop cost (task 0436 R2).
|
|
466
|
-
9. **Confirming run (at most once).** After all fixes, run `bun run format && bun run lint && bun run test` **at most once** to confirm. If it passes, the hop is done.
|
|
467
|
-
10. **Pipeline-awareness (R4, task 0483).**
|
|
466
|
+
9. **Confirming run — standalone invocation only (at most once).** After all fixes, run `bun run format && bun run lint && bun run test` **at most once** to confirm. If it passes, the hop is done. **Skip this step entirely when `--gate-log` is set** (see step 10) — under the pipeline nothing consumes its verdict.
|
|
467
|
+
10. **Pipeline-awareness (R4, task 0483; confirming run dropped 2026-08-31).** `--gate-log` is set only by the pipeline's `test-fix` hop, so treat it as the pipeline signal. There, `test-recheck` runs the full `${vars.qualityGateCmd}` gate immediately after this hop returns — that is the **deciding** run, the only one that writes PASS to `.spur/run/<wbs>-test-gate.status`. A confirming run inside this hop decides nothing and costs a second full gate (~105 s here) to answer a question `test-recheck` is about to answer anyway. So under `--gate-log`: **run no full gate at all.** Your self-check is what step 8 already gives you — the failing tests you re-ran individually — plus **one `bun run lint`** (~7 s: Biome + typecheck) to catch type or lint breakage your fixes introduced. Then return and let `test-recheck` judge. `test-recheck` opens with the same cheap `${vars.gateProbeCmd}` probe, so a still-red tree is caught in seconds rather than by a full suite. Historical anti-pattern this replaces: 0482 ran the gate 3× plus a standalone `bun run test`, and `test-recheck` then ran it a 5th time.
|
|
468
468
|
11. Report: list what was fixed (file + one-line summary per fix). If any error could not be resolved, report it explicitly — do not suppress.
|
|
469
469
|
- **Invariants:** Never bypass with `--no-verify`, `--force`, or new `biome-ignore`/`eslint-disable` suppressions. Never skip or `.skip` a test to make the suite green. Fix the root cause, not the symptom. Never claim green on `bun run lint` alone — a formatter-only diff passes `lint` but fails the formatter; run `bun run format` (or assert it produces no diff) before declaring the gate clean. **Never re-run the full gate more than once per confirming pass** (R4) — use targeted probes during the fix loops and let the pipeline's `test-recheck` state be the deciding run.
|
|
470
470
|
- **MANDATORY Exit Condition.** The ONLY way to complete successfully:
|
|
@@ -46,7 +46,7 @@ one thing and yields, so the **pipeline (not the agent) owns the loop**.
|
|
|
46
46
|
| Stage | Operation | Defined in |
|
|
47
47
|
| ------- | ----------- | ------------ |
|
|
48
48
|
| `implement` | `/sp:dev-run --mode implement <wbs>` — write the code that satisfies the task; author `## Solution`. | [dev-operations.md §4 run](dev-operations.md) → `sp:code-implementation` |
|
|
49
|
-
| `test` → (`test-fix` ↔ `test-recheck`) → `review` \| `failed` | **Project quality gate** (not `/sp:dev-unit`). Soft shell probe of `${vars.qualityGateCmd}` (default `bun run
|
|
49
|
+
| `test` → (`test-fix` ↔ `test-recheck`) → `review` \| `failed` | **Project quality gate** (not `/sp:dev-unit`). Soft shell probe of `${vars.qualityGateCmd}` (default `bun run spur-check`) — green path pays **one** full gate run. On FAIL: bounded `/sp:dev-fixall` loop (`qualityGateMaxFixAttempts`, default 2) with soft recheck; exhausted attempts route to pipeline `failed`. `/sp:dev-unit` remains **coverage gap-fill** (router C3/C5 / standalone). | [dev-operations.md §10 fixall](dev-operations.md); unit op still §1 |
|
|
50
50
|
| `review` | `/sp:dev-review <wbs>` — SECUA-framework review of the diff. | [dev-operations.md §2 review](dev-operations.md) |
|
|
51
51
|
| `verify` | `sp:code-verification` — requirements traceability + verdict. | [dev-operations.md §3 verify](dev-operations.md) |
|
|
52
52
|
|
|
@@ -97,6 +97,7 @@ cache-conservation discipline (`plugins/sp/skills/dogfood-testing/references/mon
|
|
|
97
97
|
## Step 2: Pipeline run
|
|
98
98
|
|
|
99
99
|
> **Pre-launch size-gate pre-check (R1 / 0478).** Before launching `spur workflow run task-pipeline.yaml`, probe the task's `## Plan` checklist item count (`spur task show <wbs> --json`). The default cap is 8 items (`maxImplementPlanItems: 8`). If the plan item count exceeds 8:
|
|
100
|
+
>
|
|
100
101
|
> - Without `--auto`: warn the operator before calling `spur workflow run` and prompt for confirmation or a plan-item override via `--vars '{"maxImplementPlanItems":"<count>"}'`.
|
|
101
102
|
> - With `--auto`: automatically append `"maxImplementPlanItems": "<count>"` to `--vars` and log a single-line notice (e.g. `Notice: task <wbs> has N plan items (>8 default cap); injecting maxImplementPlanItems override`).
|
|
102
103
|
|
|
@@ -313,6 +314,12 @@ the partial work still in the working tree. The failure output names the partial
|
|
|
313
314
|
[`done-housekeeping.md`](done-housekeeping.md) F6 — it carries the provenance obligations
|
|
314
315
|
(honest `done_reason`, verdict regeneration).
|
|
315
316
|
|
|
317
|
+
**Inline path (task 0727).** There is no `<runId>-implement-partial.md` artifact on the inline
|
|
318
|
+
driver path — it is written only by the subprocess `agent.run` action. The inline equivalent is
|
|
319
|
+
the dispatch-timeout contract in [`inline-pipeline-driver.md`](inline-pipeline-driver.md):
|
|
320
|
+
resume from the partial tree, never restart the stage inline; the partial working tree is the
|
|
321
|
+
recovery input.
|
|
322
|
+
|
|
316
323
|
**3. Match the executor to the size (task 0487 R3).** Size and executor capability are one
|
|
317
324
|
decision, not two. **≥ 6 requirements or ≥ 9 Plan items → a `reviewer`-role executor (the
|
|
318
325
|
`reviewer` row of [`roles.md`](../../../references/roles.md) declares the floor), or split the
|
|
@@ -52,7 +52,7 @@ Entered before `spur task batch-create --file <json>` (idea-pipeline `batch-crea
|
|
|
52
52
|
decomposition gate `/sp:dev-plan` also routes through since D5-K).
|
|
53
53
|
|
|
54
54
|
- [ ] The batch JSON is a bare array (not an object with a `tasks` key).
|
|
55
|
-
- [ ] Each entry validates locally against `
|
|
55
|
+
- [ ] Each entry validates locally against `task-batch.schema.json` (`additionalProperties: false` — no unknown keys).
|
|
56
56
|
- [ ] Each entry has a non-empty `name` (required).
|
|
57
57
|
- [ ] `feature_id` matches an existing feature (or is intentionally deferred with operator awareness).
|
|
58
58
|
- [ ] `parent_wbs` is set when the task is a child of a decomposition parent.
|
|
@@ -72,6 +72,16 @@ Entered before `task-pipeline.yaml` `precheck` state runs `spur task check <wbs>
|
|
|
72
72
|
- [ ] The `## Plan` section is an ordered checklist (not prose).
|
|
73
73
|
- [ ] The `## Design` section, if present, does not contradict the parent feature's design.
|
|
74
74
|
- [ ] No `TODO`, `TBD`, or `???` placeholders in Requirements, AC, Design, or Plan.
|
|
75
|
+
- [ ] The evidence-channel precheck (0726 R2) status file is consulted by the
|
|
76
|
+
pipeline guard: `plugins/sp/scripts/task-evidence-precheck.ts` parses the task
|
|
77
|
+
content for an exact `evidence-channel: history_tool_call.args_raw[pi]`
|
|
78
|
+
declaration and, when present, counts live pi rows with `args_raw` on
|
|
79
|
+
`.spur/spur.db` via bun:sqlite. Tasks without a declaration pass without opening
|
|
80
|
+
SQLite; unknown declarations, a missing database/table, and a zero count write
|
|
81
|
+
FAIL. Both precheck guard conjuncts (`precheck-size.status` and
|
|
82
|
+
`precheck-evidence.status`) must read PASS — tasks declaring a live-data
|
|
83
|
+
evidence channel must import real history (safe importer, non-dry-run) before
|
|
84
|
+
implementation begins.
|
|
75
85
|
|
|
76
86
|
## review gate
|
|
77
87
|
|
|
@@ -90,6 +100,12 @@ Entered before `task-pipeline.yaml` `review` state dispatches `sp:code-verificat
|
|
|
90
100
|
|
|
91
101
|
Entered before `task-pipeline.yaml` `verify` state produces a task verdict.
|
|
92
102
|
|
|
103
|
+
- [ ] The verify answer file (`.spur/run/<wbs>-verify-answer.txt`) is lint-clean before
|
|
104
|
+
verdict derivation: `plugins/sp/scripts/verify-answer-lint.ts <wbs>` (0726 R3)
|
|
105
|
+
rejects missing/duplicate/unknown R IDs, AC identities that are not an exact task
|
|
106
|
+
checklist label or linked-feature scenario title, invalid status/evidence-type
|
|
107
|
+
values, and empty evidence on any row. A lint failure fails the verify
|
|
108
|
+
step (fail-closed) before `spur task verdict` runs.
|
|
93
109
|
- [ ] `spur task check <wbs> --strict-core --json` returns PASS.
|
|
94
110
|
- [ ] Every AC scenario has a corresponding verify command that exited 0.
|
|
95
111
|
- [ ] The `## Solution` section is filled (not the placeholder comment).
|
|
@@ -41,6 +41,12 @@ command, skill, script, or second workflow.
|
|
|
41
41
|
the YAML parsed in step 1, shown only for the active state.
|
|
42
42
|
- **Refresh cadence** = stage boundaries only (when the current state changes after a transition),
|
|
43
43
|
never per action.
|
|
44
|
+
- **Transition reconciliation (task 0727)** = at every stage boundary the host must
|
|
45
|
+
**mark the finished stage completed and the next stage in_progress** in the host todo list.
|
|
46
|
+
This reconciliation is **host-owned and execution-surface-independent**: it fires identically
|
|
47
|
+
whether the stage ran via native subagent, host-inline execution, or the post-dispatch host
|
|
48
|
+
fallback, so a run can never terminate with earlier stages stuck `in_progress` (task 0726
|
|
49
|
+
ended 0/11 with precheck and implement still open).
|
|
44
50
|
- **Source of truth** = the CLI projection for layer 1; the YAML parsed in step 1 for layer 2.
|
|
45
51
|
Never hand-copy or hand-derive the state list into the driver, a command, a skill, or a script.
|
|
46
52
|
5. For task execution only, record lifecycle provenance before entering the FSM:
|
|
@@ -119,6 +125,19 @@ before the subagent starts, log the reason and use host fallback. If a started s
|
|
|
119
125
|
leaves invalid artifacts, do **not** replay the stage in the host — follow the YAML error policy so
|
|
120
126
|
partial mutations are not duplicated.
|
|
121
127
|
|
|
128
|
+
**Timeout boundary (task 0727):** a dispatched subagent is governed by
|
|
129
|
+
**the host platform's subagent limit, not the YAML timeoutMs** — `timeoutMs` stays not-applicable
|
|
130
|
+
for host execution only — and before dispatch the driver must
|
|
131
|
+
**record the governing timeout boundary and its source before dispatch** in the run log
|
|
132
|
+
(e.g. `host timeout <ms> (<platform subagent limit|yaml timeoutMs>)`). If the dispatch reaches that
|
|
133
|
+
boundary, **a dispatch timeout is a started-subagent failure**: the no-replay rule above and the
|
|
134
|
+
stage's declared YAML error policy govern (implement's default `fail` policy routes the run to
|
|
135
|
+
`failed`); it is never a host re-execution. Recovery follows the
|
|
136
|
+
[timed-out implement runbook](execution-workflow.md)'s inline-path equivalent:
|
|
137
|
+
**resume from the partial tree, never restart the stage inline** — no
|
|
138
|
+
`<runId>-implement-partial.md` artifact is written on this path, so the partial working tree
|
|
139
|
+
itself is the recovery input.
|
|
140
|
+
|
|
122
141
|
**Host-owned interaction:** the host alone executes operator-confirmation actions, owns
|
|
123
142
|
`pause: true`, and surfaces approve/taste/ask decisions. A subagent that discovers missing authority
|
|
124
143
|
or an operator decision returns a blocker; the host pauses at the current state and presents it. The
|
|
@@ -129,6 +148,12 @@ subagent form above) to `.spur/run/<run-id>.log`, where `<id>` is the current YA
|
|
|
129
148
|
log start/failure and the ignored timeout value so an inline run remains auditable without
|
|
130
149
|
fabricating an `AgentRunTracedResult`.
|
|
131
150
|
|
|
151
|
+
Run-log stamps (task 0727): every appended line is prefixed with an **ISO-8601 UTC** timestamp
|
|
152
|
+
(`YYYY-MM-DDTHH:MM:SSZ`, e.g. `2026-08-31T17:51:11Z`); the exact-template provenance lines above
|
|
153
|
+
keep their exact content after the stamp prefix. This normalization is contractual:
|
|
154
|
+
**bare local-clock stamps are prohibited** — a hand-appended `[stage 12:31]` form mixes timezones
|
|
155
|
+
in one file and makes the run unauditable (task 0726 mixed both forms).
|
|
156
|
+
|
|
132
157
|
Transition guards are not advisory. Execute the declared guard exactly, in order, with the same
|
|
133
158
|
resolved variables and artifacts. `--no-lifecycle` remains bookkeeping only; the YAML's task checks,
|
|
134
159
|
verdict gate, record step, and done guard all remain authoritative.
|