@gobing-ai/spur 0.3.69 → 0.3.70

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/config/corpus-baseline.json +0 -18
  3. package/config/plugin-scripts.json +8 -0
  4. package/config/workflow-composition-baseline.json +123 -377
  5. package/config/workflows/task-pipeline.yaml +32 -3
  6. package/package.json +9 -9
  7. package/plugins/sp/plugin.json +1 -1
  8. package/plugins/sp/scripts/task-evidence-precheck.ts +181 -0
  9. package/plugins/sp/scripts/verify-answer-lint.ts +386 -0
  10. package/plugins/sp/skills/code-verification/SKILL.md +11 -10
  11. package/plugins/sp/skills/code-verification/references/verdict-schema.md +10 -0
  12. package/plugins/sp/skills/spur-cli/references/history.md +1 -0
  13. package/plugins/sp/skills/spur-dev/references/gate-checklists.md +16 -0
  14. package/spur.js +1262 -260
  15. package/web/_astro/BoardApp.DHj-03Dp.js +1 -0
  16. package/web/_astro/{BoardApp.BQFbkeqq.js → BoardApp.DOadeEJV.js} +77 -77
  17. package/web/_astro/{TaskDetail.Dl2Eaj1w.js → TaskDetail.DKzpkDj5.js} +1 -1
  18. package/web/_astro/{arc.uG14rp8A.js → arc.Bsa0gprH.js} +1 -1
  19. package/web/_astro/{architectureDiagram-3BPJPVTR.Dye6uD_x.js → architectureDiagram-3BPJPVTR.LOUZCDBC.js} +1 -1
  20. package/web/_astro/{blockDiagram-GPEHLZMM.B9Pkh7Hb.js → blockDiagram-GPEHLZMM.CGee4isG.js} +1 -1
  21. package/web/_astro/{c4Diagram-AAUBKEIU.C2x7SC_X.js → c4Diagram-AAUBKEIU.B2CrQJ_O.js} +1 -1
  22. package/web/_astro/channel.BGL7KSHC.js +1 -0
  23. package/web/_astro/{chunk-2J33WTMH.D2p4-nWk.js → chunk-2J33WTMH.DJvf8RiP.js} +1 -1
  24. package/web/_astro/{chunk-4BX2VUAB.S-6jf33o.js → chunk-4BX2VUAB.B_l35O-Y.js} +1 -1
  25. package/web/_astro/{chunk-55IACEB6.DVXj4Fdh.js → chunk-55IACEB6.DqbxMswU.js} +1 -1
  26. package/web/_astro/{chunk-727SXJPM.Dz689FMN.js → chunk-727SXJPM.D_7ZGNjE.js} +1 -1
  27. package/web/_astro/{chunk-AQP2D5EJ.KxYj5TnI.js → chunk-AQP2D5EJ.Cq6yZ0s8.js} +1 -1
  28. package/web/_astro/{chunk-FMBD7UC4.itTQyHQB.js → chunk-FMBD7UC4.BWrlUPYi.js} +1 -1
  29. package/web/_astro/{chunk-ND2GUHAM.euSrbJf5.js → chunk-ND2GUHAM.BPLaXFZH.js} +1 -1
  30. package/web/_astro/{chunk-QZHKN3VN.OWASJRQy.js → chunk-QZHKN3VN.BtmAEAwZ.js} +1 -1
  31. package/web/_astro/{classDiagram-4FO5ZUOK.BLvrlpNO.js → classDiagram-4FO5ZUOK.CWycT7Ia.js} +1 -1
  32. package/web/_astro/{classDiagram-v2-Q7XG4LA2.BLvrlpNO.js → classDiagram-v2-Q7XG4LA2.CWycT7Ia.js} +1 -1
  33. package/web/_astro/{cose-bilkent-S5V4N54A.XBF-rmyD.js → cose-bilkent-S5V4N54A.DyxjSWon.js} +1 -1
  34. package/web/_astro/{cynefin-OW5HDTMX.DlCx762Z.js → cynefin-OW5HDTMX.DnNfLBpX.js} +1 -1
  35. package/web/_astro/{dagre-BM42HDAG.D17Rshxv.js → dagre-BM42HDAG.BFAvnjKD.js} +1 -1
  36. package/web/_astro/{diagram-2AECGRRQ.AhBIVJC8.js → diagram-2AECGRRQ.BAoNCFeB.js} +1 -1
  37. package/web/_astro/{diagram-5GNKFQAL.C9ximjyC.js → diagram-5GNKFQAL.DiiKe7BP.js} +1 -1
  38. package/web/_astro/{diagram-KO2AKTUF.CZb7Ru_9.js → diagram-KO2AKTUF.t_uOQWVP.js} +1 -1
  39. package/web/_astro/{diagram-LMA3HP47.BW7LwqoS.js → diagram-LMA3HP47.SB9997cP.js} +1 -1
  40. package/web/_astro/{diagram-OG6HWLK6.XC025W0V.js → diagram-OG6HWLK6.Bgsxgf5G.js} +1 -1
  41. package/web/_astro/{erDiagram-TEJ5UH35.CpMXmBDP.js → erDiagram-TEJ5UH35.DO9OAOAc.js} +1 -1
  42. package/web/_astro/{flowDiagram-I6XJVG4X.D2ednJWg.js → flowDiagram-I6XJVG4X.CIb41ZyL.js} +1 -1
  43. package/web/_astro/{ganttDiagram-6RSMTGT7.BjL9FGKO.js → ganttDiagram-6RSMTGT7.Dc8Lk0fV.js} +1 -1
  44. package/web/_astro/{gitGraphDiagram-PVQCEYII.B90g1VGk.js → gitGraphDiagram-PVQCEYII.CpPmMWrl.js} +1 -1
  45. package/web/_astro/index.C-t8kB0T.css +1 -0
  46. package/web/_astro/{infoDiagram-5YYISTIA.RqgycKtQ.js → infoDiagram-5YYISTIA.DxBXfQSf.js} +1 -1
  47. package/web/_astro/{ishikawaDiagram-YF4QCWOH.Ctn-zt6a.js → ishikawaDiagram-YF4QCWOH.gA6OKtXq.js} +1 -1
  48. package/web/_astro/{journeyDiagram-JHISSGLW.DJhT8Ctp.js → journeyDiagram-JHISSGLW.CMEyLMmm.js} +1 -1
  49. package/web/_astro/{kanban-definition-UN3LZRKU.BY1QdejI.js → kanban-definition-UN3LZRKU.CKXlUNCr.js} +1 -1
  50. package/web/_astro/{linear.Di7YObSt.js → linear.CpTGdVx3.js} +1 -1
  51. package/web/_astro/{mermaid.core.CbxtJS3Q.js → mermaid.core.c8SSCsE5.js} +4 -4
  52. package/web/_astro/{mindmap-definition-RKZ34NQL.CxvR4g_J.js → mindmap-definition-RKZ34NQL.BH59HDw6.js} +1 -1
  53. package/web/_astro/{pieDiagram-4H26LBE5.jNWqnBHH.js → pieDiagram-4H26LBE5.DmgFXkZR.js} +1 -1
  54. package/web/_astro/{quadrantDiagram-W4KKPZXB.BCp12MbA.js → quadrantDiagram-W4KKPZXB.DyfB7N4J.js} +1 -1
  55. package/web/_astro/{requirementDiagram-4Y6WPE33.Dxhm4TyR.js → requirementDiagram-4Y6WPE33.DosszoFo.js} +1 -1
  56. package/web/_astro/{sankeyDiagram-5OEKKPKP.BtQXp4J9.js → sankeyDiagram-5OEKKPKP.pBkHhvVN.js} +1 -1
  57. package/web/_astro/{sequenceDiagram-3UESZ5HK.BWEM1R_Q.js → sequenceDiagram-3UESZ5HK.BE8tmkNR.js} +1 -1
  58. package/web/_astro/{stateDiagram-AJRCARHV.BG3wUkWB.js → stateDiagram-AJRCARHV.C4n7vDiG.js} +1 -1
  59. package/web/_astro/{stateDiagram-v2-BHNVJYJU.BLtMeFVP.js → stateDiagram-v2-BHNVJYJU.BpgwYlRy.js} +1 -1
  60. package/web/_astro/{timeline-definition-PNZ67QCA.D5fHo0az.js → timeline-definition-PNZ67QCA.DvcVKc-W.js} +1 -1
  61. package/web/_astro/{vennDiagram-CIIHVFJN.0DcuMluU.js → vennDiagram-CIIHVFJN.BLOk-UOl.js} +1 -1
  62. package/web/_astro/{wardleyDiagram-YWT4CUSO.BZ-dxgHm.js → wardleyDiagram-YWT4CUSO.CIzB7M2o.js} +1 -1
  63. package/web/_astro/{xychartDiagram-2RQKCTM6.Bg-XWF7z.js → xychartDiagram-2RQKCTM6.CGC_lZ19.js} +1 -1
  64. package/web/index.html +2 -2
  65. package/web/_astro/BoardApp.CTkqrhWd.js +0 -1
  66. package/web/_astro/channel.Dsvulp7W.js +0 -1
  67. package/web/_astro/index.BVXdIsZV.css +0 -1
@@ -218,6 +218,25 @@ states:
218
218
  echo "FAIL" > "$SIZE_FILE";
219
219
  fi &&
220
220
  exit 0
221
+ # 0726 R2: task evidence precheck — deterministic live-data
222
+ # evidence-channel proof before implement dispatch. Writes PASS/FAIL to
223
+ # .spur/run/<wbs>-precheck-evidence.status. Always exit 0 (soft action);
224
+ # the precheck→implement guard reads the file, so a missing checker
225
+ # fails closed (writes FAIL, never PASS).
226
+ - kind: shell
227
+ options:
228
+ command: >-
229
+ EVID_FILE=".spur/run/$wbs-precheck-evidence.status" &&
230
+ mkdir -p .spur/run &&
231
+ if [ -f plugins/sp/scripts/task-evidence-precheck.ts ]; then
232
+ bun plugins/sp/scripts/task-evidence-precheck.ts "$wbs"
233
+ --spur-bin "$spurBin";
234
+ else
235
+ echo "task-evidence-precheck failed closed —" >&2 &&
236
+ echo "plugins/sp/scripts/task-evidence-precheck.ts absent." >&2 &&
237
+ echo "FAIL" > "$EVID_FILE";
238
+ fi &&
239
+ exit 0
221
240
 
222
241
  - id: implement
223
242
  description: >
@@ -536,7 +555,17 @@ states:
536
555
  priority: ${vars.taskPriority}
537
556
  compareExecutorWith: implement
538
557
  timeoutMs: ${vars.stepTimeoutMs}
539
- answerFile: .spur/run/${vars.wbs}-verify-answer.txt
558
+ expectFile: .spur/run/${vars.wbs}-verify-answer.txt
559
+ # 0726 R3: hard lint gate over the verifier-owned answer — shape and
560
+ # evidence-row identity, before the verdict derivation reads it.
561
+ # Hard action: a malformed answer halts the sequence here instead of
562
+ # poisoning the verdict parse downstream.
563
+ - kind: shell
564
+ options:
565
+ command: >-
566
+ bun plugins/sp/scripts/verify-answer-lint.ts "$wbs"
567
+ --answer ".spur/run/$wbs-verify-answer.txt"
568
+ --spur-bin "$spurBin"
540
569
  - kind: shell
541
570
  options:
542
571
  command: "$spurBin task verdict $wbs --from-answer .spur/run/$wbs-verify-answer.txt"
@@ -679,11 +708,11 @@ transitions:
679
708
  # ── precheck: size PASS + task check → implement; else → failed ──
680
709
  - from: precheck
681
710
  to: implement
682
- description: Deterministic size and task checks are green — begin implementation.
711
+ description: Deterministic size, evidence, and task checks are green — begin implementation.
683
712
  guard:
684
713
  kind: shell
685
714
  options:
686
- command: 'test "$(cat .spur/run/$wbs-precheck-size.status 2>/dev/null)" = PASS && $spurBin task check $wbs'
715
+ command: 'test "$(cat .spur/run/$wbs-precheck-size.status 2>/dev/null)" = PASS && test "$(cat .spur/run/$wbs-precheck-evidence.status 2>/dev/null)" = PASS && $spurBin task check $wbs'
687
716
  - from: precheck
688
717
  to: failed
689
718
  description: Size and/or task check failed — stop before implement.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@gobing-ai/spur",
3
- "version": "0.3.69",
3
+ "version": "0.3.70",
4
4
  "description": "Spur CLI — local-first harness for mainstream coding agents: constraint checking, workflow orchestration, agent health, and history analytics. Bun-native; exposes the `spur` command.",
5
5
  "keywords": [
6
6
  "spur",
@@ -53,14 +53,14 @@
53
53
  },
54
54
  "devDependencies": {
55
55
  "@commander-js/extra-typings": "^14.0.0",
56
- "@gobing-ai/ts-db": "^0.4.48",
57
- "@gobing-ai/ts-ai-runner": "^0.4.48",
58
- "@gobing-ai/ts-dual-workflow-engine": "^0.4.48",
59
- "@gobing-ai/ts-infra": "^0.4.48",
60
- "@gobing-ai/ts-llm-jsonl-importer": "^0.4.48",
61
- "@gobing-ai/ts-rule-engine": "^0.4.48",
62
- "@gobing-ai/ts-runtime": "^0.4.48",
63
- "@gobing-ai/ts-utils": "^0.4.48",
56
+ "@gobing-ai/ts-db": "^0.4.49",
57
+ "@gobing-ai/ts-ai-runner": "^0.4.49",
58
+ "@gobing-ai/ts-dual-workflow-engine": "^0.4.49",
59
+ "@gobing-ai/ts-infra": "^0.4.49",
60
+ "@gobing-ai/ts-llm-jsonl-importer": "^0.4.49",
61
+ "@gobing-ai/ts-rule-engine": "^0.4.49",
62
+ "@gobing-ai/ts-runtime": "^0.4.49",
63
+ "@gobing-ai/ts-utils": "^0.4.49",
64
64
  "@types/bun": "1.3.14",
65
65
  "@types/figlet": "^1.7.0",
66
66
  "@types/node-notifier": "8.0.5",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "sp",
3
- "version": "0.3.69",
3
+ "version": "0.3.70",
4
4
  "description": "Spur — a local-first harness engineering toolkit that wraps mainstream coding agents with constraint checking, workflow orchestration, and history analytics.",
5
5
  "extensions": {
6
6
  "pi": ["./hooks/pi/guard-extension.ts"]
@@ -0,0 +1,181 @@
1
+ #!/usr/bin/env bun
2
+ /**
3
+ * task-evidence-precheck — deterministic evidence-channel precheck (R2, task 0726).
4
+ *
5
+ * Parses the task content for an exact `evidence-channel:` declaration and proves the
6
+ * declared live-data channel exists in the local spur database before implementation
7
+ * begins. Currently exactly one channel is allowlisted:
8
+ *
9
+ * evidence-channel: history_tool_call.args_raw[pi]
10
+ *
11
+ * …satisfied only when the fixed query
12
+ *
13
+ * SELECT COUNT(*) FROM history_tool_call WHERE args_raw IS NOT NULL AND source = 'pi'
14
+ *
15
+ * returns a positive count on `<cwd>/.spur/spur.db` — i.e. a live non-dry-run pi import
16
+ * has already preserved tool-call `args_raw` (0722 R1). Unknown declarations, a missing
17
+ * database, a missing table, and a zero count all fail closed.
18
+ *
19
+ * A task without any `evidence-channel:` declaration passes without opening SQLite —
20
+ * the check only gates tasks that declare a live-data evidence channel.
21
+ *
22
+ * Always exits 0 (soft action). Both precheck→implement guards in task-pipeline.yaml
23
+ * read the status file; a missing or failing checker writes FAIL, so readiness fails
24
+ * closed.
25
+ *
26
+ * Ships with the plugin to arbitrary projects; node-builtin + bun:sqlite only —
27
+ * no workspace imports.
28
+ *
29
+ * Usage:
30
+ * bun plugins/sp/scripts/task-evidence-precheck.ts <wbs> [--spur-bin <path>]
31
+ *
32
+ * Env: SPUR_BIN
33
+ */
34
+
35
+ import { Database } from 'bun:sqlite';
36
+ import { execFileSync } from 'node:child_process';
37
+ import { existsSync, mkdirSync, writeFileSync } from 'node:fs';
38
+ import { join } from 'node:path';
39
+ import { fileURLToPath } from 'node:url';
40
+
41
+ /** Exact task-content declaration that activates the live-evidence gate (0726 R2). */
42
+ const DECLARATION_PREFIX = 'evidence-channel:';
43
+
44
+ /** The only allowlisted live-data channel (0726 R2). */
45
+ const EVIDENCE_CHANNEL = 'history_tool_call.args_raw[pi]';
46
+
47
+ /** Declaration text as it must appear in the task body. */
48
+ const DECLARATION = `${DECLARATION_PREFIX} ${EVIDENCE_CHANNEL}`;
49
+
50
+ /** The only live-data query this precheck is allowed to run — fixed, never task-authored. */
51
+ const EVIDENCE_QUERY = "SELECT COUNT(*) AS n FROM history_tool_call WHERE args_raw IS NOT NULL AND source = 'pi'";
52
+
53
+ // ─── CLI (same spur-bin chain as task-size-precheck.ts) ─────────────────────
54
+
55
+ function usage(): never {
56
+ console.error('Usage: bun plugins/sp/scripts/task-evidence-precheck.ts <wbs> [--spur-bin <path>]');
57
+ process.exit(1);
58
+ }
59
+
60
+ function defaultSpurBin(): string {
61
+ if (process.env.SPUR_BIN) return process.env.SPUR_BIN;
62
+ const local = fileURLToPath(new URL('../../../apps/cli/src/index.ts', import.meta.url));
63
+ if (existsSync(local)) return `bun ${local}`;
64
+ return 'spur';
65
+ }
66
+
67
+ function parseArgs(argv: string[]): { wbs: string; spurBin: string } {
68
+ let spurBin = defaultSpurBin();
69
+ let wbs = '';
70
+ let i = 0;
71
+ while (i < argv.length) {
72
+ const arg = argv[i];
73
+ if (arg === '--spur-bin') {
74
+ spurBin = argv[i + 1] ?? defaultSpurBin();
75
+ i += 2;
76
+ } else if (!arg.startsWith('--')) {
77
+ wbs = arg;
78
+ i++;
79
+ } else {
80
+ i++;
81
+ }
82
+ }
83
+ if (!wbs) usage();
84
+ return { wbs, spurBin };
85
+ }
86
+
87
+ /**
88
+ * Split a multi-token `spurBin` (`<runtime> <mainModule>`) the same way
89
+ * `runSpurJson` does in feature-sync-bounded.ts — execFileSync's first arg is
90
+ * one executable path, not a shell command line.
91
+ */
92
+ function runSpur(spurBin: string, args: string[]): string {
93
+ const [file = 'spur', ...lead] = spurBin.split(/\s+/).filter(Boolean);
94
+ return execFileSync(file, [...lead, ...args], {
95
+ encoding: 'utf-8',
96
+ stdio: ['pipe', 'pipe', 'pipe'],
97
+ });
98
+ }
99
+
100
+ function writeStatus(wbs: string, status: 'PASS' | 'FAIL'): void {
101
+ const statusDir = join(process.cwd(), '.spur', 'run');
102
+ if (!existsSync(statusDir)) mkdirSync(statusDir, { recursive: true });
103
+ writeFileSync(join(statusDir, `${wbs}-precheck-evidence.status`), `${status}\n`);
104
+ }
105
+
106
+ function fail(wbs: string, reasons: string[]): void {
107
+ writeStatus(wbs, 'FAIL');
108
+ console.error(`task-evidence-precheck: FAIL`);
109
+ for (const r of reasons) {
110
+ console.error(` ${r}`);
111
+ }
112
+ process.exit(0);
113
+ }
114
+
115
+ function main(): void {
116
+ const { wbs, spurBin } = parseArgs(process.argv.slice(2));
117
+
118
+ let taskContent: string;
119
+ try {
120
+ const result = runSpur(spurBin, ['task', 'show', wbs, '--json']);
121
+ const task = JSON.parse(result);
122
+ taskContent = task.content ?? task.body ?? '';
123
+ } catch {
124
+ fail(wbs, [`could not fetch task ${wbs} via ${spurBin} — evidence channel unverifiable`]);
125
+ }
126
+
127
+ // Collect every declaration token. A repeated exact declaration still gates the
128
+ // single fixed query; any non-allowlisted token is an unknown declaration.
129
+ const declarations: string[] = [];
130
+ for (const match of taskContent.matchAll(/evidence-channel:\s*(\S+)/g)) {
131
+ declarations.push(match[1] ?? '');
132
+ }
133
+ const unknown = declarations.filter((d) => d !== EVIDENCE_CHANNEL);
134
+ if (unknown.length > 0) {
135
+ fail(wbs, [
136
+ `unknown evidence-channel declaration(s): ${unknown.join(', ')}`,
137
+ `allowlisted declaration: ${DECLARATION}`,
138
+ ]);
139
+ }
140
+ if (declarations.length === 0) {
141
+ writeStatus(wbs, 'PASS');
142
+ console.error(`task-evidence-precheck: PASS — no evidence-channel declaration; live-data gate not active`);
143
+ process.exit(0);
144
+ }
145
+
146
+ const dbPath = join(process.cwd(), '.spur', 'spur.db');
147
+ if (!existsSync(dbPath)) {
148
+ fail(wbs, [`spur database not found at ${dbPath} — run a real history import first`]);
149
+ }
150
+
151
+ let count: number;
152
+ try {
153
+ const db = new Database(dbPath, { readonly: true });
154
+ try {
155
+ const row = db.query(EVIDENCE_QUERY).get() as { n: number } | undefined;
156
+ count = row?.n ?? 0;
157
+ } finally {
158
+ db.close();
159
+ }
160
+ } catch (e) {
161
+ fail(wbs, [
162
+ `evidence query failed on ${dbPath}: ${e instanceof Error ? e.message : String(e)}`,
163
+ 'history_tool_call table missing or unreadable — run a real history import first',
164
+ ]);
165
+ }
166
+
167
+ if (!(count > 0)) {
168
+ fail(wbs, [
169
+ `0 live pi rows with args_raw (query: ${EVIDENCE_QUERY})`,
170
+ 'run a non-dry-run pi history import with a safe importer before implementing',
171
+ ]);
172
+ }
173
+
174
+ writeStatus(wbs, 'PASS');
175
+ console.error(
176
+ `task-evidence-precheck: PASS — ${count} live pi history_tool_call row(s) with args_raw (declaration: ${DECLARATION})`,
177
+ );
178
+ process.exit(0);
179
+ }
180
+
181
+ main();
@@ -0,0 +1,386 @@
1
+ #!/usr/bin/env bun
2
+ /**
3
+ * verify-answer-lint — deterministic pre-verdict answer lint (0726 R3).
4
+ *
5
+ * Runs AFTER the verify agent exits and BEFORE `spur task verdict --from-answer`.
6
+ * The verifier owns the answer file (`.spur/run/<wbs>-verify-answer.txt`): it creates
7
+ * it with `Verdict: PARTIAL`, appends one complete requirement/AC row at a time, and
8
+ * only replaces the first verdict line once every row is certified. Because the file
9
+ * is now append-progress instead of a single captured blob, malformed rows can reach
10
+ * the verdict step — this lint rejects each invalid class with a row-level message:
11
+ *
12
+ * - missing, duplicate, or unknown requirement IDs (vs the task's Requirements)
13
+ * - AC IDs that do not exactly match the task's AC checklist label or a linked
14
+ * feature scenario title
15
+ * - status / evidence-type values the verdict parser would drop
16
+ * - empty evidence
17
+ *
18
+ * Compound evidence types (`test + command`) stay valid — normalization mirrors
19
+ * `packages/app/src/services/task-verdict.ts` exactly, so anything this lint accepts
20
+ * is also accepted by `spur task verdict --from-answer` (and vice versa).
21
+ *
22
+ * Exits non-zero on any finding, with bounded diagnostics (first 10). Writes nothing.
23
+ *
24
+ * Ships with the plugin to arbitrary projects; node-builtin only — no workspace imports.
25
+ *
26
+ * Usage:
27
+ * bun plugins/sp/scripts/verify-answer-lint.ts <wbs> --answer <path> [--spur-bin <path>]
28
+ *
29
+ * Env: SPUR_BIN
30
+ */
31
+
32
+ import { execFileSync } from 'node:child_process';
33
+ import { existsSync, readFileSync } from 'node:fs';
34
+ import { fileURLToPath } from 'node:url';
35
+
36
+ // ─── CLI (same spur-bin chain as task-evidence-precheck.ts) ─────────────────
37
+
38
+ function usage(): never {
39
+ console.error('Usage: bun plugins/sp/scripts/verify-answer-lint.ts <wbs> --answer <path> [--spur-bin <path>]');
40
+ process.exit(1);
41
+ }
42
+
43
+ function defaultSpurBin(): string {
44
+ if (process.env.SPUR_BIN) return process.env.SPUR_BIN;
45
+ const local = fileURLToPath(new URL('../../../apps/cli/src/index.ts', import.meta.url));
46
+ if (existsSync(local)) return `bun ${local}`;
47
+ return 'spur';
48
+ }
49
+
50
+ function parseArgs(argv: string[]): { wbs: string; answer: string; spurBin: string } {
51
+ let spurBin = defaultSpurBin();
52
+ let wbs = '';
53
+ let answer = '';
54
+ let i = 0;
55
+ while (i < argv.length) {
56
+ const arg = argv[i];
57
+ if (arg === '--spur-bin') {
58
+ spurBin = argv[i + 1] ?? defaultSpurBin();
59
+ i += 2;
60
+ } else if (arg === '--answer') {
61
+ answer = argv[i + 1] ?? '';
62
+ i += 2;
63
+ } else if (!arg.startsWith('--')) {
64
+ wbs = arg;
65
+ i++;
66
+ } else {
67
+ i++;
68
+ }
69
+ }
70
+ if (!wbs || !answer) usage();
71
+ return { wbs, answer, spurBin };
72
+ }
73
+
74
+ function runSpur(spurBin: string, args: string[]): string {
75
+ const [file = 'spur', ...lead] = spurBin.split(/\s+/).filter(Boolean);
76
+ return execFileSync(file, [...lead, ...args], {
77
+ encoding: 'utf-8',
78
+ stdio: ['pipe', 'pipe', 'pipe'],
79
+ });
80
+ }
81
+
82
+ // ─── Answer parsing — mirrors packages/app/src/services/task-verdict.ts ─────
83
+
84
+ interface ReqRow {
85
+ id: string;
86
+ status: string;
87
+ evidence: string;
88
+ line: number;
89
+ }
90
+ interface AcRow {
91
+ id: string;
92
+ status: string;
93
+ evidenceType: string;
94
+ evidence: string;
95
+ line: number;
96
+ }
97
+
98
+ function splitTableCells(line: string): string[] {
99
+ return line
100
+ .split(/(?<!\\)\|/)
101
+ .map((c) => c.replace(/\\\|/g, '|').trim())
102
+ .filter(Boolean);
103
+ }
104
+
105
+ function normalizeReqStatus(raw: string): string | null {
106
+ if (/\bMET\b/.test(raw)) return 'MET';
107
+ if (/\bPARTIAL\b/.test(raw)) return 'PARTIAL';
108
+ if (/\bUNMET\b/.test(raw)) return 'UNMET';
109
+ return null;
110
+ }
111
+
112
+ function normalizeAcStatus(raw: string): string | null {
113
+ if (/\bMET\b/.test(raw)) return 'MET';
114
+ if (/\bPARTIAL\b/.test(raw)) return 'PARTIAL';
115
+ if (/\bUNMET\b/.test(raw)) return 'UNMET';
116
+ if (/\bN\/A\b/.test(raw) || /\bNA\b/.test(raw)) return 'N/A';
117
+ return null;
118
+ }
119
+
120
+ function normalizeEvidenceTypeToken(normalized: string): string | null {
121
+ if (normalized === 'test') return 'test';
122
+ if (normalized === 'command') return 'command';
123
+ if (
124
+ normalized === 'static-ref' ||
125
+ normalized === 'static' ||
126
+ normalized === 'doc' ||
127
+ normalized === 'docs' ||
128
+ normalized === 'documentation'
129
+ ) {
130
+ return 'static-ref';
131
+ }
132
+ if (normalized === 'manual-review' || normalized === 'manual') return 'manual-review';
133
+ if (normalized === 'llm-judge' || normalized === 'judge') return 'llm-judge';
134
+ if (normalized === 'n/a' || normalized === 'na') return 'n/a';
135
+ return null;
136
+ }
137
+
138
+ const EVIDENCE_TYPE_PRECEDENCE = ['test', 'command', 'static-ref', 'manual-review', 'llm-judge', 'n/a'] as const;
139
+
140
+ function normalizeEvidenceType(raw: string): string | null {
141
+ const normalized = raw.toLowerCase().trim();
142
+ const single = normalizeEvidenceTypeToken(normalized);
143
+ if (single !== null) return single;
144
+ const parts = normalized.split(/[+,/]/).filter((p) => p.trim());
145
+ const tokens = parts.map((part) => normalizeEvidenceTypeToken(part.trim())).filter((t) => t !== null);
146
+ if (tokens.length < 2 || tokens.length !== parts.length) return null;
147
+ return EVIDENCE_TYPE_PRECEDENCE.find((candidate) => tokens.includes(candidate)) ?? null;
148
+ }
149
+
150
+ interface AnswerTables {
151
+ verdict: { value: string; line: number } | null;
152
+ reqs: ReqRow[];
153
+ acs: AcRow[];
154
+ }
155
+
156
+ function parseAnswer(text: string): AnswerTables {
157
+ const out: AnswerTables = { verdict: null, reqs: [], acs: [] };
158
+ const lines = text.split('\n');
159
+ let reqTable = false;
160
+ let acTable = false;
161
+
162
+ for (let i = 0; i < lines.length; i++) {
163
+ const trimmed = lines[i]?.trim() ?? '';
164
+ const lineNo = i + 1;
165
+
166
+ // A markdown heading closes whichever table is open (mirrors the verdict parser).
167
+ if ((reqTable || acTable) && /^#{1,6}\s/.test(trimmed)) {
168
+ reqTable = false;
169
+ acTable = false;
170
+ continue;
171
+ }
172
+ if (!trimmed.startsWith('|')) continue;
173
+ const cells = splitTableCells(trimmed);
174
+ if (/^[-:]+$/.test(cells[0] ?? '')) continue;
175
+
176
+ const h0 = (cells[0] ?? '').toLowerCase();
177
+ const h1 = (cells[1] ?? '').toLowerCase();
178
+
179
+ // Requirement header: `| Req | Status | Evidence |` (id-like first cell + status column).
180
+ if (!reqTable && !acTable && cells.length >= 2) {
181
+ const idLike = h0.includes('req') || h0 === 'requirement' || h0 === 'r#' || h0 === 'r' || /^r\d+$/.test(h0);
182
+ if (idLike && (h1.includes('status') || h1 === 'verdict')) {
183
+ reqTable = true;
184
+ continue;
185
+ }
186
+ if (
187
+ (h0 === 'ac' || h0.includes('acceptance')) &&
188
+ h1.includes('status') &&
189
+ (cells[2] ?? '').toLowerCase().includes('evidence')
190
+ ) {
191
+ acTable = true;
192
+ continue;
193
+ }
194
+ }
195
+
196
+ if (reqTable && cells.length >= 2) {
197
+ // An AC header following the requirement table closes it (mirrors the parser).
198
+ if ((h0 === 'ac' || h0.includes('acceptance')) && h1.includes('status')) {
199
+ reqTable = false;
200
+ acTable = true;
201
+ continue;
202
+ }
203
+ out.reqs.push({
204
+ id: cells[0] ?? '',
205
+ status: (cells[1] ?? '').toUpperCase(),
206
+ evidence: cells[2] ?? '',
207
+ line: lineNo,
208
+ });
209
+ continue;
210
+ }
211
+ if (acTable && cells.length >= 3) {
212
+ out.acs.push({
213
+ id: cells[0] ?? '',
214
+ status: cells[1] ?? '',
215
+ evidenceType: cells[2] ?? '',
216
+ evidence: cells[3] ?? '',
217
+ line: lineNo,
218
+ });
219
+ }
220
+ }
221
+
222
+ const verdictMatches = [...text.matchAll(/^\s*Verdict:\s*(\S+)\s*$/gim)];
223
+ if (verdictMatches.length === 1) {
224
+ out.verdict = {
225
+ value: verdictMatches[0]?.[1] ?? '',
226
+ line: text.slice(0, verdictMatches[0]?.index ?? 0).split('\n').length,
227
+ };
228
+ }
229
+ return out;
230
+ }
231
+
232
+ // ─── Task-side identity extraction ───────────────────────────────────────────
233
+
234
+ function sectionBetween(text: string, heading: string): string {
235
+ const marker = new RegExp(`^#{1,6}\\s+${heading}\\s*$`, 'im');
236
+ const match = marker.exec(text);
237
+ if (!match || match.index === undefined) return '';
238
+ const rest = text.slice(match.index);
239
+ const lineEnd = rest.indexOf('\n');
240
+ const afterHeading = lineEnd === -1 ? '' : rest.slice(lineEnd + 1);
241
+ const next = afterHeading.search(/^#{1,6}\s/m);
242
+ return next === -1 ? afterHeading : afterHeading.slice(0, next);
243
+ }
244
+
245
+ function extractRequirementIds(taskContent: string): string[] {
246
+ const section = sectionBetween(taskContent, 'Requirements');
247
+ const ids = new Set<string>();
248
+ for (const m of section.matchAll(/\*\*(R\d+(?:\.\d+)*)\s*\./g)) ids.add(m[1] ?? '');
249
+ for (const m of section.matchAll(/\*\*(R\d+(?:\.\d+)*)\*\*/g)) ids.add(m[1] ?? '');
250
+ return [...ids];
251
+ }
252
+
253
+ function extractAcIdentities(taskContent: string, featureContent: string | null): string[] {
254
+ const identities = new Set<string>();
255
+ const section = sectionBetween(taskContent, 'Acceptance Criteria');
256
+ for (const m of section.matchAll(/^[-*]\s+\[[ x]\]\s+(.+?)\s*(?::|$)/gm)) {
257
+ const label = (m[1] ?? '').trim();
258
+ if (!label) continue;
259
+ identities.add(label);
260
+ const leading = label.split(/\s+/)[0] ?? '';
261
+ if (leading && leading !== label) identities.add(leading);
262
+ }
263
+ if (featureContent !== null) {
264
+ for (const m of featureContent.matchAll(/^[ \t]*Scenario:\s*(.+)\s*$/gm)) {
265
+ const title = (m[1] ?? '').trim();
266
+ if (title) identities.add(title);
267
+ }
268
+ }
269
+ return [...identities];
270
+ }
271
+
272
+ // ─── Main ────────────────────────────────────────────────────────────────────
273
+
274
+ function main(): void {
275
+ const { wbs, answer, spurBin } = parseArgs(process.argv.slice(2));
276
+ const findings: string[] = [];
277
+ const add = (msg: string): void => {
278
+ if (findings.length < 10) findings.push(msg);
279
+ };
280
+
281
+ if (!existsSync(answer)) {
282
+ console.error(`verify-answer-lint: FAIL — answer file not found: ${answer}`);
283
+ process.exit(1);
284
+ }
285
+ const raw = readFileSync(answer, 'utf8');
286
+ if (!raw.trim()) {
287
+ console.error(`verify-answer-lint: FAIL — answer file is empty: ${answer}`);
288
+ process.exit(1);
289
+ }
290
+
291
+ const tables = parseAnswer(raw);
292
+ if (tables.verdict === null) {
293
+ add('no `Verdict:` line (expected exactly one `Verdict: PASS|PARTIAL|FAIL` line)');
294
+ } else if (!/^(PASS|PARTIAL|FAIL)$/i.test(tables.verdict.value)) {
295
+ add(`line ${tables.verdict.line}: invalid Verdict value "${tables.verdict.value}" (PASS | PARTIAL | FAIL)`);
296
+ }
297
+
298
+ let taskContent = '';
299
+ let featureId = '';
300
+ try {
301
+ const task = JSON.parse(runSpur(spurBin, ['task', 'show', wbs, '--json'])) as {
302
+ content?: string;
303
+ body?: string;
304
+ feature_id?: string;
305
+ frontmatter?: { feature_id?: string };
306
+ };
307
+ taskContent = task.content ?? task.body ?? '';
308
+ featureId = task.feature_id ?? task.frontmatter?.feature_id ?? '';
309
+ } catch {
310
+ console.error(`verify-answer-lint: FAIL — could not fetch task ${wbs} via ${spurBin}`);
311
+ process.exit(1);
312
+ }
313
+ if (!taskContent) {
314
+ console.error(`verify-answer-lint: FAIL — task ${wbs} returned no content via ${spurBin}`);
315
+ process.exit(1);
316
+ }
317
+
318
+ let featureContent: string | null = null;
319
+ if (featureId) {
320
+ try {
321
+ const feature = JSON.parse(runSpur(spurBin, ['feature', 'show', featureId, '--json'])) as {
322
+ content?: string;
323
+ };
324
+ featureContent = feature.content ?? '';
325
+ } catch {
326
+ featureContent = null; // checklist labels still apply; scenario titles unavailable
327
+ }
328
+ }
329
+
330
+ const reqIds = extractRequirementIds(taskContent);
331
+ const acIdentities = extractAcIdentities(taskContent, featureContent);
332
+
333
+ // Requirement rows: completeness, no unknowns, no duplicates, valid status, non-empty evidence.
334
+ const seenReq = new Set<string>();
335
+ for (const row of tables.reqs) {
336
+ if (!reqIds.includes(row.id))
337
+ add(`line ${row.line}: unknown requirement ID "${row.id}" (task declares: ${reqIds.join(', ') || 'none'})`);
338
+ else if (seenReq.has(row.id)) add(`line ${row.line}: duplicate requirement row "${row.id}"`);
339
+ seenReq.add(row.id);
340
+ if (normalizeReqStatus(row.status) === null)
341
+ add(`line ${row.line}: "${row.id}" invalid status "${row.status}" (MET | PARTIAL | UNMET)`);
342
+ if (!row.evidence.trim()) add(`line ${row.line}: "${row.id}" has empty evidence`);
343
+ }
344
+ for (const id of reqIds) {
345
+ if (!seenReq.has(id)) add(`missing requirement row for "${id}"`);
346
+ }
347
+
348
+ // AC rows: identity must exactly match a checklist label/token or a scenario title;
349
+ // status and evidence type must normalize; evidence non-empty. AC completeness is the
350
+ // verifier's authoring contract, not a lint rejection class (0726 R3).
351
+ const seenAc = new Set<string>();
352
+ for (const row of tables.acs) {
353
+ if (!acIdentities.includes(row.id)) {
354
+ add(
355
+ `line ${row.line}: AC ID "${row.id.slice(0, 60)}" matches no task AC checklist label or scenario title`,
356
+ );
357
+ } else if (seenAc.has(row.id)) {
358
+ add(`line ${row.line}: duplicate AC row "${row.id.slice(0, 60)}"`);
359
+ }
360
+ seenAc.add(row.id);
361
+ if (normalizeAcStatus(row.status) === null)
362
+ add(`line ${row.line}: invalid AC status "${row.status}" (MET | PARTIAL | UNMET | N/A)`);
363
+ if (normalizeEvidenceType(row.evidenceType) === null)
364
+ add(
365
+ `line ${row.line}: invalid evidence type "${row.evidenceType}" (test | command | static-ref | manual-review | llm-judge | n/a, or a + compound)`,
366
+ );
367
+ if (!row.evidence.trim()) add(`line ${row.line}: AC "${row.id.slice(0, 40)}" has empty evidence`);
368
+ }
369
+
370
+ if (findings.length > 0) {
371
+ console.error(
372
+ `verify-answer-lint: FAIL — ${findings.length}${findings.length >= 10 ? '+' : ''} finding(s) in ${answer}`,
373
+ );
374
+ for (const f of findings) console.error(` ${f}`);
375
+ process.exit(1);
376
+ }
377
+
378
+ const reqCount = tables.reqs.length;
379
+ const acCount = tables.acs.length;
380
+ console.error(
381
+ `verify-answer-lint: PASS — ${reqCount} requirement row(s), ${acCount} AC row(s), verdict ${tables.verdict?.value ?? '?'}`,
382
+ );
383
+ process.exit(0);
384
+ }
385
+
386
+ main();