@gobing-ai/spur 0.3.68 → 0.3.70

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/config/corpus-baseline.json +0 -18
  3. package/config/plugin-scripts.json +8 -0
  4. package/config/workflow-composition-baseline.json +123 -377
  5. package/config/workflows/basic.yaml +1 -1
  6. package/config/workflows/docs-pipeline.yaml +1 -1
  7. package/config/workflows/feature-dev.yaml +1 -1
  8. package/config/workflows/idea-pipeline.yaml +1 -1
  9. package/config/workflows/task-pipeline.yaml +62 -20
  10. package/config/workflows/wayfinder-resolution.yaml +1 -1
  11. package/config/workflows/wrapup-pipeline.yaml +1 -1
  12. package/package.json +9 -9
  13. package/plugins/sp/plugin.json +1 -1
  14. package/plugins/sp/scripts/surface-drift-inventory.ts +2 -15
  15. package/plugins/sp/scripts/task-evidence-precheck.ts +181 -0
  16. package/plugins/sp/scripts/task-size-precheck.ts +21 -79
  17. package/plugins/sp/scripts/verify-answer-lint.ts +386 -0
  18. package/plugins/sp/skills/code-verification/SKILL.md +11 -10
  19. package/plugins/sp/skills/code-verification/references/verdict-schema.md +10 -0
  20. package/plugins/sp/skills/spur-cli/references/history.md +1 -0
  21. package/plugins/sp/skills/spur-dev/references/cross-cutting.md +2 -1
  22. package/plugins/sp/skills/spur-dev/references/gate-checklists.md +16 -0
  23. package/spur.js +1275 -292
  24. package/web/_astro/BoardApp.DHj-03Dp.js +1 -0
  25. package/web/_astro/{BoardApp.BQFbkeqq.js → BoardApp.DOadeEJV.js} +77 -77
  26. package/web/_astro/{TaskDetail.Dl2Eaj1w.js → TaskDetail.DKzpkDj5.js} +1 -1
  27. package/web/_astro/{arc.uG14rp8A.js → arc.Bsa0gprH.js} +1 -1
  28. package/web/_astro/{architectureDiagram-3BPJPVTR.Dye6uD_x.js → architectureDiagram-3BPJPVTR.LOUZCDBC.js} +1 -1
  29. package/web/_astro/{blockDiagram-GPEHLZMM.B9Pkh7Hb.js → blockDiagram-GPEHLZMM.CGee4isG.js} +1 -1
  30. package/web/_astro/{c4Diagram-AAUBKEIU.C2x7SC_X.js → c4Diagram-AAUBKEIU.B2CrQJ_O.js} +1 -1
  31. package/web/_astro/channel.BGL7KSHC.js +1 -0
  32. package/web/_astro/{chunk-2J33WTMH.D2p4-nWk.js → chunk-2J33WTMH.DJvf8RiP.js} +1 -1
  33. package/web/_astro/{chunk-4BX2VUAB.S-6jf33o.js → chunk-4BX2VUAB.B_l35O-Y.js} +1 -1
  34. package/web/_astro/{chunk-55IACEB6.DVXj4Fdh.js → chunk-55IACEB6.DqbxMswU.js} +1 -1
  35. package/web/_astro/{chunk-727SXJPM.Dz689FMN.js → chunk-727SXJPM.D_7ZGNjE.js} +1 -1
  36. package/web/_astro/{chunk-AQP2D5EJ.KxYj5TnI.js → chunk-AQP2D5EJ.Cq6yZ0s8.js} +1 -1
  37. package/web/_astro/{chunk-FMBD7UC4.itTQyHQB.js → chunk-FMBD7UC4.BWrlUPYi.js} +1 -1
  38. package/web/_astro/{chunk-ND2GUHAM.euSrbJf5.js → chunk-ND2GUHAM.BPLaXFZH.js} +1 -1
  39. package/web/_astro/{chunk-QZHKN3VN.OWASJRQy.js → chunk-QZHKN3VN.BtmAEAwZ.js} +1 -1
  40. package/web/_astro/{classDiagram-4FO5ZUOK.BLvrlpNO.js → classDiagram-4FO5ZUOK.CWycT7Ia.js} +1 -1
  41. package/web/_astro/{classDiagram-v2-Q7XG4LA2.BLvrlpNO.js → classDiagram-v2-Q7XG4LA2.CWycT7Ia.js} +1 -1
  42. package/web/_astro/{cose-bilkent-S5V4N54A.XBF-rmyD.js → cose-bilkent-S5V4N54A.DyxjSWon.js} +1 -1
  43. package/web/_astro/{cynefin-OW5HDTMX.DlCx762Z.js → cynefin-OW5HDTMX.DnNfLBpX.js} +1 -1
  44. package/web/_astro/{dagre-BM42HDAG.D17Rshxv.js → dagre-BM42HDAG.BFAvnjKD.js} +1 -1
  45. package/web/_astro/{diagram-2AECGRRQ.AhBIVJC8.js → diagram-2AECGRRQ.BAoNCFeB.js} +1 -1
  46. package/web/_astro/{diagram-5GNKFQAL.C9ximjyC.js → diagram-5GNKFQAL.DiiKe7BP.js} +1 -1
  47. package/web/_astro/{diagram-KO2AKTUF.CZb7Ru_9.js → diagram-KO2AKTUF.t_uOQWVP.js} +1 -1
  48. package/web/_astro/{diagram-LMA3HP47.BW7LwqoS.js → diagram-LMA3HP47.SB9997cP.js} +1 -1
  49. package/web/_astro/{diagram-OG6HWLK6.XC025W0V.js → diagram-OG6HWLK6.Bgsxgf5G.js} +1 -1
  50. package/web/_astro/{erDiagram-TEJ5UH35.CpMXmBDP.js → erDiagram-TEJ5UH35.DO9OAOAc.js} +1 -1
  51. package/web/_astro/{flowDiagram-I6XJVG4X.D2ednJWg.js → flowDiagram-I6XJVG4X.CIb41ZyL.js} +1 -1
  52. package/web/_astro/{ganttDiagram-6RSMTGT7.BjL9FGKO.js → ganttDiagram-6RSMTGT7.Dc8Lk0fV.js} +1 -1
  53. package/web/_astro/{gitGraphDiagram-PVQCEYII.B90g1VGk.js → gitGraphDiagram-PVQCEYII.CpPmMWrl.js} +1 -1
  54. package/web/_astro/index.C-t8kB0T.css +1 -0
  55. package/web/_astro/{infoDiagram-5YYISTIA.RqgycKtQ.js → infoDiagram-5YYISTIA.DxBXfQSf.js} +1 -1
  56. package/web/_astro/{ishikawaDiagram-YF4QCWOH.Ctn-zt6a.js → ishikawaDiagram-YF4QCWOH.gA6OKtXq.js} +1 -1
  57. package/web/_astro/{journeyDiagram-JHISSGLW.DJhT8Ctp.js → journeyDiagram-JHISSGLW.CMEyLMmm.js} +1 -1
  58. package/web/_astro/{kanban-definition-UN3LZRKU.BY1QdejI.js → kanban-definition-UN3LZRKU.CKXlUNCr.js} +1 -1
  59. package/web/_astro/{linear.Di7YObSt.js → linear.CpTGdVx3.js} +1 -1
  60. package/web/_astro/{mermaid.core.CbxtJS3Q.js → mermaid.core.c8SSCsE5.js} +4 -4
  61. package/web/_astro/{mindmap-definition-RKZ34NQL.CxvR4g_J.js → mindmap-definition-RKZ34NQL.BH59HDw6.js} +1 -1
  62. package/web/_astro/{pieDiagram-4H26LBE5.jNWqnBHH.js → pieDiagram-4H26LBE5.DmgFXkZR.js} +1 -1
  63. package/web/_astro/{quadrantDiagram-W4KKPZXB.BCp12MbA.js → quadrantDiagram-W4KKPZXB.DyfB7N4J.js} +1 -1
  64. package/web/_astro/{requirementDiagram-4Y6WPE33.Dxhm4TyR.js → requirementDiagram-4Y6WPE33.DosszoFo.js} +1 -1
  65. package/web/_astro/{sankeyDiagram-5OEKKPKP.BtQXp4J9.js → sankeyDiagram-5OEKKPKP.pBkHhvVN.js} +1 -1
  66. package/web/_astro/{sequenceDiagram-3UESZ5HK.BWEM1R_Q.js → sequenceDiagram-3UESZ5HK.BE8tmkNR.js} +1 -1
  67. package/web/_astro/{stateDiagram-AJRCARHV.BG3wUkWB.js → stateDiagram-AJRCARHV.C4n7vDiG.js} +1 -1
  68. package/web/_astro/{stateDiagram-v2-BHNVJYJU.BLtMeFVP.js → stateDiagram-v2-BHNVJYJU.BpgwYlRy.js} +1 -1
  69. package/web/_astro/{timeline-definition-PNZ67QCA.D5fHo0az.js → timeline-definition-PNZ67QCA.DvcVKc-W.js} +1 -1
  70. package/web/_astro/{vennDiagram-CIIHVFJN.0DcuMluU.js → vennDiagram-CIIHVFJN.BLOk-UOl.js} +1 -1
  71. package/web/_astro/{wardleyDiagram-YWT4CUSO.BZ-dxgHm.js → wardleyDiagram-YWT4CUSO.CIzB7M2o.js} +1 -1
  72. package/web/_astro/{xychartDiagram-2RQKCTM6.Bg-XWF7z.js → xychartDiagram-2RQKCTM6.CGC_lZ19.js} +1 -1
  73. package/web/index.html +2 -2
  74. package/web/_astro/BoardApp.CTkqrhWd.js +0 -1
  75. package/web/_astro/channel.Dsvulp7W.js +0 -1
  76. package/web/_astro/index.BVXdIsZV.css +0 -1
@@ -24,7 +24,7 @@ failureStates:
24
24
  vars:
25
25
  # `task` is reserved by the engine runtime — use taskLabel for the free-form label.
26
26
  taskLabel: "task"
27
- agent: "omp"
27
+ agent: "auto"
28
28
  stepTimeoutMs: "1800000"
29
29
  # Override per project: `--vars '{"qualityGateCmd":"bun run lint && bun run test"}'`
30
30
  qualityGateCmd: "bun run check"
@@ -30,7 +30,7 @@ vars:
30
30
  wbs: "0000"
31
31
  profile: "standard"
32
32
  spurBin: "spur"
33
- agent: "omp"
33
+ agent: "auto"
34
34
  stepTimeoutMs: "1800000"
35
35
  # Proof-state bracket (task 0704, mirroring task-pipeline 0612/0703). `proofDigest` is the
36
36
  # canonical capture at verify entry; `proofDigestNow` is the live re-capture compared against
@@ -44,7 +44,7 @@ failureStates:
44
44
  - failed
45
45
  vars:
46
46
  featureId: ""
47
- agent: "omp"
47
+ agent: "auto"
48
48
  profile: "standard"
49
49
  spurBin: "spur"
50
50
  stepTimeoutMs: "1800000"
@@ -65,7 +65,7 @@ vars:
65
65
  spurBin: "spur"
66
66
  # R8 (0366): injected by WorkflowAppService.run(); stamps discovery artifact provenance.
67
67
  __runId: ""
68
- agent: "omp"
68
+ agent: "auto"
69
69
  planningAgent: ""
70
70
  stepTimeoutMs: "1800000"
71
71
 
@@ -54,15 +54,16 @@ vars:
54
54
  # runs and `workflow validate` resolve the template. (See AGENTS.md / ADR-026.)
55
55
  spurBin: "spur"
56
56
  # Agent the pipeline's agent.run steps invoke. Override per run with
57
- # `--vars '{"agent":"claude"}'`. Pinned (not left to the AiRunner's <default>
58
- # selection) so a broken/misconfigured agent on the box can't silently capture the run.
59
- agent: "omp"
60
- # Implement-only executor override (R1, task 0454). Resolves to this YAML literal
61
- # unless overridden. The `--agent` flag (execution-batch.md §3.2) forwards into BOTH
57
+ # `--vars '{"agent":"claude"}'`. `auto` is the reserved config-resolving selector:
58
+ # `agent.default` role -> tier -> cheapest USABLE executor. A named literal here would
59
+ # pin a box-specific binary into tracked SSOT and escape that usability ladder.
60
+ agent: "auto"
61
+ # Implement-only executor override (R1, task 0454). Resolves like `agent` unless
62
+ # overridden. The `--agent` flag (execution-batch.md §3.2) forwards into BOTH
62
63
  # `agent` and `implementAgent`, so a pinned `--agent X` reaches implement too. To
63
64
  # pin ONLY implement while other hops keep the default, pass
64
65
  # `--vars '{"implementAgent":"omp-zai"}'`.
65
- implementAgent: "omp"
66
+ implementAgent: "auto"
66
67
  # Step-level timeout for agentic hops (review / verify / test-fix) in ms.
67
68
  # Soft quality-gate shells are unbounded by this var (host shell only).
68
69
  # Raised 600s → 1800s (task 0398 R4 / H6 dogfood). Override:
@@ -174,6 +175,10 @@ states:
174
175
  # R1 (0453): auto-profile precheck reopens a done feature before task check.
175
176
  # Under profile=auto, resolve feature_id, sync (preferred) or update to active.
176
177
  # Under non-auto, leave R4 message to guide the operator.
178
+ # R3 (0723): a real reactivation failure is surfaced, not swallowed —
179
+ # the default 'fail' onEnter policy halts the sequence and routes the
180
+ # run to `failed` before implementation. Verbs stay single-shot:
181
+ # one sync, then one update fallback, never retried in a loop.
177
182
  - kind: shell
178
183
  options:
179
184
  command: >-
@@ -181,17 +186,23 @@ states:
181
186
  FID=$($spurBin task show $wbs --json 2>/dev/null |
182
187
  jq -r '.feature_id // .frontmatter.feature_id // empty' 2>/dev/null);
183
188
  if [ -n "$FID" ]; then
184
- $spurBin feature sync "$FID" --force --json 2>/dev/null ||
185
- $spurBin feature update "$FID" active 2>/dev/null || true;
189
+ if ! $spurBin feature sync "$FID" --force 2>/dev/null; then
190
+ if ! $spurBin feature update "$FID" active 2>/dev/null; then
191
+ echo "precheck: FAIL - feature reactivation $FID failed;" >&2;
192
+ echo "precheck: feature sync + feature update both errored" >&2;
193
+ exit 1;
194
+ fi;
195
+ fi;
186
196
  fi;
187
197
  fi;
188
198
  exit 0
189
- # R2 (0454): task size precheck — evaluate R-item and Plan-item counts.
190
- # Writes PASS/FAIL to .spur/run/<wbs>-precheck-size.status. Always exit 0
191
- # (soft check, like doctor). The precheck→implement guard reads the file.
192
- # Temporary config-only bypass (0723): keep deterministic size limits but defer
193
- # the executor-tier policy to the dedicated task-pipeline upgrade. This avoids
194
- # the script's second `spur agent doctor` call while the precheck path is repaired.
199
+ # R2 (0454, 0723): task size precheck — deterministic count-only
200
+ # evaluation of R-item and Plan-item counts. No executor-tier policy:
201
+ # dispatch-time requiresCapabilities at `agent.run` is the
202
+ # authoritative capability check. Writes PASS/FAIL to
203
+ # .spur/run/<wbs>-precheck-size.status. Always exit 0 (soft action);
204
+ # the precheck→implement guard reads the file, so a missing checker
205
+ # fails closed (writes FAIL, never PASS).
195
206
  - kind: shell
196
207
  options:
197
208
  command: >-
@@ -199,10 +210,31 @@ states:
199
210
  mkdir -p .spur/run &&
200
211
  if [ -f plugins/sp/scripts/task-size-precheck.ts ]; then
201
212
  bun plugins/sp/scripts/task-size-precheck.ts "$wbs"
202
- --spur-bin "$spurBin" --max-reqs "$maxImplementReqs" --max-plan-items "$maxImplementPlanItems";
213
+ --spur-bin "$spurBin" --max-reqs "$maxImplementReqs"
214
+ --max-plan-items "$maxImplementPlanItems";
203
215
  else
204
- echo "task-size-precheck skippedplugins/sp/scripts/task-size-precheck.ts not present in project." >&2 &&
205
- echo "PASS" > "$SIZE_FILE";
216
+ echo "task-size-precheck failed closed checker script" >&2 &&
217
+ echo "plugins/sp/scripts/task-size-precheck.ts absent." >&2 &&
218
+ echo "FAIL" > "$SIZE_FILE";
219
+ fi &&
220
+ exit 0
221
+ # 0726 R2: task evidence precheck — deterministic live-data
222
+ # evidence-channel proof before implement dispatch. Writes PASS/FAIL to
223
+ # .spur/run/<wbs>-precheck-evidence.status. Always exit 0 (soft action);
224
+ # the precheck→implement guard reads the file, so a missing checker
225
+ # fails closed (writes FAIL, never PASS).
226
+ - kind: shell
227
+ options:
228
+ command: >-
229
+ EVID_FILE=".spur/run/$wbs-precheck-evidence.status" &&
230
+ mkdir -p .spur/run &&
231
+ if [ -f plugins/sp/scripts/task-evidence-precheck.ts ]; then
232
+ bun plugins/sp/scripts/task-evidence-precheck.ts "$wbs"
233
+ --spur-bin "$spurBin";
234
+ else
235
+ echo "task-evidence-precheck failed closed —" >&2 &&
236
+ echo "plugins/sp/scripts/task-evidence-precheck.ts absent." >&2 &&
237
+ echo "FAIL" > "$EVID_FILE";
206
238
  fi &&
207
239
  exit 0
208
240
 
@@ -523,7 +555,17 @@ states:
523
555
  priority: ${vars.taskPriority}
524
556
  compareExecutorWith: implement
525
557
  timeoutMs: ${vars.stepTimeoutMs}
526
- answerFile: .spur/run/${vars.wbs}-verify-answer.txt
558
+ expectFile: .spur/run/${vars.wbs}-verify-answer.txt
559
+ # 0726 R3: hard lint gate over the verifier-owned answer — shape and
560
+ # evidence-row identity, before the verdict derivation reads it.
561
+ # Hard action: a malformed answer halts the sequence here instead of
562
+ # poisoning the verdict parse downstream.
563
+ - kind: shell
564
+ options:
565
+ command: >-
566
+ bun plugins/sp/scripts/verify-answer-lint.ts "$wbs"
567
+ --answer ".spur/run/$wbs-verify-answer.txt"
568
+ --spur-bin "$spurBin"
527
569
  - kind: shell
528
570
  options:
529
571
  command: "$spurBin task verdict $wbs --from-answer .spur/run/$wbs-verify-answer.txt"
@@ -666,11 +708,11 @@ transitions:
666
708
  # ── precheck: size PASS + task check → implement; else → failed ──
667
709
  - from: precheck
668
710
  to: implement
669
- description: Deterministic size and task checks are green — begin implementation.
711
+ description: Deterministic size, evidence, and task checks are green — begin implementation.
670
712
  guard:
671
713
  kind: shell
672
714
  options:
673
- command: 'test "$(cat .spur/run/$wbs-precheck-size.status 2>/dev/null)" = PASS && $spurBin task check $wbs'
715
+ command: 'test "$(cat .spur/run/$wbs-precheck-size.status 2>/dev/null)" = PASS && test "$(cat .spur/run/$wbs-precheck-evidence.status 2>/dev/null)" = PASS && $spurBin task check $wbs'
674
716
  - from: precheck
675
717
  to: failed
676
718
  description: Size and/or task check failed — stop before implement.
@@ -27,7 +27,7 @@ vars:
27
27
  wbs: "0000"
28
28
  profile: "standard"
29
29
  spurBin: "spur"
30
- agent: "omp"
30
+ agent: "auto"
31
31
  stepTimeoutMs: "1800000"
32
32
  approval: "required"
33
33
  resolutionMode: "research"
@@ -49,7 +49,7 @@ vars:
49
49
  profile: "standard"
50
50
  merge: "false"
51
51
  spurBin: "spur"
52
- agent: "omp"
52
+ agent: "auto"
53
53
  stepTimeoutMs: "1800000"
54
54
  # Corpus-aware quality gate for the feature-transition hop (R1, task 0625).
55
55
  # The per-task gate (`spur-check`) deliberately excludes corpus-check — the only
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@gobing-ai/spur",
3
- "version": "0.3.68",
3
+ "version": "0.3.70",
4
4
  "description": "Spur CLI — local-first harness for mainstream coding agents: constraint checking, workflow orchestration, agent health, and history analytics. Bun-native; exposes the `spur` command.",
5
5
  "keywords": [
6
6
  "spur",
@@ -53,14 +53,14 @@
53
53
  },
54
54
  "devDependencies": {
55
55
  "@commander-js/extra-typings": "^14.0.0",
56
- "@gobing-ai/ts-db": "^0.4.48",
57
- "@gobing-ai/ts-ai-runner": "^0.4.48",
58
- "@gobing-ai/ts-dual-workflow-engine": "^0.4.48",
59
- "@gobing-ai/ts-infra": "^0.4.48",
60
- "@gobing-ai/ts-llm-jsonl-importer": "^0.4.48",
61
- "@gobing-ai/ts-rule-engine": "^0.4.48",
62
- "@gobing-ai/ts-runtime": "^0.4.48",
63
- "@gobing-ai/ts-utils": "^0.4.48",
56
+ "@gobing-ai/ts-db": "^0.4.49",
57
+ "@gobing-ai/ts-ai-runner": "^0.4.49",
58
+ "@gobing-ai/ts-dual-workflow-engine": "^0.4.49",
59
+ "@gobing-ai/ts-infra": "^0.4.49",
60
+ "@gobing-ai/ts-llm-jsonl-importer": "^0.4.49",
61
+ "@gobing-ai/ts-rule-engine": "^0.4.49",
62
+ "@gobing-ai/ts-runtime": "^0.4.49",
63
+ "@gobing-ai/ts-utils": "^0.4.49",
64
64
  "@types/bun": "1.3.14",
65
65
  "@types/figlet": "^1.7.0",
66
66
  "@types/node-notifier": "8.0.5",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "sp",
3
- "version": "0.3.68",
3
+ "version": "0.3.70",
4
4
  "description": "Spur — a local-first harness engineering toolkit that wraps mainstream coding agents with constraint checking, workflow orchestration, and history analytics.",
5
5
  "extensions": {
6
6
  "pi": ["./hooks/pi/guard-extension.ts"]
@@ -473,11 +473,7 @@ export function executeScripts(root: string = PLUGIN_ROOT): void {
473
473
  writeFileSync(
474
474
  fake,
475
475
  `#!/bin/sh
476
- if [ "$1" = task ] && [ "$2" = show ]; then
477
- printf '%s' '{"content":"### Requirements\\n- [ ] R1. x\\n### Plan\\n- [ ] p1"}'
478
- else
479
- printf '%s' '{"agents":[{"capabilityTier":"standard"}]}'
480
- fi
476
+ printf '%s' '{"content":"### Requirements\\n- [ ] R1. x\\n### Plan\\n- [ ] p1"}'
481
477
  `,
482
478
  );
483
479
  chmodSync(fake, 0o755);
@@ -485,7 +481,7 @@ fi
485
481
  try {
486
482
  execFileSync(
487
483
  process.execPath,
488
- [join(root, 'scripts', 'task-size-precheck.ts'), '0487', '--spur-bin', fake, '--executor', 'standard'],
484
+ [join(root, 'scripts', 'task-size-precheck.ts'), '0487', '--spur-bin', fake],
489
485
  { cwd: dir, encoding: 'utf8', timeout: 30_000, stdio: 'pipe' },
490
486
  );
491
487
  const content = readFileSync(statusPath, 'utf8');
@@ -687,15 +683,6 @@ export function probeJsonShapes(run: CliRunner = runCli): void {
687
683
  { file: 'plugins/sp/scripts/surface-drift-inventory.ts', line: 1 },
688
684
  );
689
685
  }
690
- const doctor = jsonEnvelopeShapes['spur agent doctor omp'];
691
- const capOk = (doctor?.keys ?? []).some((k) => k.endsWith('.capabilityTier'));
692
- record(
693
- 'agent doctor <name> --json -> agents[0].capabilityTier (asserted by task-size-precheck.ts:130)',
694
- 'json-exec(field-presence)',
695
- capOk ? 'ok' : 'mismatch',
696
- capOk ? 'field present in live envelope' : 'field ABSENT from live envelope',
697
- { file: 'plugins/sp/scripts/task-size-precheck.ts', line: 130 },
698
- );
699
686
  // Curated prose flag-claims: assertions phrased as prose ("no explicit `--flag`") that the
700
687
  // generic backtick-span extractor cannot scope to a command. Extend this list when a prose
701
688
  // claim is found; each entry is verified against the live help capture.
@@ -0,0 +1,181 @@
1
+ #!/usr/bin/env bun
2
+ /**
3
+ * task-evidence-precheck — deterministic evidence-channel precheck (R2, task 0726).
4
+ *
5
+ * Parses the task content for an exact `evidence-channel:` declaration and proves the
6
+ * declared live-data channel exists in the local spur database before implementation
7
+ * begins. Currently exactly one channel is allowlisted:
8
+ *
9
+ * evidence-channel: history_tool_call.args_raw[pi]
10
+ *
11
+ * …satisfied only when the fixed query
12
+ *
13
+ * SELECT COUNT(*) FROM history_tool_call WHERE args_raw IS NOT NULL AND source = 'pi'
14
+ *
15
+ * returns a positive count on `<cwd>/.spur/spur.db` — i.e. a live non-dry-run pi import
16
+ * has already preserved tool-call `args_raw` (0722 R1). Unknown declarations, a missing
17
+ * database, a missing table, and a zero count all fail closed.
18
+ *
19
+ * A task without any `evidence-channel:` declaration passes without opening SQLite —
20
+ * the check only gates tasks that declare a live-data evidence channel.
21
+ *
22
+ * Always exits 0 (soft action). Both precheck→implement guards in task-pipeline.yaml
23
+ * read the status file; a missing or failing checker writes FAIL, so readiness fails
24
+ * closed.
25
+ *
26
+ * Ships with the plugin to arbitrary projects; node-builtin + bun:sqlite only —
27
+ * no workspace imports.
28
+ *
29
+ * Usage:
30
+ * bun plugins/sp/scripts/task-evidence-precheck.ts <wbs> [--spur-bin <path>]
31
+ *
32
+ * Env: SPUR_BIN
33
+ */
34
+
35
+ import { Database } from 'bun:sqlite';
36
+ import { execFileSync } from 'node:child_process';
37
+ import { existsSync, mkdirSync, writeFileSync } from 'node:fs';
38
+ import { join } from 'node:path';
39
+ import { fileURLToPath } from 'node:url';
40
+
41
+ /** Exact task-content declaration that activates the live-evidence gate (0726 R2). */
42
+ const DECLARATION_PREFIX = 'evidence-channel:';
43
+
44
+ /** The only allowlisted live-data channel (0726 R2). */
45
+ const EVIDENCE_CHANNEL = 'history_tool_call.args_raw[pi]';
46
+
47
+ /** Declaration text as it must appear in the task body. */
48
+ const DECLARATION = `${DECLARATION_PREFIX} ${EVIDENCE_CHANNEL}`;
49
+
50
+ /** The only live-data query this precheck is allowed to run — fixed, never task-authored. */
51
+ const EVIDENCE_QUERY = "SELECT COUNT(*) AS n FROM history_tool_call WHERE args_raw IS NOT NULL AND source = 'pi'";
52
+
53
+ // ─── CLI (same spur-bin chain as task-size-precheck.ts) ─────────────────────
54
+
55
+ function usage(): never {
56
+ console.error('Usage: bun plugins/sp/scripts/task-evidence-precheck.ts <wbs> [--spur-bin <path>]');
57
+ process.exit(1);
58
+ }
59
+
60
+ function defaultSpurBin(): string {
61
+ if (process.env.SPUR_BIN) return process.env.SPUR_BIN;
62
+ const local = fileURLToPath(new URL('../../../apps/cli/src/index.ts', import.meta.url));
63
+ if (existsSync(local)) return `bun ${local}`;
64
+ return 'spur';
65
+ }
66
+
67
+ function parseArgs(argv: string[]): { wbs: string; spurBin: string } {
68
+ let spurBin = defaultSpurBin();
69
+ let wbs = '';
70
+ let i = 0;
71
+ while (i < argv.length) {
72
+ const arg = argv[i];
73
+ if (arg === '--spur-bin') {
74
+ spurBin = argv[i + 1] ?? defaultSpurBin();
75
+ i += 2;
76
+ } else if (!arg.startsWith('--')) {
77
+ wbs = arg;
78
+ i++;
79
+ } else {
80
+ i++;
81
+ }
82
+ }
83
+ if (!wbs) usage();
84
+ return { wbs, spurBin };
85
+ }
86
+
87
+ /**
88
+ * Split a multi-token `spurBin` (`<runtime> <mainModule>`) the same way
89
+ * `runSpurJson` does in feature-sync-bounded.ts — execFileSync's first arg is
90
+ * one executable path, not a shell command line.
91
+ */
92
+ function runSpur(spurBin: string, args: string[]): string {
93
+ const [file = 'spur', ...lead] = spurBin.split(/\s+/).filter(Boolean);
94
+ return execFileSync(file, [...lead, ...args], {
95
+ encoding: 'utf-8',
96
+ stdio: ['pipe', 'pipe', 'pipe'],
97
+ });
98
+ }
99
+
100
+ function writeStatus(wbs: string, status: 'PASS' | 'FAIL'): void {
101
+ const statusDir = join(process.cwd(), '.spur', 'run');
102
+ if (!existsSync(statusDir)) mkdirSync(statusDir, { recursive: true });
103
+ writeFileSync(join(statusDir, `${wbs}-precheck-evidence.status`), `${status}\n`);
104
+ }
105
+
106
+ function fail(wbs: string, reasons: string[]): void {
107
+ writeStatus(wbs, 'FAIL');
108
+ console.error(`task-evidence-precheck: FAIL`);
109
+ for (const r of reasons) {
110
+ console.error(` ${r}`);
111
+ }
112
+ process.exit(0);
113
+ }
114
+
115
+ function main(): void {
116
+ const { wbs, spurBin } = parseArgs(process.argv.slice(2));
117
+
118
+ let taskContent: string;
119
+ try {
120
+ const result = runSpur(spurBin, ['task', 'show', wbs, '--json']);
121
+ const task = JSON.parse(result);
122
+ taskContent = task.content ?? task.body ?? '';
123
+ } catch {
124
+ fail(wbs, [`could not fetch task ${wbs} via ${spurBin} — evidence channel unverifiable`]);
125
+ }
126
+
127
+ // Collect every declaration token. A repeated exact declaration still gates the
128
+ // single fixed query; any non-allowlisted token is an unknown declaration.
129
+ const declarations: string[] = [];
130
+ for (const match of taskContent.matchAll(/evidence-channel:\s*(\S+)/g)) {
131
+ declarations.push(match[1] ?? '');
132
+ }
133
+ const unknown = declarations.filter((d) => d !== EVIDENCE_CHANNEL);
134
+ if (unknown.length > 0) {
135
+ fail(wbs, [
136
+ `unknown evidence-channel declaration(s): ${unknown.join(', ')}`,
137
+ `allowlisted declaration: ${DECLARATION}`,
138
+ ]);
139
+ }
140
+ if (declarations.length === 0) {
141
+ writeStatus(wbs, 'PASS');
142
+ console.error(`task-evidence-precheck: PASS — no evidence-channel declaration; live-data gate not active`);
143
+ process.exit(0);
144
+ }
145
+
146
+ const dbPath = join(process.cwd(), '.spur', 'spur.db');
147
+ if (!existsSync(dbPath)) {
148
+ fail(wbs, [`spur database not found at ${dbPath} — run a real history import first`]);
149
+ }
150
+
151
+ let count: number;
152
+ try {
153
+ const db = new Database(dbPath, { readonly: true });
154
+ try {
155
+ const row = db.query(EVIDENCE_QUERY).get() as { n: number } | undefined;
156
+ count = row?.n ?? 0;
157
+ } finally {
158
+ db.close();
159
+ }
160
+ } catch (e) {
161
+ fail(wbs, [
162
+ `evidence query failed on ${dbPath}: ${e instanceof Error ? e.message : String(e)}`,
163
+ 'history_tool_call table missing or unreadable — run a real history import first',
164
+ ]);
165
+ }
166
+
167
+ if (!(count > 0)) {
168
+ fail(wbs, [
169
+ `0 live pi rows with args_raw (query: ${EVIDENCE_QUERY})`,
170
+ 'run a non-dry-run pi history import with a safe importer before implementing',
171
+ ]);
172
+ }
173
+
174
+ writeStatus(wbs, 'PASS');
175
+ console.error(
176
+ `task-evidence-precheck: PASS — ${count} live pi history_tool_call row(s) with args_raw (declaration: ${DECLARATION})`,
177
+ );
178
+ process.exit(0);
179
+ }
180
+
181
+ main();
@@ -1,24 +1,23 @@
1
1
  #!/usr/bin/env bun
2
2
  /**
3
- * task-size-precheck — pipeline size precheck guard (R2, task 0454) plus the
4
- * size-vs-executor-capability gate (R3, task 0487).
3
+ * task-size-precheck — pipeline size precheck guard (R2, task 0454; count-only
4
+ * since task 0723).
5
5
  *
6
6
  * Shells `spur task show <wbs> --json`, evaluates R-item count and Plan
7
- * checklist count against limits, writes PASS/FAIL to status file. With
8
- * `--executor`, also shells `spur agent doctor <exec> --json` and blocks a large
9
- * task routed to a sub-`capable-1` executor.
7
+ * checklist count against limits, writes PASS/FAIL to status file. No executor
8
+ * or doctor involvement: executor liveness/routing/capabilities are attested
9
+ * fail-closed at the `agent.run` dispatch boundary, not predicted here.
10
10
  *
11
- * Always exits 0 (soft check, like doctor). The precheck→implement guard in
12
- * task-pipeline.yaml reads the status file.
11
+ * Always exits 0 (soft action). The precheck→implement guard in
12
+ * task-pipeline.yaml reads the status file; a missing or failing checker writes
13
+ * FAIL, so readiness fails closed.
13
14
  *
14
15
  * Ships with the plugin to arbitrary projects, so it stays node-builtin-only —
15
- * no workspace imports. The capability tier therefore comes from the CLI rather
16
- * than from `getExecutorTier` directly; `spur agent doctor --json` exposes it as
17
- * `capabilityTier` precisely so the inference regex is not duplicated here.
16
+ * no workspace imports.
18
17
  *
19
18
  * Usage:
20
19
  * bun plugins/sp/scripts/task-size-precheck.ts <wbs> [--spur-bin <path>]
21
- * [--max-reqs <n>] [--max-plan-items <n>] [--executor <name>]
20
+ * [--max-reqs <n>] [--max-plan-items <n>]
22
21
  *
23
22
  * Env: SPUR_BIN, MAX_IMPLEMENT_REQS, MAX_IMPLEMENT_PLAN_ITEMS
24
23
  */
@@ -27,7 +26,6 @@ import { execFileSync } from 'node:child_process';
27
26
  import { existsSync, mkdirSync, writeFileSync } from 'node:fs';
28
27
  import { join } from 'node:path';
29
28
  import { fileURLToPath } from 'node:url';
30
- import { STAGE_FLOOR_TIER, TIER_ORDER } from './stage-registry-adapter';
31
29
 
32
30
  // ─── Regex (sync with packages/app/src/services/task-size-precheck.ts) ───────
33
31
 
@@ -37,32 +35,11 @@ const R_ITEM_RE = /^\s*-\s*\[[ xX]\]\s*(\*\*)?R\d+\./m;
37
35
  /** Matches checklist items under the Plan section. */
38
36
  const CHECKLIST_ITEM_RE = /^\s*-\s*\[[ xX]\]/m;
39
37
 
40
- /**
41
- * Large-task thresholds for the capability gate (R3, task 0487) — the DEFAULT
42
- * caps, not the overridable `--max-*` limits. Raising the caps says "I accept a
43
- * big task"; it does not make a flash-tier model able to finish one inside
44
- * `implementTimeoutMs`.
45
- */
46
- const LARGE_TASK_REQS = 5;
47
- const LARGE_TASK_PLAN_ITEMS = 8;
48
-
49
- /**
50
- * Capability tiers strong enough for a large task (R3, task 0487). The floor is
51
- * the `review` stage's Layer-1 tier — `reviewer` per `references/roles.md`,
52
- * read via the stage-registry adapter (0538 R4: no tier literal here; roles.md
53
- * is the pointer). Tiers at or above the floor pass. An unreachable roles.md
54
- * degrades to the pre-reconcile band — fail-closed for a safety gate.
55
- */
56
- const CAPABLE_TIERS: ReadonlySet<string> = (() => {
57
- const floor = STAGE_FLOOR_TIER.get('review') ?? 'capable-1';
58
- return new Set(TIER_ORDER.slice(Math.max(0, TIER_ORDER.indexOf(floor))));
59
- })();
60
-
61
38
  // ─── CLI ─────────────────────────────────────────────────────────────────────
62
39
 
63
40
  function usage(): never {
64
41
  console.error(
65
- 'Usage: bun plugins/sp/scripts/task-size-precheck.ts <wbs> [--spur-bin <path>] [--max-reqs <n>] [--max-plan-items <n>] [--executor <name>]',
42
+ 'Usage: bun plugins/sp/scripts/task-size-precheck.ts <wbs> [--spur-bin <path>] [--max-reqs <n>] [--max-plan-items <n>]',
66
43
  );
67
44
  process.exit(1);
68
45
  }
@@ -87,13 +64,14 @@ function parseArgs(argv: string[]): {
87
64
  spurBin: string;
88
65
  maxReqs: number;
89
66
  maxPlanItems: number;
90
- executor: string;
91
67
  } {
92
68
  let spurBin = defaultSpurBin();
93
69
  let wbs = '';
94
- let maxReqs = Number(process.env.MAX_IMPLEMENT_REQS) || 5;
95
- let maxPlanItems = Number(process.env.MAX_IMPLEMENT_PLAN_ITEMS) || 8;
96
- let executor = '';
70
+ // Doubled deterministic ceiling (0723 operator decision): 10 R-items / 16
71
+ // Plan items keep in sync with DEFAULT_TASK_SIZE_LIMITS in
72
+ // packages/app/src/services/task-size-precheck.ts (asserted by test).
73
+ let maxReqs = Number(process.env.MAX_IMPLEMENT_REQS) || 10;
74
+ let maxPlanItems = Number(process.env.MAX_IMPLEMENT_PLAN_ITEMS) || 16;
97
75
 
98
76
  let i = 0;
99
77
  while (i < argv.length) {
@@ -102,13 +80,10 @@ function parseArgs(argv: string[]): {
102
80
  spurBin = argv[i + 1] ?? defaultSpurBin();
103
81
  i += 2;
104
82
  } else if (arg === '--max-reqs') {
105
- maxReqs = Number(argv[i + 1]) || 5;
83
+ maxReqs = Number(argv[i + 1]) || 10;
106
84
  i += 2;
107
85
  } else if (arg === '--max-plan-items') {
108
- maxPlanItems = Number(argv[i + 1]) || 8;
109
- i += 2;
110
- } else if (arg === '--executor') {
111
- executor = argv[i + 1] ?? '';
86
+ maxPlanItems = Number(argv[i + 1]) || 16;
112
87
  i += 2;
113
88
  } else if (!arg.startsWith('--')) {
114
89
  wbs = arg;
@@ -119,7 +94,7 @@ function parseArgs(argv: string[]): {
119
94
  }
120
95
 
121
96
  if (!wbs) usage();
122
- return { wbs, spurBin, maxReqs, maxPlanItems, executor };
97
+ return { wbs, spurBin, maxReqs, maxPlanItems };
123
98
  }
124
99
 
125
100
  /**
@@ -135,29 +110,8 @@ function runSpur(spurBin: string, args: string[]): string {
135
110
  });
136
111
  }
137
112
 
138
- /**
139
- * Capability tier of `executor` per `spur agent doctor <exec> --json`.
140
- * Unknown executor, unreadable doctor output, or an undeclared-and-uninferrable
141
- * tier all read as `standard` — conservative: a false block is one flag away,
142
- * a false pass costs a 30-minute timed-out implement.
143
- */
144
- function resolveCapabilityTier(spurBin: string, executor: string): { tier: string; resolvedName: string } {
145
- try {
146
- const out = runSpur(spurBin, ['agent', 'doctor', executor, '--json']);
147
- const row = JSON.parse(out)?.agents?.[0];
148
- const tier = row?.capabilityTier;
149
- // R1 (0622 F2/F4 residue): `doctor <role>` resolves the role to its cheapest
150
- // eligible executor (`coder` → `omp`); surface the resolved executor name in
151
- // the block message, not the role the caller passed in.
152
- const resolvedName = typeof row?.agent === 'string' && row.agent.length > 0 ? row.agent : executor;
153
- return { tier: typeof tier === 'string' && tier ? tier : 'standard', resolvedName };
154
- } catch {
155
- return { tier: 'standard', resolvedName: executor };
156
- }
157
- }
158
-
159
113
  function main(): void {
160
- const { wbs, spurBin, maxReqs, maxPlanItems, executor } = parseArgs(process.argv.slice(2));
114
+ const { wbs, spurBin, maxReqs, maxPlanItems } = parseArgs(process.argv.slice(2));
161
115
 
162
116
  // Fetch task content via spur
163
117
  let taskContent: string;
@@ -166,7 +120,7 @@ function main(): void {
166
120
  const task = JSON.parse(result);
167
121
  taskContent = task.content ?? task.body ?? '';
168
122
  } catch {
169
- // If spur fails, write FAIL and exit 0 (soft, like doctor)
123
+ // If spur fails, write FAIL and exit 0 the status file carries the verdict
170
124
  const statusDir = join(process.cwd(), '.spur', 'run');
171
125
  if (!existsSync(statusDir)) mkdirSync(statusDir, { recursive: true });
172
126
  writeFileSync(join(statusDir, `${wbs}-precheck-size.status`), 'FAIL\n');
@@ -187,18 +141,6 @@ function main(): void {
187
141
  const planItemCount = planBody.match(new RegExp(CHECKLIST_ITEM_RE.source, 'gm'))?.length ?? 0;
188
142
 
189
143
  const reasons: string[] = [];
190
- // R3 (0487): a large task on a sub-capable executor blocks even when the caller
191
- // raised the caps — the caps are an acceptance of size, not a capability grant.
192
- if (executor && (reqCount > LARGE_TASK_REQS || planItemCount > LARGE_TASK_PLAN_ITEMS)) {
193
- const { tier, resolvedName } = resolveCapabilityTier(spurBin, executor);
194
- if (!CAPABLE_TIERS.has(tier)) {
195
- reasons.push(
196
- `Task size (${reqCount} R-items / ${planItemCount} Plan items) requires a capable executor, ` +
197
- `but ${resolvedName} is tier ${tier}. ` +
198
- `Pass \`--agent <capable>\` or \`--vars '{"implementAgent":"<capable>"}'\`, or split the task.`,
199
- );
200
- }
201
- }
202
144
  if (reqCount > maxReqs) {
203
145
  reasons.push(
204
146
  `Task has ${reqCount} R-items (max ${maxReqs}). ` +