@gobing-ai/spur 0.3.91 → 0.3.92

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/config/pipeline-budgets.json +2 -2
  3. package/config/plugin-scripts.json +17 -1
  4. package/config/workflows/feature-lifecycle.yaml +14 -5
  5. package/config/workflows/feature-verification.yaml +35 -27
  6. package/config/workflows/history-anatomy.yaml +14 -25
  7. package/config/workflows/idea-pipeline.yaml +9 -5
  8. package/config/workflows/task-pipeline.yaml +185 -4
  9. package/config/workflows/wrapup-pipeline.yaml +47 -5
  10. package/package.json +9 -9
  11. package/plugins/sp/agents/super-planner.md +14 -5
  12. package/plugins/sp/commands/dev-dogfood.md +4 -4
  13. package/plugins/sp/commands/dev-fixall.md +8 -5
  14. package/plugins/sp/commands/dev-run.md +6 -0
  15. package/plugins/sp/commands/dev-runall.md +12 -6
  16. package/plugins/sp/commands/dev-verify.md +9 -0
  17. package/plugins/sp/commands/dev-verifyall.md +5 -0
  18. package/plugins/sp/lib/idea-handoff.generated.mjs +301 -300
  19. package/plugins/sp/lib/inline-run.generated.d.mts +17 -0
  20. package/plugins/sp/lib/inline-run.generated.mjs +1460 -0
  21. package/plugins/sp/plugin.json +1 -1
  22. package/plugins/sp/references/environment-lens.md +1 -1
  23. package/plugins/sp/scripts/dogfood-testing/validate-report.mjs +136 -0
  24. package/plugins/sp/scripts/dogfood-testing/validate-report.ts +196 -2
  25. package/plugins/sp/scripts/feature-verification-steps.mjs +174 -0
  26. package/plugins/sp/scripts/feature-verification-steps.ts +275 -0
  27. package/plugins/sp/scripts/history-anatomy-cache.mjs +104 -4
  28. package/plugins/sp/scripts/history-anatomy-cache.ts +137 -13
  29. package/plugins/sp/scripts/inline-pipeline-parity-check.ts +1 -0
  30. package/plugins/sp/scripts/inline-run-setup.mjs +349 -0
  31. package/plugins/sp/scripts/inline-run-setup.ts +192 -75
  32. package/plugins/sp/scripts/record-feature-sync.mjs +63 -0
  33. package/plugins/sp/scripts/record-feature-sync.ts +84 -0
  34. package/plugins/sp/scripts/residual-scan.mjs +476 -0
  35. package/plugins/sp/scripts/residual-scan.ts +614 -0
  36. package/plugins/sp/scripts/task-evidence-precheck.ts +8 -3
  37. package/plugins/sp/scripts/task-size-precheck.ts +8 -3
  38. package/plugins/sp/skills/branch-workflow/SKILL.md +1 -0
  39. package/plugins/sp/skills/branch-workflow/references/worktree-patterns.md +2 -0
  40. package/plugins/sp/skills/code-implementation/SKILL.md +17 -0
  41. package/plugins/sp/skills/code-verification/SKILL.md +21 -0
  42. package/plugins/sp/skills/code-verification/references/verdict-schema.md +1 -0
  43. package/plugins/sp/skills/dogfood-testing/SKILL.md +5 -3
  44. package/plugins/sp/skills/dogfood-testing/references/monitor-ledger.md +63 -26
  45. package/plugins/sp/skills/dogfood-testing/references/report-template.md +33 -10
  46. package/plugins/sp/skills/history-anatomy/references/modes.md +5 -3
  47. package/plugins/sp/skills/next-feature/references/ranking-rubric.md +1 -1
  48. package/plugins/sp/skills/next-router/references/routing-table.md +7 -0
  49. package/plugins/sp/skills/parallel-execution/references/dispatch-surface.md +1 -1
  50. package/plugins/sp/skills/session-review/SKILL.md +12 -2
  51. package/plugins/sp/skills/spur-cli/references/features.md +1 -1
  52. package/plugins/sp/skills/spur-cli/references/projects.md +3 -1
  53. package/plugins/sp/skills/spur-cli/references/workflows.md +39 -19
  54. package/plugins/sp/skills/spur-dev/SKILL.md +11 -4
  55. package/plugins/sp/skills/spur-dev/references/cross-cutting.md +4 -1
  56. package/plugins/sp/skills/spur-dev/references/dev-operations.md +2 -2
  57. package/plugins/sp/skills/spur-dev/references/execution-batch.md +285 -57
  58. package/plugins/sp/skills/spur-dev/references/execution-workflow.md +1 -1
  59. package/plugins/sp/skills/spur-dev/references/flag-glossary.md +19 -4
  60. package/plugins/sp/skills/spur-dev/references/inline-pipeline-driver.md +43 -16
  61. package/plugins/sp/skills/spur-dev/references/planning-workflow.md +5 -0
  62. package/spur.js +21827 -19615
  63. package/web/_astro/BoardApp.CerSBgis.js +192 -0
  64. package/web/_astro/BoardApp.eoTz0pZs.js +1 -0
  65. package/web/_astro/{TaskDetail.CXGltuT_.js → TaskDetail.DCqiC-OZ.js} +1 -1
  66. package/web/_astro/arc.CPwg6Rw0.js +1 -0
  67. package/web/_astro/{architectureDiagram-3BPJPVTR.DJ8DHkWE.js → architectureDiagram-3BPJPVTR.DM_vp_hO.js} +1 -1
  68. package/web/_astro/{blockDiagram-GPEHLZMM.D8DHK3Jl.js → blockDiagram-GPEHLZMM.DXVIiv0p.js} +1 -1
  69. package/web/_astro/{c4Diagram-AAUBKEIU.BugQbX9u.js → c4Diagram-AAUBKEIU.BbF_zCxW.js} +1 -1
  70. package/web/_astro/channel.MYZLKNwy.js +1 -0
  71. package/web/_astro/{chunk-2J33WTMH.Mv26KlVn.js → chunk-2J33WTMH.CAgQHpPC.js} +1 -1
  72. package/web/_astro/{chunk-4BX2VUAB.CD51JoT_.js → chunk-4BX2VUAB.BN-5tpw4.js} +1 -1
  73. package/web/_astro/{chunk-55IACEB6.D_PFaIEe.js → chunk-55IACEB6.CnPkEEr0.js} +1 -1
  74. package/web/_astro/{chunk-727SXJPM.DemMW1ao.js → chunk-727SXJPM.BQzQeMVm.js} +4 -4
  75. package/web/_astro/{chunk-AQP2D5EJ.Iu2V5-ex.js → chunk-AQP2D5EJ.B6xNyDnL.js} +1 -1
  76. package/web/_astro/{chunk-FMBD7UC4.MsNgSP-E.js → chunk-FMBD7UC4.C7f9Ih78.js} +1 -1
  77. package/web/_astro/{chunk-ND2GUHAM.DfbRaAlm.js → chunk-ND2GUHAM.CNV1dFXT.js} +1 -1
  78. package/web/_astro/{chunk-QZHKN3VN.BbJAQ4h-.js → chunk-QZHKN3VN.Cudn2TkJ.js} +1 -1
  79. package/web/_astro/{classDiagram-4FO5ZUOK.CPurtiC2.js → classDiagram-4FO5ZUOK.D1NwP50q.js} +1 -1
  80. package/web/_astro/{classDiagram-v2-Q7XG4LA2.CPurtiC2.js → classDiagram-v2-Q7XG4LA2.D1NwP50q.js} +1 -1
  81. package/web/_astro/{cose-bilkent-S5V4N54A.CKKdx1bM.js → cose-bilkent-S5V4N54A.B1wSL-Xb.js} +1 -1
  82. package/web/_astro/{cynefin-OW5HDTMX.CoKMTg-R.js → cynefin-OW5HDTMX.BmK52w8G.js} +1 -1
  83. package/web/_astro/{dagre-BM42HDAG.D4h4_k56.js → dagre-BM42HDAG.Bfy5CTDT.js} +2 -2
  84. package/web/_astro/diagram-2AECGRRQ.DhNnvUvX.js +43 -0
  85. package/web/_astro/diagram-5GNKFQAL.lTX5KwnS.js +10 -0
  86. package/web/_astro/{diagram-KO2AKTUF.BFoCkiCr.js → diagram-KO2AKTUF.CW_vMJ4z.js} +3 -3
  87. package/web/_astro/{diagram-LMA3HP47.exHn9OVx.js → diagram-LMA3HP47.B_8ZGF67.js} +1 -1
  88. package/web/_astro/{diagram-OG6HWLK6.CeqO34nN.js → diagram-OG6HWLK6.BppnHsdS.js} +1 -1
  89. package/web/_astro/{erDiagram-TEJ5UH35.D_v7HqxR.js → erDiagram-TEJ5UH35.BEuHXcjJ.js} +5 -5
  90. package/web/_astro/{flowDiagram-I6XJVG4X.EUmrpbwh.js → flowDiagram-I6XJVG4X.CH-UlnGr.js} +4 -4
  91. package/web/_astro/{ganttDiagram-6RSMTGT7.BOCF5lII.js → ganttDiagram-6RSMTGT7.BO81S85v.js} +1 -1
  92. package/web/_astro/{gitGraphDiagram-PVQCEYII.Di7otYZD.js → gitGraphDiagram-PVQCEYII.XnPxPPZN.js} +1 -1
  93. package/web/_astro/index.Hjbr15fG.css +1 -0
  94. package/web/_astro/{infoDiagram-5YYISTIA.TcBkCAJk.js → infoDiagram-5YYISTIA.JyjYRu_T.js} +1 -1
  95. package/web/_astro/{ishikawaDiagram-YF4QCWOH.D-2y4M0c.js → ishikawaDiagram-YF4QCWOH.BBRBF-Fo.js} +5 -5
  96. package/web/_astro/{journeyDiagram-JHISSGLW.DPbJI_n2.js → journeyDiagram-JHISSGLW.C_iymSyp.js} +1 -1
  97. package/web/_astro/{kanban-definition-UN3LZRKU.EFxhQ9Fj.js → kanban-definition-UN3LZRKU.DdfW-Oqt.js} +7 -7
  98. package/web/_astro/{linear.DSAsQLzs.js → linear.C2_IkbZT.js} +1 -1
  99. package/web/_astro/mermaid.core.GAOYeSR0.js +303 -0
  100. package/web/_astro/{mindmap-definition-RKZ34NQL.CJY1N_7V.js → mindmap-definition-RKZ34NQL.DAZIxQSK.js} +2 -2
  101. package/web/_astro/{pieDiagram-4H26LBE5.567ZNoL2.js → pieDiagram-4H26LBE5.CN8sIhKM.js} +3 -3
  102. package/web/_astro/{quadrantDiagram-W4KKPZXB.mqfz9-MY.js → quadrantDiagram-W4KKPZXB.3dGcX5GP.js} +1 -1
  103. package/web/_astro/{requirementDiagram-4Y6WPE33.Bv1Gv9In.js → requirementDiagram-4Y6WPE33.BV2y4dd6.js} +3 -3
  104. package/web/_astro/{sankeyDiagram-5OEKKPKP.B6Gs4X4r.js → sankeyDiagram-5OEKKPKP.Cqo15Tvo.js} +4 -4
  105. package/web/_astro/{sequenceDiagram-3UESZ5HK.BhYj4v-m.js → sequenceDiagram-3UESZ5HK.CROCPMJB.js} +1 -1
  106. package/web/_astro/{stateDiagram-AJRCARHV.BPbBnkpw.js → stateDiagram-AJRCARHV.RfXZrkFE.js} +1 -1
  107. package/web/_astro/{stateDiagram-v2-BHNVJYJU.C4squMNK.js → stateDiagram-v2-BHNVJYJU.CPXmbBs9.js} +1 -1
  108. package/web/_astro/{timeline-definition-PNZ67QCA.C_SwIHgl.js → timeline-definition-PNZ67QCA.DdgKTiO8.js} +3 -3
  109. package/web/_astro/{vennDiagram-CIIHVFJN.Bz4NZGpQ.js → vennDiagram-CIIHVFJN.CPNVSHF1.js} +5 -5
  110. package/web/_astro/{wardleyDiagram-YWT4CUSO.CozMVZ3i.js → wardleyDiagram-YWT4CUSO.CQhA0Jyr.js} +3 -3
  111. package/web/_astro/{xychartDiagram-2RQKCTM6.BwMGBwjB.js → xychartDiagram-2RQKCTM6.n61BWyy4.js} +1 -1
  112. package/web/index.html +2 -2
  113. package/web/_astro/BoardApp.BEDWpzsr.js +0 -188
  114. package/web/_astro/BoardApp.DQG2xfEz.js +0 -1
  115. package/web/_astro/arc.C0rrflm_.js +0 -1
  116. package/web/_astro/channel.SRrg1P-w.js +0 -1
  117. package/web/_astro/diagram-2AECGRRQ.DJ0h9zgw.js +0 -43
  118. package/web/_astro/diagram-5GNKFQAL.DvPk1jYd.js +0 -10
  119. package/web/_astro/index.Bx6GY4RH.css +0 -1
  120. package/web/_astro/mermaid.core.kAZjgJHG.js +0 -301
@@ -52,6 +52,23 @@ When this skill is entered via `/sp:dev-run --mode implement <wbs>` (the form
52
52
  The structural guard is the slash form itself (`--mode implement`). Prose in the workflow YAML
53
53
  `agent.run` `input` is the wrong place for this rule; it belongs here and in `dev-run.md`.
54
54
 
55
+ ## Escalation contract (task 0933, implement step only)
56
+
57
+ When the pipeline runs implement headless, you have a bounded channel to surface a question
58
+ instead of guessing. The rules, in decision order:
59
+
60
+ - **Decide without asking** when the answer follows from the task's frozen Design /
61
+ Requirements / Q&A, the project and global instructions, or the codebase itself. Ambiguity
62
+ you can resolve from evidence is not a question.
63
+ - **Escalate only when a requirement or design ambiguity would change scope, correctness, or
64
+ authorization.** Style preferences, implementation details, and curiosity are not escalations.
65
+ - **To escalate:** write ONE concise question — with options and a recommendation — to the
66
+ escalation file, then exit 0 **without further edits**. The pipeline pauses the run and an
67
+ operator answers; the transcript of prior Q/A pairs lives at the `--escalation-file` path
68
+ (the ask loop is bounded, default 2 — a third pause fails the task).
69
+ - **Read `--escalation-file` if it exists and treat its answers as binding.** They are the
70
+ operator's recorded decisions for this pass, not suggestions.
71
+
55
72
  ## One WBS per implement pass (task 0487 R1)
56
73
 
57
74
  The target WBS is the **only** task you implement. Sibling tasks in the corpus are context you do
@@ -259,13 +259,34 @@ the deterministic Testing writer `spur task record` (section authorship never ha
259
259
 
260
260
  ```bash
261
261
  # write .spur/run/<wbs>-verdict.json (shape in references/verdict-schema.md), then:
262
+ # F96 residual sweep (observe-only): scan + fold BEFORE record, under every --fix mode.
263
+ RESIDUAL=$(superskill script path sp residual-scan.mjs)
264
+ node "$RESIDUAL" scan --wbs <wbs> --base .spur/run/<wbs>-base.sha
265
+ node "$RESIDUAL" fold --wbs <wbs> --verdict .spur/run/<wbs>-verdict.json
262
266
  spur task record <wbs> --verdict-file .spur/run/<wbs>-verdict.json # renders ## Testing
263
267
  ```
264
268
 
269
+ > **Residual fold (F96).** `scan` reads `.spur/run/` artifacts and writes `residuals.json` +
270
+ > `residual-report.md`; `fold` rewrites the **just-written** verdict artifact's `residual-sweep`
271
+ > check in place (PASS → PARTIAL when a blocking residual exists), so `record` transcribes the
272
+ > downgraded verdict — a manual verify cannot certify what the pipeline would reject. Resolve the
273
+ > script via `superskill script path sp residual-scan.mjs`; shipped surfaces never reference
274
+ > `plugins/sp/scripts/` directly (script-contract-check rule 4). When the task reaches `done`
275
+ > through `--next`, run `residual-scan settle` (links follow-up tasks; best-effort).
276
+
265
277
  > **Corrections: the answer file is the source of truth.** `spur task record` re-transcribes
266
278
  > `## Testing` from the verdict artifact — direct `--section Testing` writes are futile. Fix
267
279
  > `.spur/run/<wbs>-verify-answer.txt` → `spur task verdict <wbs> --from-answer <file>` → re-record.
268
280
 
281
+ > **Scenario-key carry-forward (standalone `--force` re-verifies).** When you author a fresh
282
+ > verdict artifact for a task whose `## Testing` already carries feature scenario-title rows
283
+ > (`Scenario: <title>` / `AC-N` keys with MET status), **copy those rows into the new artifact**
284
+ > keyed the same way — record re-transcribes `## Testing` wholesale, so a fresh artifact keyed by
285
+ > bare `R1`-style ids drops the scenario keys `spur feature check` needs for satisfaction
286
+ > (`L4.scenario-unverified` regresses; 0921/D63). Since 0936, `record` warns on stderr (exit 0)
287
+ > for each dropped MET-matched scenario key and when the new rows match no feature scenario —
288
+ > treat those warnings as a re-key instruction, not noise.
289
+
269
290
  > **Do not write `## Review` directly, ever.** The `## Review` section is owned by the
270
291
  > `review` coordinator (`/sp:dev-review` → `sp:super-reviewer`), which merges
271
292
  > `functional-review` + `code-verification` review mode + `code-improvement` fragments. The
@@ -136,6 +136,7 @@ Wave C verification can emit the following additive `checks[]` rows:
136
136
  | `evidence-rule-pass` | All behavior-bearing AC rows had executable evidence or were explicitly non-behavioral. |
137
137
  | `evidence-rule-failed` | One or more MET behavior-bearing AC rows lacked `test` / `command` evidence and were downgraded to PARTIAL. |
138
138
  | `cli-golden-path-present` | CLI-surface tasks supplied, or failed to supply, one golden-path command evidence row. |
139
+ | `residual-sweep` | Post-verdict residual scan (F96): `fail` when blocking leftovers exist (P1–P3 findings, added diff markers, unchecked boxes); evidence lists blocking/deferrable/advisory/housekeeping counts plus blocking and deferrable item ids. A `fail` downgrades an otherwise-PASS verdict to PARTIAL via the fold step. |
139
140
 
140
141
  ## How the gate reads it
141
142
 
@@ -194,8 +194,9 @@ On **every** step resolve:
194
194
 
195
195
  The final report MUST include a `### 3. Monitor Ledger` section containing those rows (cardinality:
196
196
  row count == the declared executed steps). Cardinality, full methodology, column contract,
197
- token/cache estimation, multi-source Cost honesty, the cache-health finding rule, and the
198
- **cache-conservation discipline** live in
197
+ token/context-reuse estimation, multi-source Cost honesty, the evidence-based reuse and
198
+ pipeline-provenance observation rules (supported 0912 baseline findings, owner handoffs,
199
+ INSUFFICIENT_EVIDENCE limits — task 0913), and the **cache-conservation discipline** live in
199
200
  **[monitor-ledger.md](references/monitor-ledger.md)** — apply conservation while monitoring; low
200
201
  cache% is usually the driver re-fetching data it already holds.
201
202
 
@@ -532,7 +533,8 @@ Do NOT:
532
533
  - [references/report-template.md](references/report-template.md) — report section contract,
533
534
  mandatory footer, task-sink L3 rule.
534
535
  - [references/monitor-ledger.md](references/monitor-ledger.md) — live-ledger column contract,
535
- token/cache estimation, cache-health finding rule.
536
+ token/context-reuse estimation, evidence-based reuse observation rule, pipeline-run provenance
537
+ observation rules.
536
538
 
537
539
  ## Platform Notes
538
540
 
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: monitor-ledger
3
- description: "The dogfood monitor methodology + on-disk live ledger column contract + dual-write + token/cache estimation heuristic + the cache-health finding rule. The on-disk ledger is the single source of truth the report is assembled from — recorded live, per step, never reconstructed."
3
+ description: "The dogfood monitor methodology + on-disk live ledger column contract + dual-write + token/cache estimation heuristic + evidence-based context-reuse observation rules. The on-disk ledger is the single source of truth the report is assembled from — recorded live, per step, never reconstructed."
4
4
  see_also:
5
5
  - dogfood-testing
6
6
  - report-template
@@ -99,14 +99,16 @@ in the report's §6 Findings (no exemption applies).
99
99
  | `Finding` | One-line finding surfaced at this step, or `—`. A finding does **not** change `Outcome`. |
100
100
  | `Fresh Tokens` | Estimated fresh context for the step. Prefix with `~`. |
101
101
  | `Cached Tokens` | Estimated reused context for the step. Prefix with `~`. |
102
- | `Cache %` | `Cached Tokens / (Fresh Tokens + Cached Tokens)`, rounded to the nearest whole percent. An `~unknown` row carries `—`, never `0%` — unknown basis is not an observed zero. |
102
+ | `Cache %` | Estimated context-reuse share: `Cached Tokens / (Fresh Tokens + Cached Tokens)`, rounded to the nearest whole percent. This is a **heuristic estimate** of how much context the step reused — it is not observed provider cache behavior (task 0913, R3). An `~unknown` row carries `—`, never `0%` — unknown basis is not an observed zero. |
103
103
  | `Basis` | Observable basis for the estimate: command output, prior file read reused, generated text, etc. |
104
104
  | `Wall-clock` | Elapsed time for the step. |
105
105
 
106
- ## Token + cache estimation heuristic
106
+ ## Token + context-reuse estimation heuristic
107
107
 
108
108
  A skill **cannot read its own exact token meter** — derive an estimate and label every number
109
- `~estimate`. The accepted methodology is deterministic from the ledger rows:
109
+ `~estimate`. These chars/4 figures estimate **context volume and reuse within the driver's own
110
+ session**; they do not measure provider cache hits, cache cost, or realized savings (task 0913,
111
+ R3). The accepted methodology is deterministic from the ledger rows:
110
112
 
111
113
  1. Estimate **Fresh Tokens** from new material consumed or produced by the step:
112
114
  - text read from files or command output: `ceil(characters / 4)`, rounded to the nearest 100;
@@ -122,8 +124,9 @@ A skill **cannot read its own exact token meter** — derive an estimate and lab
122
124
  both sums (or surfaced as a separate unknown bucket), never folded in as `Cached ~0`:
123
125
  `aggregate cache% = round(sum(Cached Tokens) / sum(Fresh Tokens + Cached Tokens) * 100)`.
124
126
 
125
- The **trend across runs** is the signal, not the absolute value: rising cache% = the testee is
126
- reusing context efficiently; falling cache% = context bloat creeping in.
127
+ The **trend across runs** is the signal, not the absolute value: a falling estimated reuse share
128
+ suggests context bloat creeping in; a rising one, efficient reference reuse. Trend claims from
129
+ estimates stay labeled as such — they never become causal waste findings on their own.
127
130
 
128
131
  > Never print a precise token number you cannot substantiate. The numbers exist to show a *trend*,
129
132
  > not to bill anyone.
@@ -169,27 +172,58 @@ missing, mark the row `~unknown`, exclude it from the aggregate (or surface it a
169
172
  bucket), and explain the missing basis — never fold it in as `Cached Tokens = ~0`: unknown cache use
170
173
  is not an observed zero-percent hit rate, and a low-cache diagnosis needs observed data.
171
174
 
172
- ## Cache-health finding rule
173
-
174
- Cache% is the operational signal for testee-tuning:
175
-
176
- - Any **individual step with cache% < 40%** → it is re-reading files or re-sending prompt context
177
- unnecessarily. Emit a **P3** finding naming that step, **even if the step succeeded**.
178
- - A run with **aggregate cache% < 50%** → the testee is a tuning candidate regardless of the
179
- PASS/PARTIAL/FAIL verdict. Emit a **P3** finding: "Low cache hit rate — candidate for
180
- context-window or prompt trimming."
181
-
182
- These feed the report's §6 Findings (see [report-template.md](report-template.md)).
175
+ ## Context-reuse observation rule (task 0913, R3 — replaces the fixed-threshold cache-health rule)
176
+
177
+ The estimated reuse share is a **heuristic** (chars/4 over the driver's own session); it cannot
178
+ prove provider cache behavior, unnecessary re-reading, or realized savings. Threshold crossings
179
+ (40%, 50%, or any other fixed line) therefore **never auto-generate a causal waste finding**.
180
+ Instead:
181
+
182
+ - **Estimated-only evidence** (chars/4 ledger, no external meter): a low estimated reuse share may
183
+ be reported only as a **labeled hypothesis** finding carrying the `[unverifiable]` tag, naming
184
+ the confirmation it needs (per-step provider telemetry, a real meter, or a measured reread
185
+ trace). Example shape: `P3 — estimated reuse share 36%, below the run's own trend — hypothesis:
186
+ repeated task-list refetching; confirm with per-step metering` `[unverifiable]`.
187
+ - **Measured evidence** (ccusage session/day delta, or agent usage fields in tool results): report
188
+ the measured figures with their scope label and confidence, and keep them out of per-step ledger
189
+ cells. A waste finding may then name the measured evidence — still without claiming causality
190
+ the meter does not show.
191
+ - **Unavailable evidence**: print `Meter: n/a` and totals `n/a` rather than a fabricated share;
192
+ route the missing-measurement gap to its owner as an improvement proposal.
193
+
194
+ Unobservable chained-step cost keeps its **mandatory** P3 finding (`P3 — chained-step cost not
195
+ observable`, task 0278 R3) — that rule reports missing evidence, not causal waste, and stays.
196
+
197
+ These findings feed the report's §6 Findings (see [report-template.md](report-template.md)).
198
+
199
+ ## Pipeline-run provenance observations (task 0913, R5)
200
+
201
+ When the testee drives a task pipeline, the driver may adopt the supported findings from the
202
+ 0912 workflow baseline (`docs/reports/i31/0912-workflow-baseline.md`) as bounded observations —
203
+ with artifact anchors and owner handoffs, never as invented performance conclusions:
204
+
205
+ - **Run-row closure / structured emission (supported, pilot-selected):** if the observed pipeline
206
+ run leaves a non-terminal run row or emits no `action_runs` rows, record the observation with
207
+ the run id as anchor and hand off to the owners named in the baseline (driver adoption: D62;
208
+ row-closure defect: P). These are the baseline's pilot-eligible findings (F1/F2).
209
+ - **Performance conclusions (INSUFFICIENT_EVIDENCE in the baseline):** no token/USD, percentage,
210
+ speedup, or fleet claim may be derived from a dogfood run. Record the limitation explicitly
211
+ (what was unmeasured and why) and route the missing evidence to the baseline's named gaps
212
+ (session/cost joins E6; scoped-gate experiment F3/F4; fleet serve session S5).
213
+ - Historical comparison, recurrence, and cohort trends stay with history-anatomy and existing
214
+ doctor tooling — the dogfood report owns only what its own ledger observed.
183
215
 
184
216
  ## Cache-conservation discipline (how to keep cache% high)
185
217
 
186
- The cache-health rule above *detects* waste; this section is the mitigation. The dogfooding driver
218
+ The context-reuse observation rule above interprets the signal; this section is the mitigation. The
219
+ dogfooding driver
187
220
  (the agent running Phase 2/3) controls most of the cache% it later reports — low cache% is usually
188
221
  the driver re-fetching data it already holds. Apply these while monitoring each step:
189
222
 
190
223
  ### Driver cache checklist (task 0278 R7)
191
224
 
192
- When aggregate cache% risks falling under 50%, apply this checklist **before** re-reading:
225
+ When the estimated reuse share looks low (or trends down across runs), apply this checklist
226
+ **before** re-reading:
193
227
 
194
228
  | # | Action | Why |
195
229
  | --- | -------- | ----- |
@@ -198,12 +232,13 @@ When aggregate cache% risks falling under 50%, apply this checklist **before** r
198
232
  | 3 | Prefer `--json` CLI over re-parsing freeform prose | Smaller, stable payloads |
199
233
  | 4 | Dual-write ledger rows without re-reading the whole report each step | Append/patch; don't full-file re-load |
200
234
  | 5 | Skip redundant `bun test` full suite between steps when a focused file suite already green | Run the broad suite once at the end |
201
- | 6 | For batch testees (`verifyall` / `runall` / `refineall`): freeze `task list --json` once at resolve | Re-listing the set per task is the #1 sub-50% cache pattern on feature dogfoods |
202
- | 7 | On re-verify of done tasks: re-read only cited `file:line` anchors, not full Solution blobs | Anchor-first re-verify keeps cache% above the 50% floor |
235
+ | 6 | For batch testees (`verifyall` / `runall` / `refineall`): freeze `task list --json` once at resolve | Re-listing the set per task is the #1 low-reuse-estimate pattern on feature dogfoods |
236
+ | 7 | On re-verify of done tasks: re-read only cited `file:line` anchors, not full Solution blobs | Anchor-first re-verify keeps the reuse estimate honest and high |
203
237
 
204
238
  1. **Reuse CLI output already in context.** If a prior step (or a prior tool call this step)
205
239
  captured `spur task show`/`check`/`list` output, do **not** re-invoke the same command for that
206
- data — reference the prior result. Re-invocation is the #1 cause of sub-40% steps. Only re-fetch
240
+ data — reference the prior result. Re-invocation is the #1 cause of low estimated-reuse steps.
241
+ Only re-fetch
207
242
  when the underlying state *changed* (e.g. you just wrote a section and need the new
208
243
  `requiredSections`).
209
244
  2. **Don't re-ground shared scaffolding per step.** Command docs, the skill preamble, and the
@@ -217,7 +252,8 @@ When aggregate cache% risks falling under 50%, apply this checklist **before** r
217
252
  useful as a trend if it reflects what actually happened.
218
253
 
219
254
  The point is not to game the number — it is to drive the testee (and your own monitoring) toward
220
- reusing context, which is the real cost saving the cache% signal stands for.
255
+ reusing context, which is the context-efficiency gain the estimated reuse share stands for. The
256
+ share itself remains a heuristic trend signal (task 0913, R3), not a measured saving.
221
257
 
222
258
  ## Worked ledger example
223
259
 
@@ -230,6 +266,7 @@ reusing context, which is the real cost saving the cache% signal stands for.
230
266
  | 4 profile | 1 | PASS | — | — | ~500 | ~350 | 41% | command output + prior profile reused | ~2s |
231
267
  ```
232
268
 
233
- Aggregate: total = `3700 + 2050 = 5750`; cached = `2050`; cache% =
234
- `round(2050 / 5750 * 100) = 36%` `[~estimate]` — below the 50% floor, so emit the P3 cache-health
235
- finding.
269
+ Aggregate: total = `3700 + 2050 = 5750`; cached = `2050`; estimated reuse share =
270
+ `round(2050 / 5750 * 100) = 36%` `[~estimate]`. A low estimated reuse share is reported only as a
271
+ labeled-hypothesis `[unverifiable]` finding (or backed by a real meter) — never as an automatic
272
+ causal waste claim (task 0913, R3).
@@ -96,8 +96,21 @@ Skeleton is written in Phase 1; filled as the run progresses; finalized in Phase
96
96
  - **Mode:** `observe-only (--max-retry 0)` | `fix (--max-retry N)`
97
97
  - **Task under test:** WBS + title (if applicable)
98
98
  - **Run id:** `<run_id>` · **Live:** `<live_path>` · **Report:** `<report_path>`
99
+ - **Source evidence:** `<commit sha>` + dirty-state (`clean` | `porcelain hash` | `unrecorded`)
100
+ - **Definition:** resolved definition digest, or `path+hash`; `unknown` when not resolvable
101
+ - **Execution mode:** `<observe-only | fix>` · driver/testee separation noted for chained legs
102
+ - **Measurement scope:** `<driver ledger (chars/4, per-step) | ccusage <day\|session> | agent usage fields>`; external meter name or `none`
103
+ - **Sample coverage:** `<k>/<n>` executed steps observed; testee workflow/run/session IDs `<ids | unknown>`
99
104
  ```
100
105
 
106
+ **Provenance and comparability (task 0913, R4 — additive).** The provenance lines above are
107
+ optional in `@1.2`: legacy reports without them remain readable and valid, but are **not
108
+ automatically eligible for comparison**. Before treating two reports as comparable, require
109
+ compatible measurement scope, execution mode, and resolved definition identity; otherwise emit
110
+ **`not comparable`** and route the missing evidence to its owner. Unsupported identifiers are
111
+ recorded as `unknown` — never invented. Driver cost and testee/chained-leg cost stay separate
112
+ rows; pilot comparisons must not mix them.
113
+
101
114
  When `status` is `aborted` or the report is partial, add under §1:
102
115
 
103
116
  ```
@@ -122,14 +135,22 @@ When `status` is `aborted` or the report is partial, add under §1:
122
135
 
123
136
  **Cost honesty rules:**
124
137
 
125
- - Always include ledger-derived `~estimate` total / cached / cache% with a **Method** line and
138
+ - Separate the three evidence classes in every Cost block (task 0913, R3): **measured** usage
139
+ (real meter, scope-labeled), **heuristic estimates** (chars/4 ledger, labeled `~estimate`,
140
+ confidence `LOW`), and **unavailable** values rendered `n/a` — never a fabricated number or a
141
+ share computed over unknown rows.
142
+ - Always include ledger-derived `~estimate` total / cached / reuse share with a **Method** line and
126
143
  **confidence** (`LOW` when estimate-only; `MEDIUM` when a real meter is also present).
127
144
  - Optional meters when available (never invent):
128
145
  - `ccusage` session/daily delta — label scope (`day` / `session`), **not** per-step
129
146
  - agent usage fields if present in tool results
130
- - If no meter: print `Meter: n/a` explicitly.
147
+ - If no meter: print `Meter: n/a` explicitly; if no ledger row is observable, print totals and
148
+ share as `n/a` rather than folding unknowns into zero.
131
149
  - Never present an unsubstantiated precise integer as billed/metered cost.
132
- - Aggregate cache% MUST equal the ledger formula (see §3); otherwise the report is invalid.
150
+ - Observable totals MUST satisfy `total = fresh + cached` and the share MUST equal the ledger
151
+ formula over observable rows within display rounding (±1 point); the validator (task 0913)
152
+ rejects violations. The chars/4 figures estimate the driver session's context reuse — they are
153
+ **not** provider cache measurements and establish no realized savings.
133
154
  - **Chained-step segmentation (@1.2):** when a derived step is implement-heavy (the step runs a
134
155
  pipeline leg, writes code, or mutates more than its own arguments), its cost MUST be a separate
135
156
  ledger row tagged `chained:<step>` and kept out of the driver's row. If the chained leg ran in a
@@ -251,8 +272,8 @@ from the trailing feasibility tag:
251
272
  ```
252
273
 
253
274
  Omitting the class preserves the current line shape; untagged findings remain valid and the
254
- protocol stays `sp:dogfood-testing@1.2` (the validator gains no required field; the cache-health
255
- P3 above needs no class).
275
+ protocol stays `sp:dogfood-testing@1.2` (the validator gains no required field; an
276
+ evidence-based reuse observation needs no class).
256
277
 
257
278
  - `testee` — a defect in the testee's contract (the protocol the run grades). Bounded fix-mode
258
279
  may repair it, unchanged.
@@ -278,13 +299,15 @@ Severity scale:
278
299
  finding:** when a drift row (`drift:external`) is present in the ledger, a P2 finding naming the
279
300
  drifted paths is mandatory in the report (not optional). The finding states the run's evidence is
280
301
  degraded, not voided. See [SKILL.md §Workspace-drift guard](../SKILL.md#workspace-drift-guard-r2--task-0296).
281
- - **P3** — efficiency / DX / observation (includes the cache-health rule below).
302
+ - **P3** — efficiency / DX / observation (includes evidence-based context-reuse observations below).
282
303
  - **P4** — nice-to-have, cosmetic, or speculative.
283
304
 
284
- **Cache-health rule** (from [monitor-ledger.md](monitor-ledger.md)): if aggregate cache% < 50% or any
285
- step < 40%, emit a **P3** — "Low cache hit rate — candidate for context-window or prompt trimming"
286
- with the offending step(s). Absolute token totals from the heuristic are trend-only (`[unverifiable]`
287
- as billable cost proof is expected).
305
+ **Context-reuse observation rule** (task 0913 — replaces the fixed-threshold cache-health rule;
306
+ see [monitor-ledger.md](monitor-ledger.md)): estimated reuse shares (chars/4) never auto-generate
307
+ causal waste findings. Report a low estimated share only as a labeled-hypothesis finding carrying
308
+ `[unverifiable]` and naming the confirmation needed, or cite measured meter evidence with its
309
+ scope. Absolute token totals from the heuristic are trend-only (`[unverifiable]` as billable cost
310
+ proof is expected).
288
311
 
289
312
  **Migration grep rule.** When dogfooding migrations or retired surfaces, distinguish intentional
290
313
  legacy-term mentions in guidance from live routed surfaces. Pair any broad grep for old skill or
@@ -1,7 +1,9 @@
1
1
  # Mode contract — `sp:history-anatomy` (HA-S1, 0658)
2
2
 
3
3
  The skill resolves exactly two modes. Everything else fails loud. This matrix is the enforcement
4
- surface the workflow (0660) and the skill share; keep the vocabulary frozen.
4
+ surface the workflow (0660) and the skill share; keep the vocabulary frozen. Execution note
5
+ (0920): argument validation runs deterministically in the `history-anatomy-cache` helper `paths`
6
+ command before the workflow starts; the model hop no longer performs it.
5
7
 
6
8
  ## Mode vocabulary (frozen)
7
9
 
@@ -26,7 +28,7 @@ Rejected arguments (each fails loud, naming the offending argument):
26
28
  | --- | --- |
27
29
  | focus text (positional) | Daily mode has no focus string. |
28
30
  | `--since` / `--until` | Daily always uses the calendar-day window. |
29
- | `--output` | Daily always writes to the run directory (see 0660). |
31
+ | `--output` | Daily always writes to `docs/report/<date>-history-anatomy.md` (the executing helper's default; 0660). |
30
32
 
31
33
  A daily invocation must print the normalized **inclusive ISO bounds** and the timezone used, so the
32
34
  wall-clock window is auditable.
@@ -40,7 +42,7 @@ Requires a **non-empty focus** and **two ordered inclusive bounds**.
40
42
  | focus (positional) | Required; a missing or empty focus fails loud. |
41
43
  | `--since <iso>` | Required; the inclusive lower bound. |
42
44
  | `--until <iso>` | Required; must be present with `--since`; must not be earlier than `--since`. |
43
- | `--output <path>` | Optional; when present, writes to that explicit path. When absent, writes to the run directory. |
45
+ | `--output <path>` | Optional; when present, writes to that explicit path. When absent, writes to `docs/report/<date>-history-anatomy.md` (the executing helper's default). |
44
46
 
45
47
  Rejected arguments (each fails loud, naming the offending argument):
46
48
 
@@ -40,7 +40,7 @@ and no command-derived number or `file:line` citation is a defect in the report,
40
40
  ## The gated list
41
41
 
42
42
  Gated features are listed **separately, never ranked**, each with its gate reason from the
43
- actionability pass (`blocked: 0142 — external trigger` / `no open tasks` / `all tasks terminal`).
43
+ actionability pass (`blocked: <wbs> — external trigger` / `no open tasks` / `all tasks terminal`).
44
44
  Features whose gate reason is "all tasks terminal" are T4 candidates — say so once, in the sync-first
45
45
  block, rather than repeating per row.
46
46
 
@@ -113,6 +113,13 @@ token, deterministic).
113
113
  | C3 | A5/A6 when Testing empty/N/A **and** verify would fail for missing tests — only if prior implement claims code exists | Coverage/test signal: `bun test` fail attributed to task paths OR explicit "insufficient tests" in prior verify verdict artifact `.spur/run/<wbs>-verdict.json` | test fail / coverage gap | `/sp:dev-unit <wbs> --auto` | continue |
114
114
  | C4 | A3/A5/A6 when operator or task tags mention rules, OR `spur rule run` last report dirty in `.spur/` if present | `spur rule run` (default project preset) non-zero with findings | rule findings | **HITL STOP** — print rule summary; suggest `/sp:rule-scan` or `rule-add`/`rule-refine` (do not auto-author rules) | continue |
115
115
  | C5 | A6 only | Existing `.spur/run/<wbs>-verdict.json` with FAIL and findings pointing at coverage | verdict artifact | `/sp:dev-unit <wbs>` then re-verify on next invocation (`--once` friendly) | `/sp:dev-verify …` |
116
+ | C6 | A4/A5 | `.spur/run/<wbs>-verdict.json` has a failing (PARTIAL/FAIL) `residual-sweep` check — the bounded remediation loop already ran and failed (F96) | folded verdict artifact | **HITL STOP** — print `.spur/run/<wbs>-residual-report.md` and the recovery command `/sp:dev-run <wbs>`; never auto-dispatch a fix (repeating the loop unattended burns quota without new information) | continue |
117
+
118
+ **C-row precedence note (F96):** C6 outranks C2/C3/C5 for the same task — a residual-sweep failure
119
+ means the workspace holds unfinished task residue, so fixall/unit reruns would either clean it
120
+ unintentionally or re-certify a folded verdict. C6 fires before any fix/unit dispatch; recovery is
121
+ the operator re-running the pipeline (`/sp:dev-run <wbs>`), which re-enters the normal
122
+ verify → test-fix remediation budget.
116
123
 
117
124
  **Explicit non-probes in v1:** no freeform chat history; no always-on full `bun run test` for every
118
125
  call; no git dirtiness as a route (optional advisory print only).
@@ -37,7 +37,7 @@ native subagent.
37
37
  | 1 | **Different model or coding agent required** | The step needs a model or a coding agent the host session cannot provide (`--model`, `--agent`). | "verify on o3" where the host is Claude Code; "run this through omp" from a non-omp host. |
38
38
  | 2 | **Headless or unattended step** | The step must run without a live session - scheduled, detached, or driven by a non-interactive caller. | A batch launched by `spur workflow run --async` with no operator attached. |
39
39
  | 3 | **Durable auditable run record required** | The dispatch must produce a persisted run record (cost ledger, trace, exit code) for after-the-fact audit. | `spur agent run` writes `.spur/run/` artifacts; a native subagent does not. |
40
- | 4 | **Workspace or credential isolation required** | The step must run in a separate workspace, worktree, or credential scope from the orchestrating session. | A destructive step isolated to a throwaway worktree; a step that must not inherit the session's `cwd` secrets. |
40
+ | 4 | **Workspace or credential isolation required** | The step must run in a separate workspace, worktree, or credential scope from the orchestrating session. | A destructive step isolated to a throwaway worktree; a step that must not inherit the session's `cwd` secrets; `--mode parallel` batch fan-out, one worktree per task ([execution-batch.md § Parallel isolation](../../spur-dev/references/execution-batch.md#parallel-isolation---mode-parallel)). |
41
41
 
42
42
  ## The naming requirement
43
43
 
@@ -84,7 +84,14 @@ three buckets — never skip triage and start fixing from the raw findings list.
84
84
  Apply the placement rule in
85
85
  [the environment-improvement mapping](../../references/environment-lens.md): automate with a
86
86
  check when possible, place coding standards on the review path, and keep always-loaded steering
87
- as navigation pointers.
87
+ as navigation pointers. When the session drove a task pipeline, at most three bounded
88
+ diagnostic questions may be drawn from the supported observations (F1/F2) and measured
89
+ overhead candidates (F3/F4) in the 0912 workflow baseline
90
+ (`docs/reports/i31/0912-workflow-baseline.md`) — e.g. repeated gate runs (F4), full-loss
91
+ test-fix timeouts (F3), or unemitted/non-terminal run rows (F1/F2) — each citing its artifact
92
+ anchor and owner handoff (driver adoption D62, row-closure defect P). Areas the baseline marks
93
+ INSUFFICIENT_EVIDENCE (token/USD, percentages, fleet) stay excluded; record the limitation
94
+ instead of a performance conclusion.
88
95
  5. **Render the report.** Use the exact compact output contract below. Omit empty table rows, not
89
96
  headings; write `None observed` when a section has no supported entry.
90
97
 
@@ -123,7 +130,10 @@ session. Do not list ordinary implementation steps as issues.
123
130
  ### Process and environment improvements
124
131
 
125
132
  For each supported proposal, name its owner surface, expected impact, verification method, and
126
- reversibility. Proposals remain report-only: apply no change and create no task.
133
+ reversibility. Proposals remain report-only: apply no change and create no task. A pipeline
134
+ observation adopted from the 0912 workflow baseline cites its anchor
135
+ (`docs/reports/i31/0912-workflow-baseline.md`) and owner handoff, and carries no unsupported
136
+ performance claim.
127
137
 
128
138
  ### Triage (only when `--triage` was passed)
129
139
 
@@ -28,7 +28,7 @@ what* or *how to write a scenario*, this skill.
28
28
  | `list` | List features, filtered | `--status <s>` `--priority <p>` `--folder` `--json` |
29
29
  | `move <id>` | Re-parent a subtree (cascade-rename of descendants) | `--parent <id>` `--dry-run` `--folder` `--json` |
30
30
  | `refresh` | Rebuild INDEX + each feature `## Tasks` table from task edges (**docs only**; no status change) | `--feature <id>` `--all` `--folder` `--json` |
31
- | `check [id]` | Validate one feature / the tree; the 4-layer gate; `--fix` repairs structural findings in place | `--strict` `--fix` `--folder` `--json` |
31
+ | `check [id]` | Validate one feature / the tree; the 4-layer gate; `--fix` repairs structural findings in place | `--strict` `--fix` `--as <status>` `--folder` `--json` |
32
32
  | `sync [id]` | Align feature **lifecycle status** with linked task states (real transitions + guards) | `--all` `--dry-run` `--force` `--folder` `--json` |
33
33
 
34
34
  **`refresh` vs `sync` (do not conflate):**
@@ -19,6 +19,7 @@ shapes live in `apps/cli/src/commands/projects.ts`.
19
19
  | ---- | ------- | --------- |
20
20
  | `add <path>` | Upsert an existing path in the registry | `--name <name>` `--json` |
21
21
  | `remove <target>` | Remove an entry by display name or path | `--json` |
22
+ | `clean` | Purge missing project folders and terminate lingering processes (alias: `refresh`) | `--no-terminate-processes` `--json` |
22
23
  | `list` | List entries with live running status | `--json` `--fleet` |
23
24
  | `start <target>` | Start or reuse a detached project server | `--port <n>` `--json` |
24
25
  | `stop <target>` | Best-effort stop the listener and clear its recorded port | `--json` |
@@ -35,7 +36,8 @@ exit `0`; validation, registry, spawn, health, or lookup failure is exit `1`.
35
36
  - `add` requires an existing path, resolves a relative path from the current working directory, and
36
37
  defaults the display name to its basename. It upserts; it does not start a server. The current
37
38
  source does not enforce a `.spur/` marker or directory type.
38
- - `list` probes recorded ports and heals stale entries to `port: 0` before reporting `running`.
39
+ - `list` automatically heals tilde paths, verifies project directories exist on disk (purging missing entries and terminating lingering processes), and heals stale entries to `port: 0` before reporting `running`.
40
+ - `clean` (alias `refresh`) explicitly purges non-existent project directories from `projects.json`, probes live ports for removed entries, and cleanly terminates orphaned listening processes (SIGTERM with bounded wait and SIGKILL escalation). Supports `--no-terminate-processes` to skip process kills.
39
41
  - `list --fleet` (0835/0858) additionally resolves each project's `agent.fleet` section from that
40
42
  project's `.spur/config.yaml` under the existing verb (no new noun). Per project it prints one line
41
43
  per member: instance id (the spec id / mailbox identity), `role`, resolved `executor`,
@@ -110,11 +110,11 @@ The skill's logic divides by **whether the LLM adds value**:
110
110
  | --------- | --------- | ----- | ------------------ |
111
111
  | `validate` | `spur workflow validate` (CLI) | `<file> [--no-schema]` | Schema + semantic verdict |
112
112
  | `run` | `spur workflow run` (CLI) | `<file> [--run-id <id>] [--vars <json>] [--dry-run] [--async] [--no-plan] [--quiet/--silent/--verbose] [--detail <level>] [--trace-file] [--no-log] [--steer]` | Terminal state reached (sync) or run started (async); trace readable |
113
- | `continue` | `spur workflow continue` (CLI) | `[run-id] [--yes] [--answer <yes\|no\|cancel>] [--async] [--no-log]` | Resume a paused or interrupted run (omit id -> most recent resumable); `--answer` injects a gate answer before guard re-evaluation and is required headless (0901 R3); `--async` detaches the resume (0901 R4) |
113
+ | `continue` | `spur workflow continue` (CLI) | `[run-id] [--yes] [--answer <yes\|no\|cancel>] [--answer-text <text>] [--async] [--no-log]` | Resume a paused or interrupted run (omit id -> most recent resumable); `--answer` injects a confirm/select gate answer and `--answer-text` answers an input gate with free text (H1 R27), both before guard re-evaluation; one of them is required headless (0901 R3) and each is validated against the pending gate kind; `--async` detaches the resume (0901 R4) |
114
114
  | `cancel` | `spur workflow cancel` (CLI) | `<run-id>` | Single non-terminal run marked failed (SIGTERM async worker when live) |
115
115
  | `clean` | `spur workflow clean` (CLI) | `[--older-than <min>] [--force] [--logs] [--dry-run]` | Bulk-finalize stale `running`/`pending` runs as failed **and** reclaim retained run logs older than `workflow.logRetentionDays` (30d default) |
116
116
  | `list` | `spur workflow list` (CLI) | — | Available workflow **YAML definition files** (not run records) |
117
- | `trace` | `spur workflow trace` (CLI) | `[run-id] [--workflow <n>] [--status <s>] [--since <iso>] [--last <n>] [--follow] [--poll <ms>] [--output]` | Run history list or per-run timeline |
117
+ | `trace` | `spur workflow trace` (CLI) | `[run-id] [--workflow <n>] [--status <s>] [--since <iso>] [--last <n>] [--follow] [--poll <ms>] [--output] [--timeout <ms>]` | Run history list or per-run timeline |
118
118
  | `progress` | `spur workflow progress` (CLI) | `<run-id>` | The `projectWorkflowProgress` projection for that run — current state, per-action attempts, candidate next transitions, diagnostics. Read-only; the verb renders, `packages/app` derives. Unknown run id exits 1 with `Run <id> not found.` |
119
119
  | `add` | agent procedure | `"<nl-description>" [--kind <state-machine\|transition-flow>] [--file <path>]` | **Mode chosen (confirmed)** → first reconciled against existing workflows (extend an existing flow rather than duplicate) → YAML authored in real schema shape → **validated AND dry-run** (reaches the expected terminal state) → [add](workflows/operations.md#add) |
120
120
  | `refine` | agent procedure | `<workflow-file> [--intent "<goal>"] [--dry-run]` | Smallest change meeting the intent, re-validated and re-dry-run; `--dry-run` emits a diff only → [refine](workflows/operations.md#refine) |
@@ -204,7 +204,7 @@ spur workflow run ./workflows/approval.yaml --silent # errors only
204
204
  spur workflow run ./workflows/approval.yaml --verbose # transitions + correlation diagnostics
205
205
  spur workflow run ./workflows/approval.yaml --detail minimal # tersest human output
206
206
  spur workflow run ./workflows/approval.yaml --trace-file # persist redacted JSONL trace
207
- spur workflow run ./workflows/approval.yaml --no-log # opt out of the consolidated .spur/run/<RUNID>.log
207
+ spur workflow run ./workflows/approval.yaml --no-log # opt out of the run record .spur/run/<RUNID>.md + .state.json
208
208
  spur workflow run ./workflows/approval.yaml --steer # interactive steering on stdin
209
209
  ```
210
210
 
@@ -214,8 +214,8 @@ spur workflow run ./workflows/approval.yaml --steer # interactive
214
214
  per-step headers), `full` (transitions + correlation). `--verbose` is shorthand for `--detail full`.
215
215
  - **`--trace-file`** appends a redacted, schema-versioned JSONL trace under `.spur/workflow/`
216
216
  for post-run analysis - independent of human/JSON output.
217
- - **`--no-log`** opts out of writing the consolidated all-in-one run log (`.spur/run/<RUNID>.log`).
218
- By default the log is written **and retained** after the run ends; this flag skips it entirely
217
+ - **`--no-log`** opts out of writing the two-file run record (`.spur/run/<RUNID>.md` + `.state.json`).
218
+ By default the record is written **and retained** after the run ends; this flag skips it entirely
219
219
  (propagates to the `--async` detached worker). No `--keep-log` / delete-by-default exists.
220
220
  - **`--steer`** is synchronous and in-process: it cannot combine with `--json` or `--async` (exit `2`).
221
221
  It accepts steering commands on stdin at declared action boundaries for interactive control.
@@ -260,11 +260,11 @@ the operator accepts it, and never hot-edit a running workflow's shell in place.
260
260
  spur workflow validate <file> [--no-schema] [--json]
261
261
  spur workflow show <file> [--format <mermaid|todo>] [--json]
262
262
  spur workflow run <file> [--run-id <id>] [--vars <json>] [--dry-run] [--async] [--no-plan] [--quiet/--silent/--verbose] [--detail <level>] [--trace-file] [--no-log] [--steer] [--json]
263
- spur workflow continue [run-id] [--yes] [--answer <yes|no|cancel>] [--async] [--no-log] [--json]
263
+ spur workflow continue [run-id] [--yes] [--answer <yes|no|cancel>] [--answer-text <text>] [--async] [--no-log] [--json]
264
264
  spur workflow cancel <run-id> [--json]
265
265
  spur workflow clean [--older-than <minutes>] [--force] [--logs] [--dry-run] [--json]
266
266
  spur workflow list [--json]
267
- spur workflow trace [run-id] [--workflow <name>] [--status <s>] [--since <iso>] [--last <n>] [--follow] [--poll <ms>] [--output] [--json]
267
+ spur workflow trace [run-id] [--workflow <name>] [--status <s>] [--since <iso>] [--last <n>] [--follow] [--poll <ms>] [--output] [--timeout <ms>] [--json]
268
268
  spur workflow progress <run-id> [--json]
269
269
  ```
270
270
 
@@ -285,7 +285,7 @@ not advertise `--json-envelope` because its JSON projection is a kept-raw docume
285
285
  | `--verbose` | Include transitions and correlation diagnostics in human progress (implies `--detail full`). |
286
286
  | `--detail <level>` | Human detail level: `minimal`, `invocation` (default), or `full`. |
287
287
  | `--trace-file` | Append a redacted schema-versioned JSONL trace under `.spur/workflow/`. |
288
- | `--no-log` | Opt out of writing the consolidated `.spur/run/<RUNID>.log` (retained by default; propagates to `--async` workers). |
288
+ | `--no-log` | Opt out of writing the two-file run record `.spur/run/<RUNID>.md` + `.state.json` (retained by default; propagates to `--async` workers). |
289
289
  | `--steer` | Accept in-process steering commands on stdin at declared action boundaries (sync only; incompatible with `--json`/`--async`). |
290
290
 
291
291
  `validate` and `run` exit non-zero on failure (`run` exits non-zero when the final status is not
@@ -302,32 +302,51 @@ Follow a live run to terminal (human streaming mode):
302
302
  ```bash
303
303
  spur workflow trace <run-id> --follow # stream until terminal (default 1000ms poll)
304
304
  spur workflow trace <run-id> --follow --poll 500 # poll every 500ms
305
- spur workflow trace <run-id> --follow --output # stream .spur/run/<RUNID>.log instead of the DB timeline
305
+ spur workflow trace <run-id> --follow --output # stream .spur/run/<RUNID>.md instead of the DB timeline
306
+ spur workflow trace <run-id> --follow --timeout 600000 # bound the watch; timeout → one checkpoint, run continues, exit 1
306
307
  ```
307
308
 
308
309
  - **`--follow`** replays a run timeline and polls persisted state until it becomes terminal. It
309
310
  requires a `run-id` (exit `2` without one) and cannot combine with `--json` (exit `2` - it is a
310
311
  human streaming mode).
311
312
  - **`--poll <ms>`** sets the follow polling interval (default `1000`, minimum `50`; exit `2` otherwise).
312
- - **`--output`** swaps the follow source from the structured DB timeline to the consolidated all-in-one
313
- log (`.spur/run/<RUNID>.log`, tail -f equivalent), streaming new lines as they land and exiting at
314
- terminal status. It requires `--follow` and a `run-id`, is a human stream (rejects `--json`), and is
315
- a **distinct source** — it never interleaves with the DB timeline. If the log never appears (e.g. the
313
+ - **`--timeout <ms>`** (requires `--follow`, positive integer; otherwise exit `1` with `VALIDATION_FAILED`)
314
+ bounds the watch (0930): when the deadline passes before the run is terminal — including the `Run not
315
+ found` registration retry window — the follow prints one checkpoint naming the run id and last observed
316
+ status and exits `1`. A watch timeout never cancels or relaunches the run; resume by re-running the same
317
+ follow command.
318
+ - **`--output`** swaps the follow source from the structured DB timeline to the human run record
319
+ (`.spur/run/<RUNID>.md`, tail -f equivalent), streaming new lines as they land and exiting at
320
+ terminal status. A pre-0925 run with only a legacy `<RUNID>.log` is followed in place (read-only
321
+ fallback). It requires `--follow` and a `run-id`, is a human stream (rejects `--json`), and is
322
+ a **distinct source** — it never interleaves with the DB timeline. If no record file appears (e.g. the
316
323
  run was started with `--no-log`), a clear message is printed at terminal status rather than hanging.
317
- No `spur workflow monitor` verb exists; `--output` is the log-streaming surface.
324
+ No `spur workflow monitor` verb exists; `--output` is the record-streaming surface.
318
325
 
319
326
  HITL pause/resume: a run that hits a HITL action pauses; resume with `spur workflow continue [run-id]`
320
327
  (`--yes` skips confirmation). A headless `hitl.confirm` persists a default `no` before pausing -
321
- use `--answer yes|no|cancel` to inject the operator's gate answer before guard re-evaluation (0433).
322
- `--answer` is distinct from `--yes`: `--yes` skips the CLI resume prompt, `--answer` sets the HITL
323
- gate answer. 0901: resumes accept interrupted runs too (rerun-enter re-executes the interrupted
328
+ use `--answer yes|no|cancel` to inject the operator's gate answer before guard re-evaluation (0433),
329
+ or `--answer-text <text>` to answer an input gate with free text (H1 R27). `--answer` is distinct from
330
+ `--yes`: `--yes` skips the CLI resume prompt, the answer flags set the HITL gate answer. Answer flags
331
+ are validated against the pending gate kind (`--answer` requires a confirm/select gate,
332
+ `--answer-text` an input gate; both together are rejected; a mismatch exits 2 with the run still
333
+ paused), and `--answer`/`--answer-text` honor the gate's `options.var` instead of assuming the
334
+ default var. 0901: resumes accept interrupted runs too (rerun-enter re-executes the interrupted
324
335
  node and needs `resumeRerun: true` on the target state); a headless (`--json`/non-TTY) continue
325
- without `--answer` is refused (exit 2); `--async` detaches the resume and reports started/failed
336
+ without `--answer` or `--answer-text` is refused (exit 2); `--async` detaches the resume and reports started/failed
326
337
  after the worker claims the run; resumed runs write the consolidated run log and pass shell
327
338
  streams through the secret redactor + 64 KiB tail unless `--no-log`. Cancel one live/paused run
328
339
  with `cancel <run-id>`; bulk-finalize orphans stuck in
329
340
  `running`/`pending` with `clean` (`--older-than` default 30 minutes, or `--force`).
330
341
 
342
+ Replay posture (0916): no canonical workflow opts into rerun-enter today — every state relies on
343
+ the engine's refusal default, so an interrupted resume into an unmarked state is refused and the
344
+ next safe action is the printed status (resume `paused` runs directly; recover crashed `running`
345
+ runs via `clean`, which sweeps them to `interrupted`). Enabling `resumeRerun: true` on a canonical
346
+ state requires demonstrated repeatability evidence and a conscious update to the replay matrix in
347
+ `packages/app/tests/workflow/replay-matrix.test.ts`; mutating steps behind `pause: true` gates are
348
+ safe by construction because paused resumes skip-enter.
349
+
331
350
  **Schema resolution parity (0431):** `validate` and `run` both load the workflow through
332
351
  `WorkflowAppService` with the same `embeddedSchemaOptions()` map the CLI injects for
333
352
  `@gobing-ai/spur/schemas/...` refs. `run` pre-loads then calls the engine with the loaded def
@@ -344,7 +363,8 @@ redirecting `agent.run` stages (ADR-047).
344
363
  - **Run-log reclamation** (0429): removes retained `.spur/run/<RUNID>.log` files whose mtime is older
345
364
  than `workflow.logRetentionDays` in `.spur/config.yaml` (default 30 days; integer days, not minutes).
346
365
  Age is the only gate; best-effort deletes never abort the rest. Never touches
347
- `.spur/workflow/<RUNID>.jsonl` or `*-partial.md`.
366
+ `.spur/workflow/<RUNID>.jsonl` or `*-partial.md`. Scope stays legacy `.log` names only (0925 R4):
367
+ the two-file record (`.md` + `.state.json`) is not reclaimed until a pair retention policy exists.
348
368
  - **`--logs`** scopes to log reclamation only (skips stale-run finalization). `--dry-run` applies to
349
369
  both scopes (lists what would be removed, writes nothing). `--json` returns
350
370
  `{ olderThanMinutes, dryRun, cleaned, logs: { retentionDays, dryRun, reclaimed, failures } }` (with