@gobing-ai/spur 0.3.90 → 0.3.92

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (130) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/config/pipeline-budgets.json +9 -2
  3. package/config/plugin-scripts.json +17 -1
  4. package/config/workflows/decision-routing-example.yaml +134 -0
  5. package/config/workflows/feature-lifecycle.yaml +14 -5
  6. package/config/workflows/feature-verification.yaml +35 -27
  7. package/config/workflows/history-anatomy.yaml +14 -25
  8. package/config/workflows/idea-pipeline.yaml +17 -5
  9. package/config/workflows/task-pipeline.yaml +187 -4
  10. package/config/workflows/wayfinder-resolution.yaml +2 -0
  11. package/config/workflows/wrapup-pipeline.yaml +49 -5
  12. package/package.json +9 -9
  13. package/plugins/sp/README.md +1 -0
  14. package/plugins/sp/agents/expert-spur.md +2 -2
  15. package/plugins/sp/agents/super-planner.md +14 -5
  16. package/plugins/sp/commands/dev-dogfood.md +4 -4
  17. package/plugins/sp/commands/dev-fixall.md +8 -5
  18. package/plugins/sp/commands/dev-run.md +6 -0
  19. package/plugins/sp/commands/dev-runall.md +12 -6
  20. package/plugins/sp/commands/dev-verify.md +9 -0
  21. package/plugins/sp/commands/dev-verifyall.md +5 -0
  22. package/plugins/sp/lib/idea-handoff.generated.mjs +301 -300
  23. package/plugins/sp/lib/inline-run.generated.d.mts +17 -0
  24. package/plugins/sp/lib/inline-run.generated.mjs +1460 -0
  25. package/plugins/sp/plugin.json +1 -1
  26. package/plugins/sp/references/environment-lens.md +1 -1
  27. package/plugins/sp/scripts/dogfood-testing/validate-report.mjs +136 -0
  28. package/plugins/sp/scripts/dogfood-testing/validate-report.ts +196 -2
  29. package/plugins/sp/scripts/feature-verification-steps.mjs +174 -0
  30. package/plugins/sp/scripts/feature-verification-steps.ts +275 -0
  31. package/plugins/sp/scripts/history-anatomy-cache.mjs +104 -4
  32. package/plugins/sp/scripts/history-anatomy-cache.ts +137 -13
  33. package/plugins/sp/scripts/inline-pipeline-parity-check.ts +2 -0
  34. package/plugins/sp/scripts/inline-run-setup.mjs +349 -0
  35. package/plugins/sp/scripts/inline-run-setup.ts +192 -75
  36. package/plugins/sp/scripts/record-feature-sync.mjs +63 -0
  37. package/plugins/sp/scripts/record-feature-sync.ts +84 -0
  38. package/plugins/sp/scripts/residual-scan.mjs +476 -0
  39. package/plugins/sp/scripts/residual-scan.ts +614 -0
  40. package/plugins/sp/scripts/surface-drift-inventory.ts +71 -6
  41. package/plugins/sp/scripts/task-evidence-precheck.ts +8 -3
  42. package/plugins/sp/scripts/task-size-precheck.ts +8 -3
  43. package/plugins/sp/scripts/validate-flag-contracts.ts +3 -3
  44. package/plugins/sp/skills/branch-workflow/SKILL.md +1 -0
  45. package/plugins/sp/skills/branch-workflow/references/worktree-patterns.md +2 -0
  46. package/plugins/sp/skills/code-implementation/SKILL.md +17 -0
  47. package/plugins/sp/skills/code-verification/SKILL.md +21 -0
  48. package/plugins/sp/skills/code-verification/references/verdict-schema.md +1 -0
  49. package/plugins/sp/skills/dogfood-testing/SKILL.md +5 -3
  50. package/plugins/sp/skills/dogfood-testing/references/monitor-ledger.md +63 -26
  51. package/plugins/sp/skills/dogfood-testing/references/report-template.md +33 -10
  52. package/plugins/sp/skills/history-anatomy/references/modes.md +5 -3
  53. package/plugins/sp/skills/next-feature/references/ranking-rubric.md +1 -1
  54. package/plugins/sp/skills/next-router/references/routing-table.md +7 -0
  55. package/plugins/sp/skills/parallel-execution/references/dispatch-surface.md +1 -1
  56. package/plugins/sp/skills/session-review/SKILL.md +12 -2
  57. package/plugins/sp/skills/spur-cli/references/agent.md +11 -2
  58. package/plugins/sp/skills/spur-cli/references/features.md +1 -1
  59. package/plugins/sp/skills/spur-cli/references/projects.md +3 -1
  60. package/plugins/sp/skills/spur-cli/references/self.md +5 -1
  61. package/plugins/sp/skills/spur-cli/references/workflows.md +42 -20
  62. package/plugins/sp/skills/spur-dev/SKILL.md +11 -4
  63. package/plugins/sp/skills/spur-dev/references/cross-cutting.md +25 -3
  64. package/plugins/sp/skills/spur-dev/references/dev-operations.md +2 -2
  65. package/plugins/sp/skills/spur-dev/references/execution-batch.md +285 -57
  66. package/plugins/sp/skills/spur-dev/references/execution-workflow.md +14 -5
  67. package/plugins/sp/skills/spur-dev/references/flag-glossary.md +19 -4
  68. package/plugins/sp/skills/spur-dev/references/glossary.md +9 -1
  69. package/plugins/sp/skills/spur-dev/references/inline-pipeline-driver.md +51 -16
  70. package/plugins/sp/skills/spur-dev/references/planning-workflow.md +5 -0
  71. package/plugins/sp/skills/wayfinder/SKILL.md +1 -1
  72. package/spur.js +23422 -19364
  73. package/web/_astro/BoardApp.CerSBgis.js +192 -0
  74. package/web/_astro/BoardApp.eoTz0pZs.js +1 -0
  75. package/web/_astro/{TaskDetail.DwTmbQp5.js → TaskDetail.DCqiC-OZ.js} +1 -1
  76. package/web/_astro/arc.CPwg6Rw0.js +1 -0
  77. package/web/_astro/{architectureDiagram-3BPJPVTR.jvdDahWM.js → architectureDiagram-3BPJPVTR.DM_vp_hO.js} +1 -1
  78. package/web/_astro/{blockDiagram-GPEHLZMM.zSg4AmFD.js → blockDiagram-GPEHLZMM.DXVIiv0p.js} +1 -1
  79. package/web/_astro/{c4Diagram-AAUBKEIU.BkUIUQWH.js → c4Diagram-AAUBKEIU.BbF_zCxW.js} +1 -1
  80. package/web/_astro/channel.MYZLKNwy.js +1 -0
  81. package/web/_astro/{chunk-2J33WTMH.DvfQ_f50.js → chunk-2J33WTMH.CAgQHpPC.js} +1 -1
  82. package/web/_astro/{chunk-4BX2VUAB.DuI4gQqX.js → chunk-4BX2VUAB.BN-5tpw4.js} +1 -1
  83. package/web/_astro/{chunk-55IACEB6.D3BWBOpF.js → chunk-55IACEB6.CnPkEEr0.js} +1 -1
  84. package/web/_astro/{chunk-727SXJPM.3QSi0a9M.js → chunk-727SXJPM.BQzQeMVm.js} +4 -4
  85. package/web/_astro/{chunk-AQP2D5EJ.xazCQrAF.js → chunk-AQP2D5EJ.B6xNyDnL.js} +1 -1
  86. package/web/_astro/{chunk-FMBD7UC4.B2g6u4rA.js → chunk-FMBD7UC4.C7f9Ih78.js} +1 -1
  87. package/web/_astro/{chunk-ND2GUHAM.wWwWs99t.js → chunk-ND2GUHAM.CNV1dFXT.js} +1 -1
  88. package/web/_astro/{chunk-QZHKN3VN.BD5g3qa9.js → chunk-QZHKN3VN.Cudn2TkJ.js} +1 -1
  89. package/web/_astro/{classDiagram-4FO5ZUOK.C7CzCdsX.js → classDiagram-4FO5ZUOK.D1NwP50q.js} +1 -1
  90. package/web/_astro/{classDiagram-v2-Q7XG4LA2.C7CzCdsX.js → classDiagram-v2-Q7XG4LA2.D1NwP50q.js} +1 -1
  91. package/web/_astro/{cose-bilkent-S5V4N54A.Xyiau0gw.js → cose-bilkent-S5V4N54A.B1wSL-Xb.js} +1 -1
  92. package/web/_astro/{cynefin-OW5HDTMX.BeC5MWas.js → cynefin-OW5HDTMX.BmK52w8G.js} +1 -1
  93. package/web/_astro/{dagre-BM42HDAG.yZbMN9vc.js → dagre-BM42HDAG.Bfy5CTDT.js} +2 -2
  94. package/web/_astro/diagram-2AECGRRQ.DhNnvUvX.js +43 -0
  95. package/web/_astro/diagram-5GNKFQAL.lTX5KwnS.js +10 -0
  96. package/web/_astro/{diagram-KO2AKTUF.CR6k3Y3G.js → diagram-KO2AKTUF.CW_vMJ4z.js} +3 -3
  97. package/web/_astro/{diagram-LMA3HP47.x7mwu8jz.js → diagram-LMA3HP47.B_8ZGF67.js} +1 -1
  98. package/web/_astro/{diagram-OG6HWLK6.D8aTTvUr.js → diagram-OG6HWLK6.BppnHsdS.js} +1 -1
  99. package/web/_astro/{erDiagram-TEJ5UH35.BoBqcKXQ.js → erDiagram-TEJ5UH35.BEuHXcjJ.js} +5 -5
  100. package/web/_astro/{flowDiagram-I6XJVG4X.D3mTQdrU.js → flowDiagram-I6XJVG4X.CH-UlnGr.js} +4 -4
  101. package/web/_astro/{ganttDiagram-6RSMTGT7.H-cqgIh-.js → ganttDiagram-6RSMTGT7.BO81S85v.js} +1 -1
  102. package/web/_astro/{gitGraphDiagram-PVQCEYII.B6s9zbfC.js → gitGraphDiagram-PVQCEYII.XnPxPPZN.js} +1 -1
  103. package/web/_astro/index.Hjbr15fG.css +1 -0
  104. package/web/_astro/{infoDiagram-5YYISTIA.BzgCoV6P.js → infoDiagram-5YYISTIA.JyjYRu_T.js} +1 -1
  105. package/web/_astro/{ishikawaDiagram-YF4QCWOH.BZzVhy1-.js → ishikawaDiagram-YF4QCWOH.BBRBF-Fo.js} +5 -5
  106. package/web/_astro/{journeyDiagram-JHISSGLW.BV3195Py.js → journeyDiagram-JHISSGLW.C_iymSyp.js} +1 -1
  107. package/web/_astro/{kanban-definition-UN3LZRKU.BjRd2DWz.js → kanban-definition-UN3LZRKU.DdfW-Oqt.js} +7 -7
  108. package/web/_astro/{linear.BILTgS5N.js → linear.C2_IkbZT.js} +1 -1
  109. package/web/_astro/mermaid.core.GAOYeSR0.js +303 -0
  110. package/web/_astro/{mindmap-definition-RKZ34NQL.BiEjaI4-.js → mindmap-definition-RKZ34NQL.DAZIxQSK.js} +2 -2
  111. package/web/_astro/{pieDiagram-4H26LBE5.i_8V5pIn.js → pieDiagram-4H26LBE5.CN8sIhKM.js} +3 -3
  112. package/web/_astro/{quadrantDiagram-W4KKPZXB.BWaW3MHn.js → quadrantDiagram-W4KKPZXB.3dGcX5GP.js} +1 -1
  113. package/web/_astro/{requirementDiagram-4Y6WPE33.CzddBbtg.js → requirementDiagram-4Y6WPE33.BV2y4dd6.js} +3 -3
  114. package/web/_astro/{sankeyDiagram-5OEKKPKP.X2ww0e-D.js → sankeyDiagram-5OEKKPKP.Cqo15Tvo.js} +4 -4
  115. package/web/_astro/{sequenceDiagram-3UESZ5HK.DSA4kTcc.js → sequenceDiagram-3UESZ5HK.CROCPMJB.js} +1 -1
  116. package/web/_astro/{stateDiagram-AJRCARHV.D0DtFSpR.js → stateDiagram-AJRCARHV.RfXZrkFE.js} +1 -1
  117. package/web/_astro/{stateDiagram-v2-BHNVJYJU.BfQq0zQv.js → stateDiagram-v2-BHNVJYJU.CPXmbBs9.js} +1 -1
  118. package/web/_astro/{timeline-definition-PNZ67QCA.Dmlrgi1m.js → timeline-definition-PNZ67QCA.DdgKTiO8.js} +3 -3
  119. package/web/_astro/{vennDiagram-CIIHVFJN.D5mpl00Z.js → vennDiagram-CIIHVFJN.CPNVSHF1.js} +5 -5
  120. package/web/_astro/{wardleyDiagram-YWT4CUSO.Df4BdzO4.js → wardleyDiagram-YWT4CUSO.CQhA0Jyr.js} +3 -3
  121. package/web/_astro/{xychartDiagram-2RQKCTM6.DiTRreKN.js → xychartDiagram-2RQKCTM6.n61BWyy4.js} +1 -1
  122. package/web/index.html +2 -2
  123. package/web/_astro/BoardApp.CDUcHlTJ.js +0 -188
  124. package/web/_astro/BoardApp.CaCGU_uX.js +0 -1
  125. package/web/_astro/arc.BzF71EFI.js +0 -1
  126. package/web/_astro/channel.SSVY0JPQ.js +0 -1
  127. package/web/_astro/diagram-2AECGRRQ.Cmo2zQM-.js +0 -43
  128. package/web/_astro/diagram-5GNKFQAL.D033eSVi.js +0 -10
  129. package/web/_astro/index.CcU5weKX.css +0 -1
  130. package/web/_astro/mermaid.core.DBy_WKeW.js +0 -301
@@ -231,8 +231,17 @@ to executors via `agent.executors[].agent` (or the model's `<provider>/` prefix)
231
231
  write path. A provider is exhausted when any `primary|secondary|tertiary` window reports
232
232
  `usedPercent >= 100`; the window name and `resetsAt` go into the observation reason. Per-provider
233
233
  `{ "error": … }` entries are skipped (listed, never treated as recovery); healthy entries still
234
- apply. A missing codexbar binary or an unparsable payload exits `1` and changes nothing.
235
- Unmapped providers are listed and never guessed.
234
+ apply. Providers whose windows are all null carry no signal (`no-usage`): they are reported and
235
+ excluded from availability decisions — an absent signal neither disables nor recovers, and an
236
+ exhausted signal wins on shared executors. Operator-owned availability (`disabled: true` or an
237
+ operator ownership object) is never touched. A missing codexbar binary or an unparsable payload
238
+ exits `1` and changes nothing. Unmapped providers are listed and never guessed.
239
+
240
+ Each reported change carries a delivery-semantics `action` (0907): `would-apply` (dry run),
241
+ `applied` — the exact observation created by the invocation was acknowledged without a skip
242
+ (desired state confirmed; not proof of a YAML byte change), `no-op` — already satisfied or
243
+ operator-owned, `skipped` — the observation was superseded or rejected, `pending` — delivery
244
+ unconfirmed or failed. For `skipped`/`pending` the printed target is intent only.
236
245
 
237
246
  **Scheduling is external** (cron/launchd, same pattern as `spur history daily`):
238
247
 
@@ -28,7 +28,7 @@ what* or *how to write a scenario*, this skill.
28
28
  | `list` | List features, filtered | `--status <s>` `--priority <p>` `--folder` `--json` |
29
29
  | `move <id>` | Re-parent a subtree (cascade-rename of descendants) | `--parent <id>` `--dry-run` `--folder` `--json` |
30
30
  | `refresh` | Rebuild INDEX + each feature `## Tasks` table from task edges (**docs only**; no status change) | `--feature <id>` `--all` `--folder` `--json` |
31
- | `check [id]` | Validate one feature / the tree; the 4-layer gate; `--fix` repairs structural findings in place | `--strict` `--fix` `--folder` `--json` |
31
+ | `check [id]` | Validate one feature / the tree; the 4-layer gate; `--fix` repairs structural findings in place | `--strict` `--fix` `--as <status>` `--folder` `--json` |
32
32
  | `sync [id]` | Align feature **lifecycle status** with linked task states (real transitions + guards) | `--all` `--dry-run` `--force` `--folder` `--json` |
33
33
 
34
34
  **`refresh` vs `sync` (do not conflate):**
@@ -19,6 +19,7 @@ shapes live in `apps/cli/src/commands/projects.ts`.
19
19
  | ---- | ------- | --------- |
20
20
  | `add <path>` | Upsert an existing path in the registry | `--name <name>` `--json` |
21
21
  | `remove <target>` | Remove an entry by display name or path | `--json` |
22
+ | `clean` | Purge missing project folders and terminate lingering processes (alias: `refresh`) | `--no-terminate-processes` `--json` |
22
23
  | `list` | List entries with live running status | `--json` `--fleet` |
23
24
  | `start <target>` | Start or reuse a detached project server | `--port <n>` `--json` |
24
25
  | `stop <target>` | Best-effort stop the listener and clear its recorded port | `--json` |
@@ -35,7 +36,8 @@ exit `0`; validation, registry, spawn, health, or lookup failure is exit `1`.
35
36
  - `add` requires an existing path, resolves a relative path from the current working directory, and
36
37
  defaults the display name to its basename. It upserts; it does not start a server. The current
37
38
  source does not enforce a `.spur/` marker or directory type.
38
- - `list` probes recorded ports and heals stale entries to `port: 0` before reporting `running`.
39
+ - `list` automatically heals tilde paths, verifies project directories exist on disk (purging missing entries and terminating lingering processes), and heals stale entries to `port: 0` before reporting `running`.
40
+ - `clean` (alias `refresh`) explicitly purges non-existent project directories from `projects.json`, probes live ports for removed entries, and cleanly terminates orphaned listening processes (SIGTERM with bounded wait and SIGKILL escalation). Supports `--no-terminate-processes` to skip process kills.
39
41
  - `list --fleet` (0835/0858) additionally resolves each project's `agent.fleet` section from that
40
42
  project's `.spur/config.yaml` under the existing verb (no new noun). Per project it prints one line
41
43
  per member: instance id (the spec id / mailbox identity), `role`, resolved `executor`,
@@ -90,7 +90,11 @@ spur self status --json # machine-readable
90
90
  ```
91
91
 
92
92
  Reports the project's Spur configuration state (init status, feature/task counts, rule preset
93
- health) and git working-tree status. Optional `[path]` argument targets a different project
93
+ health), git working-tree status, and DecisionMaker readiness (0911): the human output adds a
94
+ `DecisionMaker:` line and `--json` exposes `decisionMaker: {enabled, provider,
95
+ credentialPresent, state, connectivity, inlineSupport}` with states `disabled` / `missing-key` /
96
+ `configured-not-probed`. Presence check only — never a live probe, never echoes the key.
97
+ Optional `[path]` argument targets a different project
94
98
  directory. Only flag is `--json`.
95
99
 
96
100
  ## What this skill is NOT
@@ -80,7 +80,9 @@ Use this skill to:
80
80
  - **Author a workflow** — turn a described process into a validated, dry-run-verified YAML definition
81
81
  in the right mode. → authoring-workflows.md
82
82
  - **Validate before trusting** — schema + semantic-check a workflow file (references, terminal
83
- reachability, template vars) before running it.
83
+ reachability, template vars, and since 0911 the HITL decision policy: `decision` only on
84
+ `hitl.confirm`/`hitl.select`, evidence mode banned in `pause: true` states/nodes, one evidence
85
+ action per state/node, existing producer nodes, valid select choices) before running it.
84
86
  - **Run a workflow** — execute a definition and read its run trace (states/nodes entered, transitions
85
87
  taken, terminal status).
86
88
  - **Refine an existing workflow** — fix a stuck guard, add a state/node, retune `iterationBound`,
@@ -108,11 +110,11 @@ The skill's logic divides by **whether the LLM adds value**:
108
110
  | --------- | --------- | ----- | ------------------ |
109
111
  | `validate` | `spur workflow validate` (CLI) | `<file> [--no-schema]` | Schema + semantic verdict |
110
112
  | `run` | `spur workflow run` (CLI) | `<file> [--run-id <id>] [--vars <json>] [--dry-run] [--async] [--no-plan] [--quiet/--silent/--verbose] [--detail <level>] [--trace-file] [--no-log] [--steer]` | Terminal state reached (sync) or run started (async); trace readable |
111
- | `continue` | `spur workflow continue` (CLI) | `[run-id] [--yes] [--answer <yes\|no\|cancel>] [--async] [--no-log]` | Resume a paused or interrupted run (omit id -> most recent resumable); `--answer` injects a gate answer before guard re-evaluation and is required headless (0901 R3); `--async` detaches the resume (0901 R4) |
113
+ | `continue` | `spur workflow continue` (CLI) | `[run-id] [--yes] [--answer <yes\|no\|cancel>] [--answer-text <text>] [--async] [--no-log]` | Resume a paused or interrupted run (omit id -> most recent resumable); `--answer` injects a confirm/select gate answer and `--answer-text` answers an input gate with free text (H1 R27), both before guard re-evaluation; one of them is required headless (0901 R3) and each is validated against the pending gate kind; `--async` detaches the resume (0901 R4) |
112
114
  | `cancel` | `spur workflow cancel` (CLI) | `<run-id>` | Single non-terminal run marked failed (SIGTERM async worker when live) |
113
115
  | `clean` | `spur workflow clean` (CLI) | `[--older-than <min>] [--force] [--logs] [--dry-run]` | Bulk-finalize stale `running`/`pending` runs as failed **and** reclaim retained run logs older than `workflow.logRetentionDays` (30d default) |
114
116
  | `list` | `spur workflow list` (CLI) | — | Available workflow **YAML definition files** (not run records) |
115
- | `trace` | `spur workflow trace` (CLI) | `[run-id] [--workflow <n>] [--status <s>] [--since <iso>] [--last <n>] [--follow] [--poll <ms>] [--output]` | Run history list or per-run timeline |
117
+ | `trace` | `spur workflow trace` (CLI) | `[run-id] [--workflow <n>] [--status <s>] [--since <iso>] [--last <n>] [--follow] [--poll <ms>] [--output] [--timeout <ms>]` | Run history list or per-run timeline |
116
118
  | `progress` | `spur workflow progress` (CLI) | `<run-id>` | The `projectWorkflowProgress` projection for that run — current state, per-action attempts, candidate next transitions, diagnostics. Read-only; the verb renders, `packages/app` derives. Unknown run id exits 1 with `Run <id> not found.` |
117
119
  | `add` | agent procedure | `"<nl-description>" [--kind <state-machine\|transition-flow>] [--file <path>]` | **Mode chosen (confirmed)** → first reconciled against existing workflows (extend an existing flow rather than duplicate) → YAML authored in real schema shape → **validated AND dry-run** (reaches the expected terminal state) → [add](workflows/operations.md#add) |
118
120
  | `refine` | agent procedure | `<workflow-file> [--intent "<goal>"] [--dry-run]` | Smallest change meeting the intent, re-validated and re-dry-run; `--dry-run` emits a diff only → [refine](workflows/operations.md#refine) |
@@ -202,7 +204,7 @@ spur workflow run ./workflows/approval.yaml --silent # errors only
202
204
  spur workflow run ./workflows/approval.yaml --verbose # transitions + correlation diagnostics
203
205
  spur workflow run ./workflows/approval.yaml --detail minimal # tersest human output
204
206
  spur workflow run ./workflows/approval.yaml --trace-file # persist redacted JSONL trace
205
- spur workflow run ./workflows/approval.yaml --no-log # opt out of the consolidated .spur/run/<RUNID>.log
207
+ spur workflow run ./workflows/approval.yaml --no-log # opt out of the run record .spur/run/<RUNID>.md + .state.json
206
208
  spur workflow run ./workflows/approval.yaml --steer # interactive steering on stdin
207
209
  ```
208
210
 
@@ -212,8 +214,8 @@ spur workflow run ./workflows/approval.yaml --steer # interactive
212
214
  per-step headers), `full` (transitions + correlation). `--verbose` is shorthand for `--detail full`.
213
215
  - **`--trace-file`** appends a redacted, schema-versioned JSONL trace under `.spur/workflow/`
214
216
  for post-run analysis - independent of human/JSON output.
215
- - **`--no-log`** opts out of writing the consolidated all-in-one run log (`.spur/run/<RUNID>.log`).
216
- By default the log is written **and retained** after the run ends; this flag skips it entirely
217
+ - **`--no-log`** opts out of writing the two-file run record (`.spur/run/<RUNID>.md` + `.state.json`).
218
+ By default the record is written **and retained** after the run ends; this flag skips it entirely
217
219
  (propagates to the `--async` detached worker). No `--keep-log` / delete-by-default exists.
218
220
  - **`--steer`** is synchronous and in-process: it cannot combine with `--json` or `--async` (exit `2`).
219
221
  It accepts steering commands on stdin at declared action boundaries for interactive control.
@@ -258,11 +260,11 @@ the operator accepts it, and never hot-edit a running workflow's shell in place.
258
260
  spur workflow validate <file> [--no-schema] [--json]
259
261
  spur workflow show <file> [--format <mermaid|todo>] [--json]
260
262
  spur workflow run <file> [--run-id <id>] [--vars <json>] [--dry-run] [--async] [--no-plan] [--quiet/--silent/--verbose] [--detail <level>] [--trace-file] [--no-log] [--steer] [--json]
261
- spur workflow continue [run-id] [--yes] [--answer <yes|no|cancel>] [--async] [--no-log] [--json]
263
+ spur workflow continue [run-id] [--yes] [--answer <yes|no|cancel>] [--answer-text <text>] [--async] [--no-log] [--json]
262
264
  spur workflow cancel <run-id> [--json]
263
265
  spur workflow clean [--older-than <minutes>] [--force] [--logs] [--dry-run] [--json]
264
266
  spur workflow list [--json]
265
- spur workflow trace [run-id] [--workflow <name>] [--status <s>] [--since <iso>] [--last <n>] [--follow] [--poll <ms>] [--output] [--json]
267
+ spur workflow trace [run-id] [--workflow <name>] [--status <s>] [--since <iso>] [--last <n>] [--follow] [--poll <ms>] [--output] [--timeout <ms>] [--json]
266
268
  spur workflow progress <run-id> [--json]
267
269
  ```
268
270
 
@@ -283,7 +285,7 @@ not advertise `--json-envelope` because its JSON projection is a kept-raw docume
283
285
  | `--verbose` | Include transitions and correlation diagnostics in human progress (implies `--detail full`). |
284
286
  | `--detail <level>` | Human detail level: `minimal`, `invocation` (default), or `full`. |
285
287
  | `--trace-file` | Append a redacted schema-versioned JSONL trace under `.spur/workflow/`. |
286
- | `--no-log` | Opt out of writing the consolidated `.spur/run/<RUNID>.log` (retained by default; propagates to `--async` workers). |
288
+ | `--no-log` | Opt out of writing the two-file run record `.spur/run/<RUNID>.md` + `.state.json` (retained by default; propagates to `--async` workers). |
287
289
  | `--steer` | Accept in-process steering commands on stdin at declared action boundaries (sync only; incompatible with `--json`/`--async`). |
288
290
 
289
291
  `validate` and `run` exit non-zero on failure (`run` exits non-zero when the final status is not
@@ -300,32 +302,51 @@ Follow a live run to terminal (human streaming mode):
300
302
  ```bash
301
303
  spur workflow trace <run-id> --follow # stream until terminal (default 1000ms poll)
302
304
  spur workflow trace <run-id> --follow --poll 500 # poll every 500ms
303
- spur workflow trace <run-id> --follow --output # stream .spur/run/<RUNID>.log instead of the DB timeline
305
+ spur workflow trace <run-id> --follow --output # stream .spur/run/<RUNID>.md instead of the DB timeline
306
+ spur workflow trace <run-id> --follow --timeout 600000 # bound the watch; timeout → one checkpoint, run continues, exit 1
304
307
  ```
305
308
 
306
309
  - **`--follow`** replays a run timeline and polls persisted state until it becomes terminal. It
307
310
  requires a `run-id` (exit `2` without one) and cannot combine with `--json` (exit `2` - it is a
308
311
  human streaming mode).
309
312
  - **`--poll <ms>`** sets the follow polling interval (default `1000`, minimum `50`; exit `2` otherwise).
310
- - **`--output`** swaps the follow source from the structured DB timeline to the consolidated all-in-one
311
- log (`.spur/run/<RUNID>.log`, tail -f equivalent), streaming new lines as they land and exiting at
312
- terminal status. It requires `--follow` and a `run-id`, is a human stream (rejects `--json`), and is
313
- a **distinct source** — it never interleaves with the DB timeline. If the log never appears (e.g. the
313
+ - **`--timeout <ms>`** (requires `--follow`, positive integer; otherwise exit `1` with `VALIDATION_FAILED`)
314
+ bounds the watch (0930): when the deadline passes before the run is terminal — including the `Run not
315
+ found` registration retry window — the follow prints one checkpoint naming the run id and last observed
316
+ status and exits `1`. A watch timeout never cancels or relaunches the run; resume by re-running the same
317
+ follow command.
318
+ - **`--output`** swaps the follow source from the structured DB timeline to the human run record
319
+ (`.spur/run/<RUNID>.md`, tail -f equivalent), streaming new lines as they land and exiting at
320
+ terminal status. A pre-0925 run with only a legacy `<RUNID>.log` is followed in place (read-only
321
+ fallback). It requires `--follow` and a `run-id`, is a human stream (rejects `--json`), and is
322
+ a **distinct source** — it never interleaves with the DB timeline. If no record file appears (e.g. the
314
323
  run was started with `--no-log`), a clear message is printed at terminal status rather than hanging.
315
- No `spur workflow monitor` verb exists; `--output` is the log-streaming surface.
324
+ No `spur workflow monitor` verb exists; `--output` is the record-streaming surface.
316
325
 
317
326
  HITL pause/resume: a run that hits a HITL action pauses; resume with `spur workflow continue [run-id]`
318
327
  (`--yes` skips confirmation). A headless `hitl.confirm` persists a default `no` before pausing -
319
- use `--answer yes|no|cancel` to inject the operator's gate answer before guard re-evaluation (0433).
320
- `--answer` is distinct from `--yes`: `--yes` skips the CLI resume prompt, `--answer` sets the HITL
321
- gate answer. 0901: resumes accept interrupted runs too (rerun-enter re-executes the interrupted
328
+ use `--answer yes|no|cancel` to inject the operator's gate answer before guard re-evaluation (0433),
329
+ or `--answer-text <text>` to answer an input gate with free text (H1 R27). `--answer` is distinct from
330
+ `--yes`: `--yes` skips the CLI resume prompt, the answer flags set the HITL gate answer. Answer flags
331
+ are validated against the pending gate kind (`--answer` requires a confirm/select gate,
332
+ `--answer-text` an input gate; both together are rejected; a mismatch exits 2 with the run still
333
+ paused), and `--answer`/`--answer-text` honor the gate's `options.var` instead of assuming the
334
+ default var. 0901: resumes accept interrupted runs too (rerun-enter re-executes the interrupted
322
335
  node and needs `resumeRerun: true` on the target state); a headless (`--json`/non-TTY) continue
323
- without `--answer` is refused (exit 2); `--async` detaches the resume and reports started/failed
336
+ without `--answer` or `--answer-text` is refused (exit 2); `--async` detaches the resume and reports started/failed
324
337
  after the worker claims the run; resumed runs write the consolidated run log and pass shell
325
338
  streams through the secret redactor + 64 KiB tail unless `--no-log`. Cancel one live/paused run
326
339
  with `cancel <run-id>`; bulk-finalize orphans stuck in
327
340
  `running`/`pending` with `clean` (`--older-than` default 30 minutes, or `--force`).
328
341
 
342
+ Replay posture (0916): no canonical workflow opts into rerun-enter today — every state relies on
343
+ the engine's refusal default, so an interrupted resume into an unmarked state is refused and the
344
+ next safe action is the printed status (resume `paused` runs directly; recover crashed `running`
345
+ runs via `clean`, which sweeps them to `interrupted`). Enabling `resumeRerun: true` on a canonical
346
+ state requires demonstrated repeatability evidence and a conscious update to the replay matrix in
347
+ `packages/app/tests/workflow/replay-matrix.test.ts`; mutating steps behind `pause: true` gates are
348
+ safe by construction because paused resumes skip-enter.
349
+
329
350
  **Schema resolution parity (0431):** `validate` and `run` both load the workflow through
330
351
  `WorkflowAppService` with the same `embeddedSchemaOptions()` map the CLI injects for
331
352
  `@gobing-ai/spur/schemas/...` refs. `run` pre-loads then calls the engine with the loaded def
@@ -342,7 +363,8 @@ redirecting `agent.run` stages (ADR-047).
342
363
  - **Run-log reclamation** (0429): removes retained `.spur/run/<RUNID>.log` files whose mtime is older
343
364
  than `workflow.logRetentionDays` in `.spur/config.yaml` (default 30 days; integer days, not minutes).
344
365
  Age is the only gate; best-effort deletes never abort the rest. Never touches
345
- `.spur/workflow/<RUNID>.jsonl` or `*-partial.md`.
366
+ `.spur/workflow/<RUNID>.jsonl` or `*-partial.md`. Scope stays legacy `.log` names only (0925 R4):
367
+ the two-file record (`.md` + `.state.json`) is not reclaimed until a pair retention policy exists.
346
368
  - **`--logs`** scopes to log reclamation only (skips stale-run finalization). `--dry-run` applies to
347
369
  both scopes (lists what would be removed, writes nothing). `--json` returns
348
370
  `{ olderThanMinutes, dryRun, cleaned, logs: { retentionDays, dryRun, reclaimed, failures } }` (with
@@ -92,9 +92,15 @@ pick task (spur task list --json)
92
92
  The pipeline (`kind: state-machine`) runs the work loop:
93
93
 
94
94
  ```
95
- precheck → implement → test → review → approve(HITL) → verify → record → done
95
+ precheck → implement[→escalate(HITL)] → test → review → approve(HITL) → verify → record → done
96
96
  ```
97
97
 
98
+ **Escalation contract (0933).** The implement agent may pause on an operator question: it writes
99
+ `.spur/run/<wbs>-question.md` and exits 0; the `escalate` hop surfaces it via HITL and pauses the
100
+ run. Resume with `spur workflow continue --answer-text <answer>` — the guard appends the Q/A to
101
+ `.spur/run/<wbs>-escalation.md` and re-enters implement (transcript via `--escalation-file`).
102
+ `maxEscalations` (default 2) bounds the loop; exhausted or empty answer → `failed`. Implement only.
103
+
98
104
  Full procedure: **[references/execution-workflow.md](references/execution-workflow.md)**.
99
105
  Host-session procedure: **[references/inline-pipeline-driver.md](references/inline-pipeline-driver.md)**.
100
106
 
@@ -162,9 +168,10 @@ CLI does.
162
168
  - Before accepting a child/watcher result, compare its run ID with the dispatched run ID and check
163
169
  current trace state through Spur. If using a run log as evidence, require its mtime to be at least
164
170
  the dispatch time. A mismatch or stale timestamp is not completion evidence. Do not scrape terminals.
165
- Bound each watch invocation to 10 minutes or 20 polls, whichever comes first; persist the last
166
- confirmed identity/state and report a checkpoint before continuing. A watcher timeout does not
167
- cancel the owned run or authorize launching a replacement.
171
+ Bound each watch invocation to `spur workflow trace <run-id> --follow --timeout 600000` — one
172
+ bounded call, no hand-rolled poll loop; a timeout prints one checkpoint line and exits 1 while the
173
+ run continues, so persist the last confirmed identity/state and report the checkpoint before
174
+ continuing. A watcher timeout does not cancel the owned run or authorize launching a replacement.
168
175
  - Mark superseded scratch with `SUPERSEDED` and a pointer to the authoritative task Design. Never
169
176
  let a scratch instruction override the live task, even when its old run is still readable.
170
177
  - Checker-policy changes require one explicit unsuppressed audit (T10); ordinary corpus commit
@@ -61,6 +61,18 @@ The explicit-rejection carve-out above was superseded by ADR-087 (task 0687): a
61
61
  substitutes tier resolution with a warning instead of rejecting — see the substitution blockquote
62
62
  above and `resolveAgent` in `packages/app/src/services/agent-service.ts`.
63
63
 
64
+ ### Run-scoped session policy (B7, summary)
65
+
66
+ When an inline resolution dispatches a native subagent, or a workflow `agent.run` executes, the
67
+ run-scoped session policy applies: coder stages default to `session: reuse`; reviewer, planner,
68
+ and scribe stages default to `fresh`; a reviewer stage may declare `session: reuse` explicitly.
69
+ A runner record without resume capability (`supportsResumeById: false`) forces a fresh dispatch,
70
+ and a disabled pinned executor re-resolves once, then starts fresh. Run traces record the session
71
+ provenance (`reused | fresh`). This is a summary only: the owning contract is
72
+ [session-pinned-dispatch.md §4](../../../../../docs/design/session-pinned-dispatch.md) with role
73
+ semantics in [roles.md](../../../references/roles.md); this section does not restate that
74
+ contract.
75
+
64
76
  ### Objective triggers override the answer
65
77
 
66
78
  The one rule resolves operator *intent*. A trigger is a detected *requirement* the chosen executor
@@ -188,8 +200,15 @@ Do not read provider auth or quota from `spur agent doctor`. The doctor resolves
188
200
  config), so it historically degraded to `status: usable · auth: no · model: unknown` for GLM-style
189
201
  executors and was useless as a preflight gate. Feature B4 removed the auth signal from the surface
190
202
  entirely (no column, no `authenticated` in `--json`) precisely so nothing can read it by mistake;
191
- the precheck probe classifies on usability alone. Exhaustion is detected mid-run by the escalation
192
- classifier, not by any preflight probe.
203
+ the precheck probe classifies on usability alone. Preflight availability does exist — as a
204
+ separate, owned surface (session-pinned-dispatch.md §3.4): `spur agent usage` captures provider
205
+ quota windows into a durable snapshot, the availability drain derives `quota.exhausted` /
206
+ `quota.recovered` observations from it, and `spur agent doctor` renders the provenance (`owner`,
207
+ `since`, `reason`, snapshot `age`). A snapshot older than `agent.usage.maxAgeMs` renders `stale`
208
+ and never enables an executor. Usability (doctor) is not authentication, and neither is a live
209
+ quota guarantee — the signals stay distinct. Mid-run exhaustion is still detected by the
210
+ escalation classifier; the preflight surface informs planning, it does not replace the in-run
211
+ detector.
193
212
 
194
213
  ### Explicit subprocess surfaces are unchanged
195
214
 
@@ -493,7 +512,10 @@ iterating one task).
493
512
 
494
513
  **The rule.** When a test fails and you are iterating to green:
495
514
 
496
- 1. Run the narrow target first: `bun test <file> --test-name-pattern <test>`.
515
+ 1. Run the narrow target first: `bun test <file> --test-name-pattern <test>`. Workspace tests need
516
+ the workspace `bunfig.toml` preload — invoke them as a subshell `(cd <workspace> && bun test …)`
517
+ or with absolute paths, because the shell's cwd persists between calls and a bare `cd` breaks
518
+ later relative-path commands (verifyall session friction, 2026-09-23).
497
519
  2. Loop on that narrow target until green.
498
520
  3. **Then** run the single full `spur-check` (or `bun run check`) as the final gate.
499
521
 
@@ -352,8 +352,8 @@ must not be changed without updating the backing skill.
352
352
  ### 13. runall
353
353
 
354
354
  - **Purpose:** Run a batch of tasks through their pipelines in dependency-correct order — resolve a set, topo-sort, run each via `task-pipeline.yaml`, inspect verdicts, apply the failure policy, emit a batch report.
355
- - **Inputs:** `--tasks <selector>` (required — explicit WBS list, status pseudo-list, `feature:<id>`, or `ready`). `--mode <sequential|parallel>` (default `sequential`; `parallel` fans out a proven-independent subset per `execution-batch.md` § Parallel Execution). `--keep-going` skips a failed task's in-batch dependents and continues independents (default halts on first failure). `--auto` sets `profile=auto` on each per-task run (skips the HITL approve gate). `--feature <id>` is sugar for `--tasks feature:<id>`; when the effective selector is feature-derived, the batch runs `spur feature check <id> --strict --json` **once** before task resolution — a non-zero strict check aborts with verdict `aborted`, zero attempted tasks, and the structured findings (task 0510 R2); scoped: `L4.scenario-unverified` (expected pre-run state of any not-yet-run feature) is reported verbatim but does not abort — any other strict error aborts. Explicit WBS/status/`ready` selectors add no feature check. The orchestrator loop itself continues in this session; interactive sequential omit/`inline` uses the host driver — host-controlled with eligible `agent.run` stages dispatching once to a native subagent and host fallback (task 0508) — while `--agent auto`/a name, parallel mode, and headless invocation keep the isolated per-task workflow subprocess boundary. `--agent <inline|auto|name>` pins the executor for the per-task stages (see [SSOT](cross-cutting.md#inline-default-execution-surface)); the value crosses into per-task `vars.agent`, not the orchestrator. `--json` emits the report as JSON. `--wrap` triggers `wrapup-pipeline.yaml` after the batch completes, `--next` chains each task to terminal status then runs the wrap hop **once for the batch**, `--continue` resumes from checkpoint.
356
- - **Three orthogonal axes (do not confuse):** `--keep-going` = batch failure policy (halt vs skip dependents); `--continue` = resume from checkpoint (pick up an interrupted batch); `--next` = per-task lifecycle chaining (advance status on a verdict — `dev-verify`/`dev-verifyall` only). `routing-table.md` offers `--continue` and `--next` as competing options for the same situation only when the batch was interrupted mid-run; otherwise they address different problems.
355
+ - **Inputs:** `--tasks <selector>` (required — explicit WBS list, status pseudo-list, `feature:<id>`, or `ready`). `--mode <sequential|parallel>` (default `sequential`; `parallel` fans out a proven-independent subset per `execution-batch.md` § Parallel Execution). `--keep-going` skips a failed task's in-batch dependents and continues independents (default halts on first failure). `--auto` sets `profile=auto` on each per-task run (skips the HITL approve gate). `--feature <id>` is sugar for `--tasks feature:<id>`; when the effective selector is feature-derived, the batch runs `spur feature check <id> --strict --json` **once** before task resolution — a non-zero strict check aborts with verdict `aborted`, zero attempted tasks, and the structured findings (task 0510 R2); scoped: `L4.scenario-unverified` (expected pre-run state of any not-yet-run feature) is reported verbatim but does not abort — any other strict error aborts. Explicit WBS/status/`ready` selectors add no feature check. The orchestrator loop itself continues in this session; interactive sequential omit/`inline` uses the host driver — host-controlled with eligible `agent.run` stages dispatching once to a native subagent and host fallback (task 0508) — while `--agent auto`/a name, parallel mode, and headless invocation keep the isolated per-task workflow subprocess boundary. `--agent <inline|auto|name>` pins the executor for the per-task stages (see [SSOT](cross-cutting.md#inline-default-execution-surface)); the value crosses into per-task `vars.agent`, not the orchestrator. `--json` emits the report as JSON. `--wrap` triggers `wrapup-pipeline.yaml` **once for the batch** over the `done` subset only (execution-batch.md Step 6: `vars.feature` only when every frozen task is `done`/`cancelled`; empty done subset skips with a reason), `--next` chains each task to terminal status then runs the same batch-once wrap, `--continue` resumes from checkpoint.
356
+ - **Three orthogonal axes (do not confuse):** `--keep-going` = batch failure policy (halt vs skip dependents); `--continue` = resume from checkpoint against the batch's original frozen identity (execution-batch.md § Batch continuation, task 0919); `--next` = per-task lifecycle chaining (advance status on a verdict — `dev-verify`/`dev-verifyall` only). `routing-table.md` offers `--continue` and `--next` as competing options for the same situation only when the batch was interrupted mid-run; otherwise they address different problems.
357
357
  - **Backing:** `sp:spur-dev` skill, `runall` operation → delegates the driver loop to the **`sp:super-planner`** agent (the batch orchestrator).
358
358
  - **Behavior:** The orchestrator reads [execution-batch.md](execution-batch.md) and drives: resolve selector → freeze set → topo-sort by `dependencies[]` (Kahn, WBS-ascending tie-break; cycle aborts) → resolve out-of-set deps by status (done → allow, else → block subtree) → run each task via `spur workflow run task-pipeline.yaml --async` + `spur workflow trace` polling → inspect terminal state + `.spur/run/<wbs>-verdict.json` → stop-the-batch default or `--keep-going` subtree skip → emit batch report. Per-task pipeline is invoked **verbatim** — no new FSM, no step edits. `--auto`/`--agent` are the only flags that cross the orchestrator→pipeline boundary (both into per-task `--vars`).
359
359
  - **Delegation:** `Skill(skill="sp:spur-dev", args="runall $ARGUMENTS")` → `sp:super-planner` agent.