@gobing-ai/spur 0.3.86 → 0.3.88

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/config/config.example.yaml +9 -0
  3. package/config/plugin-scripts.json +4 -0
  4. package/config/rules/strict/runtime-boundaries.yaml +1 -0
  5. package/config/templates/feature/default.md +2 -0
  6. package/config/templates/task/brainstorm.md +2 -2
  7. package/config/templates/task/feature-impl.md +2 -2
  8. package/config/templates/task/issue.md +2 -2
  9. package/config/templates/task/meta.md +2 -2
  10. package/config/templates/task/review.md +2 -2
  11. package/config/templates/task/standard.md +2 -2
  12. package/config/workflow-candidates.json +13 -35
  13. package/config/workflows/feature-lifecycle.yaml +6 -0
  14. package/config/workflows/feature-verification.yaml +6 -5
  15. package/config/workflows/idea-pipeline.yaml +89 -36
  16. package/config/workflows/task-pipeline.yaml +25 -0
  17. package/package.json +9 -9
  18. package/plugins/sp/README.md +9 -6
  19. package/plugins/sp/commands/dev-idea.md +9 -2
  20. package/plugins/sp/commands/dev-refactor.md +33 -0
  21. package/plugins/sp/commands/dev-refine.md +25 -7
  22. package/plugins/sp/commands/dev-refineall.md +9 -6
  23. package/plugins/sp/lib/idea-handoff.generated.mjs +160 -152
  24. package/plugins/sp/plugin.json +1 -1
  25. package/plugins/sp/references/roles.md +1 -1
  26. package/plugins/sp/scripts/idea-coverage-check.ts +168 -0
  27. package/plugins/sp/scripts/inline-pipeline-parity-check.ts +114 -3
  28. package/plugins/sp/scripts/inline-run-setup.ts +28 -18
  29. package/plugins/sp/skills/brainstorm/SKILL.md +4 -0
  30. package/plugins/sp/skills/code-refactoring/SKILL.md +155 -0
  31. package/plugins/sp/skills/code-refactoring/references/finding-schema.md +74 -0
  32. package/plugins/sp/skills/code-refactoring/references/fix-ladder.md +52 -0
  33. package/plugins/sp/skills/code-refactoring/references/focus-detection.md +44 -0
  34. package/plugins/sp/skills/code-refactoring/references/refactor-finding.schema.json +95 -0
  35. package/plugins/sp/skills/spec-decomposition/references/decomposition.md +23 -16
  36. package/plugins/sp/skills/spur-cli/references/agent.md +92 -9
  37. package/plugins/sp/skills/spur-cli/references/features/acceptance-criteria.md +4 -2
  38. package/plugins/sp/skills/spur-cli/references/features.md +6 -1
  39. package/plugins/sp/skills/spur-cli/references/workflows/authoring-workflows.md +2 -2
  40. package/plugins/sp/skills/spur-cli/references/workflows/operations.md +3 -3
  41. package/plugins/sp/skills/spur-cli/references/workflows/workflow-fit-and-tuning.md +1 -1
  42. package/plugins/sp/skills/spur-dev/SKILL.md +33 -33
  43. package/plugins/sp/skills/spur-dev/references/ac-style-guide.md +24 -0
  44. package/plugins/sp/skills/spur-dev/references/dev-operations.md +99 -26
  45. package/plugins/sp/skills/spur-dev/references/flag-glossary.md +27 -5
  46. package/plugins/sp/skills/spur-dev/references/idea-evaluation.md +11 -1
  47. package/plugins/sp/skills/spur-dev/references/inline-pipeline-driver.md +33 -1
  48. package/plugins/sp/skills/spur-dev/references/planning-workflow.md +10 -6
  49. package/plugins/sp/skills/taste-refactoring-api/SKILL.md +44 -1
  50. package/plugins/sp/skills/taste-refactoring-api/references/protocol-modes.md +31 -0
  51. package/plugins/sp/skills/taste-refactoring-architect/SKILL.md +43 -0
  52. package/plugins/sp/skills/taste-refactoring-tests/SKILL.md +42 -0
  53. package/plugins/sp/skills/taste-refactoring-ui/SKILL.md +43 -0
  54. package/schemas/spur-config.schema.json +14 -2
  55. package/schemas/task-batch.schema.json +2 -2
  56. package/spur.js +2012 -487
  57. package/web/_astro/{BoardApp.yBBcFXWP.js → BoardApp.CDUcHlTJ.js} +67 -67
  58. package/web/_astro/BoardApp.CaCGU_uX.js +1 -0
  59. package/web/_astro/{TaskDetail.DqFJbRFc.js → TaskDetail.DwTmbQp5.js} +1 -1
  60. package/web/_astro/{arc.DL-BpHoi.js → arc.BzF71EFI.js} +1 -1
  61. package/web/_astro/{architectureDiagram-3BPJPVTR.9IdQYyDq.js → architectureDiagram-3BPJPVTR.jvdDahWM.js} +1 -1
  62. package/web/_astro/{blockDiagram-GPEHLZMM.BKsFCqTl.js → blockDiagram-GPEHLZMM.zSg4AmFD.js} +1 -1
  63. package/web/_astro/{c4Diagram-AAUBKEIU.DhwI0dh1.js → c4Diagram-AAUBKEIU.BkUIUQWH.js} +1 -1
  64. package/web/_astro/channel.SSVY0JPQ.js +1 -0
  65. package/web/_astro/{chunk-2J33WTMH.B3QVmQ9S.js → chunk-2J33WTMH.DvfQ_f50.js} +1 -1
  66. package/web/_astro/{chunk-4BX2VUAB.DBSuqs9F.js → chunk-4BX2VUAB.DuI4gQqX.js} +1 -1
  67. package/web/_astro/{chunk-55IACEB6.BSYTWAYD.js → chunk-55IACEB6.D3BWBOpF.js} +1 -1
  68. package/web/_astro/{chunk-727SXJPM.cWVuxXfS.js → chunk-727SXJPM.3QSi0a9M.js} +1 -1
  69. package/web/_astro/{chunk-AQP2D5EJ.DpU_Ob3d.js → chunk-AQP2D5EJ.xazCQrAF.js} +1 -1
  70. package/web/_astro/{chunk-FMBD7UC4.BykFkyji.js → chunk-FMBD7UC4.B2g6u4rA.js} +1 -1
  71. package/web/_astro/{chunk-ND2GUHAM.DwgHlMdY.js → chunk-ND2GUHAM.wWwWs99t.js} +1 -1
  72. package/web/_astro/{chunk-QZHKN3VN.CusXUGWM.js → chunk-QZHKN3VN.BD5g3qa9.js} +1 -1
  73. package/web/_astro/{classDiagram-4FO5ZUOK.fx0ObzkN.js → classDiagram-4FO5ZUOK.C7CzCdsX.js} +1 -1
  74. package/web/_astro/{classDiagram-v2-Q7XG4LA2.fx0ObzkN.js → classDiagram-v2-Q7XG4LA2.C7CzCdsX.js} +1 -1
  75. package/web/_astro/{cose-bilkent-S5V4N54A.Z4HgOlsd.js → cose-bilkent-S5V4N54A.Xyiau0gw.js} +1 -1
  76. package/web/_astro/{cynefin-OW5HDTMX.B5dIZHJu.js → cynefin-OW5HDTMX.BeC5MWas.js} +1 -1
  77. package/web/_astro/{dagre-BM42HDAG.DT70Q_Yw.js → dagre-BM42HDAG.yZbMN9vc.js} +1 -1
  78. package/web/_astro/{diagram-2AECGRRQ.DqHA3XBF.js → diagram-2AECGRRQ.Cmo2zQM-.js} +1 -1
  79. package/web/_astro/{diagram-5GNKFQAL.BmCem957.js → diagram-5GNKFQAL.D033eSVi.js} +1 -1
  80. package/web/_astro/{diagram-KO2AKTUF.sn0-hrE0.js → diagram-KO2AKTUF.CR6k3Y3G.js} +1 -1
  81. package/web/_astro/{diagram-LMA3HP47.BSHe9tVc.js → diagram-LMA3HP47.x7mwu8jz.js} +1 -1
  82. package/web/_astro/{diagram-OG6HWLK6.DHIc-86k.js → diagram-OG6HWLK6.D8aTTvUr.js} +1 -1
  83. package/web/_astro/{erDiagram-TEJ5UH35.Bxayrs7v.js → erDiagram-TEJ5UH35.BoBqcKXQ.js} +1 -1
  84. package/web/_astro/{flowDiagram-I6XJVG4X.BkzoE_5I.js → flowDiagram-I6XJVG4X.D3mTQdrU.js} +1 -1
  85. package/web/_astro/{ganttDiagram-6RSMTGT7.okT6CvTo.js → ganttDiagram-6RSMTGT7.H-cqgIh-.js} +1 -1
  86. package/web/_astro/{gitGraphDiagram-PVQCEYII.CJuYbhC7.js → gitGraphDiagram-PVQCEYII.B6s9zbfC.js} +1 -1
  87. package/web/_astro/{infoDiagram-5YYISTIA.RqLy7nBo.js → infoDiagram-5YYISTIA.BzgCoV6P.js} +1 -1
  88. package/web/_astro/{ishikawaDiagram-YF4QCWOH.BwIcoagw.js → ishikawaDiagram-YF4QCWOH.BZzVhy1-.js} +1 -1
  89. package/web/_astro/{journeyDiagram-JHISSGLW.UB1VbWtH.js → journeyDiagram-JHISSGLW.BV3195Py.js} +1 -1
  90. package/web/_astro/{kanban-definition-UN3LZRKU.AaxMKpTk.js → kanban-definition-UN3LZRKU.BjRd2DWz.js} +1 -1
  91. package/web/_astro/{linear.Nv_xOUjP.js → linear.BILTgS5N.js} +1 -1
  92. package/web/_astro/{mermaid.core.Bc4LqQgX.js → mermaid.core.DBy_WKeW.js} +4 -4
  93. package/web/_astro/{mindmap-definition-RKZ34NQL.oKUvU_qi.js → mindmap-definition-RKZ34NQL.BiEjaI4-.js} +1 -1
  94. package/web/_astro/{pieDiagram-4H26LBE5.DQk0oo03.js → pieDiagram-4H26LBE5.i_8V5pIn.js} +1 -1
  95. package/web/_astro/{quadrantDiagram-W4KKPZXB.BdDjESDa.js → quadrantDiagram-W4KKPZXB.BWaW3MHn.js} +1 -1
  96. package/web/_astro/{requirementDiagram-4Y6WPE33.C2u9hUeH.js → requirementDiagram-4Y6WPE33.CzddBbtg.js} +1 -1
  97. package/web/_astro/{sankeyDiagram-5OEKKPKP.CDEoiJST.js → sankeyDiagram-5OEKKPKP.X2ww0e-D.js} +1 -1
  98. package/web/_astro/{sequenceDiagram-3UESZ5HK.D_hT_GAT.js → sequenceDiagram-3UESZ5HK.DSA4kTcc.js} +1 -1
  99. package/web/_astro/{stateDiagram-AJRCARHV.DI8RYG0b.js → stateDiagram-AJRCARHV.D0DtFSpR.js} +1 -1
  100. package/web/_astro/{stateDiagram-v2-BHNVJYJU.Bkxz4DnP.js → stateDiagram-v2-BHNVJYJU.BfQq0zQv.js} +1 -1
  101. package/web/_astro/{timeline-definition-PNZ67QCA.DSY-kH3-.js → timeline-definition-PNZ67QCA.Dmlrgi1m.js} +1 -1
  102. package/web/_astro/{vennDiagram-CIIHVFJN.CpaDtuGr.js → vennDiagram-CIIHVFJN.D5mpl00Z.js} +1 -1
  103. package/web/_astro/{wardleyDiagram-YWT4CUSO.DujQWvo8.js → wardleyDiagram-YWT4CUSO.Df4BdzO4.js} +1 -1
  104. package/web/_astro/{xychartDiagram-2RQKCTM6.DcM5Y4b9.js → xychartDiagram-2RQKCTM6.DiTRreKN.js} +1 -1
  105. package/web/index.html +1 -1
  106. package/web/_astro/BoardApp.CJiqp5pS.js +0 -1
  107. package/web/_astro/channel.CX5453qQ.js +0 -1
@@ -27,6 +27,14 @@ Rules:
27
27
  - **One R-number = one scenario.** Never split a requirement across multiple scenarios
28
28
  under the same R-number; never merge two requirements into one scenario.
29
29
 
30
+ ## Task-side numbering (`AC<n>`)
31
+
32
+ `R<n>` is the **feature** scenario key and the **task Requirements** key. Task `### Acceptance
33
+ Criteria` items therefore use their own namespace: `- [ ] AC1 — <title>` or `Scenario: AC1 — <title>`,
34
+ numbered task-locally. Carry a feature scenario by copying its title after the prefix (`normalizeTitle`
35
+ strips both `AC<n>` and `R<n>`, so DD-09 matching is unaffected); bind a task requirement with
36
+ `(req: R<n>)`. Legacy tasks that wrote `- [ ] R<n> —` / `Scenario: R<n> —` keep working unchanged.
37
+
30
38
  ## Two AC tiers (authoring convention)
31
39
 
32
40
  A planning convention (DD-06 "permissive start"), not a `spur feature check` feature today —
@@ -73,6 +81,10 @@ scenario, it matches by title. Rules:
73
81
  Registered user can log in with email and password" is traceable.
74
82
  - **No synonyms in cross-references.** The title in the feature file and the title in the
75
83
  task's AC reference must be byte-identical.
84
+ - **Avoid gate vocabulary in titles.** `spur task check` (L4.gate-language) rejects task sections
85
+ containing `HITL`, `approval`/`approved`, `merged`/`merge event`, `content-gate`, `GATED`, or
86
+ `capstone` as standalone words; task AC bullets copy scenario titles verbatim, so a title using
87
+ them fails every child task. Write "pause for an operator answer" instead of "HITL approval".
76
88
 
77
89
  ## Verdict AC ↔ feature scenario linkage
78
90
 
@@ -199,6 +211,18 @@ Use the canonical BDD template at `templates/bdd/gherkin.md`. Key rules:
199
211
  - **When** describes the single action under test.
200
212
  - **Then** asserts the observable outcome.
201
213
  - **And** chains additional preconditions, actions, or assertions.
214
+ - **Trace the scenario to its requirements (0887 R4).** Directly under each `Scenario:`
215
+ heading add a comment line listing the requirement-inventory ids the scenario covers;
216
+ the BDD parser skips `#` comment lines, so the form is checker-inert:
217
+
218
+ ```gherkin
219
+ Scenario: Registered user can log in with email and password
220
+ # covers: I1, I3
221
+ Given ...
222
+ ```
223
+
224
+ Every inventory item without a `[deferred: ...]` marker must be covered by at least one
225
+ scenario — `idea-coverage-check` measures this at the idea-pipeline's ac-generate boundary.
202
226
 
203
227
  Avoid:
204
228
 
@@ -85,7 +85,8 @@ each would be scope creep for one-liner procedures.
85
85
  | 13a | parallel | `dev-parallel` | `Skill()` | `sp:parallel-execution` | `--tasks <selector> [--feature <id>] [--mode <fan-out\|review-panel\|investigation>] [--agent <inline\|auto\|name>] [--json]` |
86
86
  | 14 | wrap | `dev-wrap` | `Skill()` | `spur workflow run` (wrapup-pipeline) | `<wbs> [--agent <inline\|auto\|name>] [--auto] [--merge] [--dry-run]` |
87
87
  | 15 | wrapall | `dev-wrapall` | `Skill()` | `spur workflow run` (wrapup-pipeline) | `[--since <iso>] [--feature <id>] [--status <s>] [--agent <inline\|auto\|name>] [--auto] [--merge] [--dry-run]` |
88
- | 16 | idea | `dev-idea` | `Skill()` | inline driver (`idea-pipeline`) or async workflow | `"<idea>" [--auto] [--skip-design] [--approve-taste] [--agent <inline\|auto\|name>]` |
88
+ | 16 | idea | `dev-idea` | `Skill()` | inline driver (`idea-pipeline`) or async workflow | `"<idea>" [--from-file <path>] [--auto] [--skip-design] [--approve-taste] [--agent <inline\|auto\|name>]` |
89
+ | 17 | refactor | `dev-refactor` | `Skill()` | `sp:code-refactoring` skill (thin wrapper, ADR-032) | `[<description>] [--scope <path>] [--focus <api\|architect\|tests\|ui\|auto>] [--fix <none\|blockers-first\|all>] [--check <cmd>] [--agent <inline\|auto\|name>] [--auto]` |
89
90
 
90
91
  ## Skill-backed operations
91
92
 
@@ -152,10 +153,23 @@ must not be changed without updating the backing skill.
152
153
 
153
154
  ### 5. refine
154
155
 
155
- - **Purpose:** Refine a task's requirements via structured Q&A — clarify scope, elicit missing details, tighten acceptance criteria before execution. Optional **implement-ready** depth freezes Design/Requirements/Plan so another agent can implement without inventing design.
156
- - **Inputs:** `<wbs>` (required). `--focus <mode>` narrows the gap analysis. `--depth <standard|ready>` (default **`standard`**) sets the depth bar — see [flag-glossary.md](flag-glossary.md#flag-depth). Execution defaults to inline (in-session); `--agent <inline|auto|name>` selector accepted (see [SSOT](cross-cutting.md#inline-default-execution-surface)). `--auto` skips interactive Q&A (synthesis only) and propagates down the `--next` chain. `--next`: advance to the next step — transition `backlog → todo` through the FSM **idempotently** (only when `status == backlog`; a task already at `todo` or past it skips the transition and chains anyway — `status >= todo` ⇒ already advanced) and invoke `/sp:dev-run <wbs> --mode implement --auto --next`. On a guard/refine failure, stop as review-pending.
156
+ - **Purpose:** Refine a task's requirements via structured Q&A — clarify scope, elicit missing details, tighten acceptance criteria before execution. Optional **implement-ready** depth freezes Design/Requirements/Plan so another agent can implement without inventing design, and is also the path for **evaluating and correcting an existing task** — a review-triage filing, a stale backlog capture, or any task whose claims and proposed fixes may no longer hold.
157
+ - **Inputs:** `<wbs>` (required). `--focus <mode>` narrows the gap analysis (values below). `--description <text>` injects operator framing into the Q&A/synthesis. `--depth <standard|ready>` (default **`standard`**) sets the depth bar — see [flag-glossary.md](flag-glossary.md#flag-depth). Execution defaults to inline (in-session); `--agent <inline|auto|name>` selector accepted (see [SSOT](cross-cutting.md#inline-default-execution-surface)). `--auto` skips interactive Q&A (synthesis only) and propagates down the `--next` chain. `--next`: advance to the next step — transition `backlog → todo` through the FSM **idempotently** (only when `status == backlog`; a task already at `todo` or past it skips the transition and chains anyway — `status >= todo` ⇒ already advanced) and invoke `/sp:dev-run <wbs> --mode implement --auto --next`. On a guard/refine failure, stop as review-pending.
158
+ - **Status scope:** refine targets `backlog`/`todo` tasks (the same rule `batch-preflight` applies to refineall). A task at `wip` or later has an implementation built on its current spec. Refine it only on an explicit operator request, never under `--auto` alone, and never move its status backwards.
159
+ - **`--focus` values** (hint bundles for the gap analysis/Q&A; default `all`):
160
+
161
+ | Value | Domain hints | When |
162
+ | --- | --- | --- |
163
+ | `all` | purpose, scope, constraints, dependencies, acceptance criteria, users, timeline | Complete refinement |
164
+ | `requirements` | purpose, scope, acceptance criteria | Standard refinement |
165
+ | `background` | purpose, scope | Thin tasks needing context |
166
+ | `constraints` | constraints, dependencies, timeline | Technical depth |
167
+ | `acceptance` | acceptance criteria, users | Verification focus |
168
+ | `quick` | scope, acceptance criteria | Fast pass |
169
+
170
+ Focus narrows what `--depth standard` looks at. Under `--depth ready` the full checklist still runs, and focus only orders the work.
157
171
  - **Backing:** `sp:spur-dev` skill, `refine` operation. Q&A clarifications are presented as decision briefs per [decision-brief.md](decision-brief.md).
158
- - **Behavior:** Read the task → elicit missing AC/Design/Plan through targeted Q&A (or auto-synthesis) → write each via `spur task update <wbs> --section <name> --from-file`. Done just-in-time, per task, immediately before execution. With `--next`: on success, transition status (idempotently — see Inputs) + chain to dev-run; on failure, stop and surface error.
172
+ - **Behavior:** Read the task → (ready depth) audit the existing content against the current tree → elicit missing or wrong Background/Requirements/AC/Design/Plan through targeted Q&A (or auto-synthesis) → write each via `spur task update <wbs> --section <name> --from-file`. Done just-in-time, per task, immediately before execution. With `--next`: on success, transition status (idempotently — see Inputs) + chain to dev-run; on failure, stop and surface error.
159
173
  - **Pre-synthesis skip gate (under `--auto` + `--depth standard` only):** Before invoking synthesis, run `spur task check <wbs> --json`. Filter the findings to the **refine target sections** only:
160
174
  `{Background, Requirements, Acceptance Criteria, Design, Plan}`.
161
175
  These are the anti-drift surfaces: constraints + planning that cheaper implementers must follow.
@@ -205,28 +219,72 @@ must not be changed without updating the backing skill.
205
219
  SKIP — sections already meet implement-ready checklist: depth=ready, sections-considered=[…]
206
220
  ```
207
221
 
208
- **Implement-ready checklist (all must hold for allowed target sections):**
209
- 1. **Requirements** — R-items are observable outcomes; explicit out-of-scope / non-goals; no
222
+ **Implement-ready checklist (all must hold for allowed target sections).** Each item's `id` is
223
+ the `READY_CHECKLIST_IDS` value (`packages/app/src/services/task-readiness.ts`) that the
224
+ create-time ready preparation and the idea-pipeline ready-prepare stage also use:
225
+ 1. **`requirements`** — R-items are observable outcomes; explicit out-of-scope / non-goals; no
210
226
  ambiguous “wire it up” without a named seam or file area.
211
- 2. **Design** — WHAT / WHY / WHERE; **frozen names** (types, flags, vars, paths) **or** explicit
227
+ 2. **`design`** — WHAT / WHY / WHERE; **frozen names** (types, flags, vars, paths) **or** explicit
212
228
  “no new API”; precedence / algorithm when behavior is non-obvious; **anti-patterns** (what not
213
229
  to implement); primary file/package targets; handoff to dependent tasks (WBS) if any.
214
- 3. **Plan** — ordered checklist mappable to R-items; test/verification intent called out.
215
- 4. **Acceptance Criteria** — scenarios still match feature R-titles when `feature_id` is set;
216
- Given/When/Then still executable as a verify lens.
217
- 5. **Q&A / References** — open decisions closed or explicitly deferred with owner; links to ADR /
230
+ 3. **`plan`** — ordered checklist mappable to R-items; test/verification intent called out.
231
+ 4. **`ac`** — every scenario uses an [ac-style-guide.md](ac-style-guide.md) task-side form
232
+ (`- [ ] AC<n> — <title>` or `Scenario: AC<n> — <title>`, with `(req: R<n>)`). `spur task check`
233
+ parses only checkbox and `Scenario:` lines, so any other shape (a plain `- AC1 —` bullet, a bold
234
+ heading) silently escapes its AC checks. When `feature_id` is set, each AC
235
+ title matches a feature scenario title, unless the task is deliberately `ac_altitude:
236
+ task-local`. Given/When/Then stays executable as a verify lens.
237
+ 5. **`decisions`** — Q&A open decisions closed or explicitly deferred with owner; links to ADR /
218
238
  feature / upstream tasks present when the design depends on them.
219
- 6. **Cross-task** — if `dependencies[]` exist, Design states what this task assumes from deps and
220
- what it must leave for dependents (no silent re-ownership of upstream contracts).
221
- 7. **Premise verification** — every factual claim in Background and Requirements that the Design
222
- depends on (a status, a file/table/location, an already-landed fix, a count) is checked against
223
- the **current tree** — read the file, run the query, grep the corpus. Contradictions are
224
- corrected in **this** refine (rewrite the claim, or re-point the design at ground truth), never
225
- deferred to the implementer. `--depth ready` exists so a downstream agent does not re-derive the
226
- analysis; a frozen design built on a false premise is the worst available outcome.
227
-
228
- Ready depth is for multi-package work, multi-agent implement handoffs, and costly pipeline
229
- failures — not for every small task. Default remains `standard`.
239
+ 6. **`dependencies`** — if `dependencies[]` exist, Design states what this task assumes from deps
240
+ and what it must leave for dependents (no silent re-ownership of upstream contracts). A
241
+ dependency the Design relies on but `dependencies[]` lacks is added with `spur task deps`.
242
+ 7. **`premises`** — every factual claim in Background, Requirements, Design and Plan is checked
243
+ against the **current tree** (read the file, run the query, grep the corpus), not only the gaps.
244
+ This is the **audit pass**; it matters most for a task written earlier or by another agent
245
+ (review-triage filings, stale backlog). Lenses:
246
+ - **Facts** — statuses, `file:line` references, type/field/flag names, counts, "already fixed"
247
+ or "not yet implemented" claims.
248
+ - **Sources** — a cited run, artifact, session or doc exists and says what the task claims. If
249
+ it is gone, say so and restate the claim from code.
250
+ - **Fix soundness** — each proposed fix holds for every caller and mode of the seam it touches,
251
+ does not break a currently valid path, and reuses an existing mechanism before adding one.
252
+ - **Test observability** — each AC names a test layer (and file) that can actually observe the
253
+ behavior. A test that mocks the collaborator carrying the behavior cannot.
254
+ - **Environment** — installed dependency versions match the lockfile and generated artifacts
255
+ are current. If not, the Plan gets a step-0 precondition.
256
+ - **Concurrency** — active worktrees or `wip` tasks touching the same files (`git worktree
257
+ list`, `spur task list --status wip --json`) are recorded in References/Design.
258
+ - **Scope** — the work is not already done, and not owned by another task.
259
+
260
+ Contradictions are corrected in **this** refine (rewrite the claim, or re-point the design at
261
+ ground truth), never deferred to the implementer. `--depth ready` exists so a downstream agent
262
+ does not re-derive the analysis; a frozen design built on a false premise is the worst
263
+ available outcome.
264
+
265
+ **Correction record.** When the audit changes a claim, append a dated block to Background —
266
+ `**Refine corrections (<YYYY-MM-DD>)**`, one line per correction: claim → verified reality →
267
+ resolution. Never delete an earlier block, so the next reader sees what changed and why. Scope
268
+ decisions made along the way also go into Q&A as closed decisions.
269
+
270
+ **Ready finalization.** Once every item holds:
271
+ 1. Run `spur task check <wbs> --as todo --json`. Any error means the task is not ready. Fix it,
272
+ or report `failed`.
273
+ 2. Fill metadata that is still unset: `spur task update <wbs> --priority <P0–P3>` and
274
+ `--estimate-hours <n>` (both `--json`). Never overwrite a value the operator set.
275
+ 3. Promote `backlog → todo` idempotently with `spur task update <wbs> todo --json`, the same FSM
276
+ transition `--next` uses. Create-time ready preparation skips its own promotion when the task
277
+ is already `todo`, and next-router stops routing the task back to refine. Without `--next`,
278
+ refine does not chain into run.
279
+ 4. Report the checklist as rows `{id, pass, evidence}`, one per id, in the markdown result and in
280
+ the `--json` object, alongside the corrections count and status before → after.
281
+
282
+ A decision refine cannot close under `--auto` makes the outcome `failed`. Report the concrete
283
+ question, and leave the task at `backlog`.
284
+
285
+ Ready depth is the canonical path for three jobs: the ready competency behind `spur task create`
286
+ (its recovery command), evaluating and correcting an existing task, and freezing multi-package or
287
+ multi-agent handoffs. It is not for every small task. Default remains `standard`.
230
288
 
231
289
  - **SKIP short-circuits synthesis, not `--next`.** A SKIP means no synthesis was needed — it does **not** cancel the `--next` chain. Under `--auto --next`, a SKIP still flows into the (idempotent) status transition and the chained `/sp:dev-run --mode implement`. "`refine --auto --next` on a well-specified task" is therefore effectively "run the implement→verify chain"; an operator who wanted refinement only should drop `--next`.
232
290
  - **Delegation:** `Skill(skill="sp:spur-dev", args="refine $ARGUMENTS")`
@@ -243,13 +301,13 @@ must not be changed without updating the backing skill.
243
301
  1. Resolve + **freeze** the set at kickoff (never re-query membership mid-batch).
244
302
  2. Apply `--status` filter (default `backlog` + `todo`; applied in-agent against the frozen set; `spur task list --status` takes exactly one canonical status per call — see `execution-batch.md` Step 1). Tasks already `done`/`cancelled`/`testing` are excluded unless the operator widens `--status`. Report each exclusion with reason.
245
303
  3. Topo-sort by `dependencies[]` (Kahn, WBS-ascending tie-break). Cycle → abort entire batch before any refine. Out-of-set deps: `done` → allow; else → block subtree (same as runall).
246
- 4. For each WBS in order: invoke single-task refine with shared flags **including `--depth`**. Under `--auto` + **`--depth standard`** (default), the per-task **L3 pre-synthesis SKIP gate** still applies. Under **`--depth ready`**, each task runs the implement-ready checklist (no L3-only SKIP).
304
+ 4. For each WBS in order: invoke single-task refine with shared flags **including `--depth`**. Under `--auto` + **`--depth standard`** (default), the per-task **L3 pre-synthesis SKIP gate** still applies. Under **`--depth ready`**, each task runs the implement-ready checklist (no L3-only SKIP). That includes the premises audit, the correction record and ready finalization, so a passing task leaves at `todo`.
247
305
  5. Failure policy: **stop-the-batch** (default) or `--keep-going` (skip in-batch dependents of a failed refine; continue independents).
248
- 6. Emit a batch report (markdown or `--json`) that records `depth` once at the header.
306
+ 6. Emit a batch report (markdown or `--json`) that records `depth` once at the header. Under `--depth ready`, each row also carries the corrections count, status before → after, and failed checklist ids.
249
307
  - **Per-task outcome vocabulary:** `refined` (synthesis wrote sections) | `SKIP` (already meets the active depth bar under `--auto`) | `failed` | `skipped` (dep failed under `--keep-going`) | `not-attempted` (halted) | `blocked` (unmet out-of-set dep).
250
308
  - **Batch verdict:** `clean` (all attempted tasks `refined` or `SKIP`) | `halted` (a failure stopped the batch) | `aborted` (cycle / unknown selector / empty set after filter).
251
309
  - **`--next` is not accepted** (dropped by feature H8, 2026-07-31 — see `plugins/sp/commands/dev-refineall.md` for the removal record). Chain execution explicitly: refineall, then `/sp:dev-runall --feature <id>`.
252
- - **`--auto` recommendation:** Batch refine without `--auto` requires per-task interactive Q&A and does not scale. Default operator path: `/sp:dev-refineall --feature <id> --auto`. For implement handoffs: `/sp:dev-refineall --feature <id> --auto --depth ready`.
310
+ - **`--auto` recommendation:** Batch refine without `--auto` requires per-task interactive Q&A and does not scale. Default operator path: `/sp:dev-refineall --feature <id> --auto`. For implement handoffs, or to re-audit a feature's filed tasks: `/sp:dev-refineall --feature <id> --auto --depth ready`.
253
311
  - **Delegation:** `Skill(skill="sp:spur-dev", args="refineall $ARGUMENTS")` → per task `Skill(skill="sp:spur-dev", args="refine <wbs> $SHARED_FLAGS")` (shared flags include `--depth` when set).
254
312
 
255
313
  ### 6. plan
@@ -335,11 +393,18 @@ must not be changed without updating the backing skill.
335
393
  ### 16. idea
336
394
 
337
395
  - **Purpose:** Turn a vague idea into a feature with AC and a decomposed task batch — the unified entry point for the planning half.
338
- - **Inputs:** `"<idea>"` (required, positional, quoted). Three everyday axes:
396
+ - **Inputs:** `"<idea>"` (quoted) or `--from-file <path>` — exactly one; the two are mutually
397
+ exclusive (0887 R7). Three everyday axes:
339
398
  - `--auto` — skip **objective** HITL (feature-check, batch-create); taste gates still pause.
340
399
  - `--skip-design` — design package off (system-design + task Design).
341
400
  - `--approve-taste` — with `--auto`, skip **all** remaining taste pauses this run (idea-eval + design-approval). Sets `idea_approved=true` and `design_approved=true`.
342
401
  Aliases (prefer `--approve-taste`): `--idea-approved` → `idea_approved`; `--design-approved` → `design_approved`. There is **no** `--design` force flag.
402
+ - `--from-file <path>` — read the idea text from a file instead of the positional argument
403
+ (verbatim; long/multiline asks). Mutually exclusive with `"<idea>"`.
404
+ - **Verbatim idea artifact:** before `start` executes, the driver persists the idea argument (or
405
+ `--from-file` contents) unmodified to `.spur/run/<run-id>-idea-input.md` (0887 R1); the precheck
406
+ fails the run when that file is empty or missing, and every model-bearing stage prompt treats
407
+ it as the authoritative ask (R2).
343
408
  - **Backing:** `idea-pipeline.yaml` through the inline driver for omitted/`inline`, or `spur workflow run idea-pipeline.yaml --async` for `auto`/name.
344
409
  - **Behavior:** Builds vars from the table above and drives the idea pipeline. Flow: discovery → **idea-eval** (taste; reject → cancelled) → feature-create → ac-generate → feature-check → system-design (conditional) → design-approval (taste) → decompose → batch-create (`--skip-ready`) → ready-prepare (ready checklist per created task + ready-evidence sidecar, 0788) → handoff. STOPS at handoff — no task execution, no pipeline nesting. Headless runs use one `trace --follow`; cancellation is reported stopped only when `workflow cancel --json` returns `killed: true`.
345
410
  - **Delegation:** Host-session inline driver by default; explicit executor selection uses the async workflow worker.
@@ -356,6 +421,14 @@ must not be changed without updating the backing skill.
356
421
 
357
422
  - **Taste pre-clear (`--approve-taste`):** owned with design-approval var semantics in [cross-cutting.md](cross-cutting.md) § "Design Approval Gate"; idea-eval uses the parallel `idea_approved` var. One CLI flag sets both.
358
423
 
424
+ ### 17. refactor
425
+
426
+ - **Purpose:** Lens-routed refactoring with a preservation contract — route a scope through taste lenses (api / architect / tests / ui, auto-detected by path), classify findings on the shared `refactor-finding` schema, and apply via the fix ladder without breaking preserved behavior.
427
+ - **Inputs:** `[<description>]` free-text steering, `--scope <path>` (default: working tree), `--focus <lens>` (default: auto), `--fix <policy>` (default: none), `--check <cmd>` (default: project gate), `--agent <selector>` (default: inline), `--auto` (default: off).
428
+ - **Backing:** `sp:code-refactoring` skill — the command carries zero orchestration logic (ADR-032); the skill owns focus auto-detection, lens dispatch, the P1–P4 severity map, objective/taste gates, and the fix ladder with revert-on-regression.
429
+ - **Behavior:** Green `--check` baseline → focus detection → lens dispatch → findings to `.spur/run/<run-id>-refactor-findings.json` + human report `.spur/run/<run-id>-refactor-report.md` (lens set, P1–P4 table, preservation summary, applied/reverted/deferred). `--fix blockers-first` auto-applies P1/P2 `auto`-eligible findings; `--fix all` extends to `operator`-eligible P3/P4; every fix re-runs `--check` and reverts on regression. Tests are never removed or weakened by an `auto` fix.
430
+ - **Operator taste gates:** every `cutting` or `breaking` finding pauses for an explicit operator answer in every mode — even `--auto` skips only objective gates. `--fix none` (the default) writes both artifacts with no edits.
431
+
359
432
  ---
360
433
 
361
434
  ## Inline operations
@@ -160,13 +160,22 @@ Scope the operation to all tasks under a feature id (`^[A-Z][1-9]*$`). On featur
160
160
  commands (`dev-wrapall`) it also advances the feature through legal lifecycle edges with guards
161
161
  honored.
162
162
 
163
+ ### `--check <cmd>` — validation command for iterate-and-check loops
164
+
165
+ **Anchor:** `#flag-check`.
166
+
167
+ Verification command a command iterates against (`dev-simplify`, `dev-refactor`). The command
168
+ establishes a baseline with it before the first change and re-runs it after each change.
169
+
163
170
  ### `--focus <dims>` — constrain the operation to specific dimensions
164
171
 
165
172
  **Anchor:** `#flag-focus`.
166
173
 
167
174
  Constrain the operation to a named subset of dimensions — review dimensions on `dev-review`/
168
175
  `dev-verify`/`dev-verifyall` (`all|stack|dependencies|data|flows|api|security|quality|performance`),
169
- a refine focus mode on `dev-refine`/`dev-refineall`, or a reconstruction lens on `dev-reverse`.
176
+ a refactor lens set on `dev-refactor` (`api|architect|tests|ui|auto`), a refine focus mode on
177
+ `dev-refine`/`dev-refineall` (`all|requirements|background|constraints|acceptance|quick` —
178
+ [dev-operations.md](dev-operations.md) § refine), or a reconstruction lens on `dev-reverse`.
170
179
  Narrowing reduces token cost; omitting runs
171
180
  all dimensions.
172
181
 
@@ -175,7 +184,7 @@ all dimensions.
175
184
  **Anchor:** `#flag-scope`.
176
185
 
177
186
  Limit the operation to a file or directory path (`dev-arch`, `dev-debug`, `dev-fixall`,
178
- `dev-gitmsg`, `dev-gtd`, `dev-simplify`) to bound the working set.
187
+ `dev-gitmsg`, `dev-gtd`, `dev-refactor`, `dev-simplify`) to bound the working set.
179
188
 
180
189
  ### `--all` — widen the operation to everything in its domain
181
190
 
@@ -253,7 +262,8 @@ tasks by `updated_at >= date`).
253
262
 
254
263
  **Anchor:** `#flag-fix`.
255
264
 
256
- Remediation policy on verify-family commands (`dev-verify`, `dev-verifyall`):
265
+ Remediation policy on verify-family commands (`dev-verify`, `dev-verifyall`) and the refactor
266
+ coordinator (`dev-refactor`):
257
267
  `none|blockers-first|all`. `none` reports findings without fixing; `blockers-first` fixes only P1/P2;
258
268
  `all` fixes everything found. Deprecated on `dev-review` (routes to `dev-verify --fix`).
259
269
 
@@ -294,6 +304,16 @@ verifying a task whose artifact is intentionally not yet shippable (e.g. a doc-o
294
304
  Omit the design package (system-design satellite + task `### Design`) on planning commands
295
305
  (`dev-plan`, `dev-idea`). The task is created without the design section; refine supplies it later.
296
306
 
307
+ ### `--from-file <path>` — read the idea from a file (dev-idea)
308
+
309
+ **Anchor:** `#flag-from-file`.
310
+
311
+ `dev-idea` reads the idea text from `<path>` instead of the positional argument. Mutually
312
+ exclusive with `"<idea>"` — exactly one must be present. The file's contents become the verbatim
313
+ idea text, persisted unmodified to `.spur/run/<run-id>-idea-input.md` (0887 R1) and treated as
314
+ the authoritative ask by every model-bearing stage prompt. Useful for long or multiline asks
315
+ that are awkward to quote (0887 R7).
316
+
297
317
  ### `--output <path>` — write the result to a path
298
318
 
299
319
  **Anchor:** `#flag-output`.
@@ -353,10 +373,12 @@ Orthogonal to `--focus` (which _narrows_ domains) and to `--mode` on other comma
353
373
  | Value (refine family) | Bar | `--auto` SKIP behavior |
354
374
  | --------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------- |
355
375
  | `standard` (default when omitted) | L3 structural completeness (not empty/placeholder; check-clean for target sections) | **SKIP** when no L3 findings on target sections |
356
- | `ready` | **Implement-ready** freeze: another agent can implement without inventing design (frozen names/APIs or explicit "no new API", anti-patterns, file targets, handoffs, out-of-scope) | **Do not SKIP** on L3-clean alone — run the ready checklist; rewrite sections until the bar is met |
376
+ | `ready` | **Implement-ready** freeze: another agent can implement without inventing design (frozen names/APIs or explicit "no new API", anti-patterns, file targets, handoffs, out-of-scope) | **Do not SKIP** on L3-clean alone — audit existing claims, run the ready checklist, rewrite sections until the bar is met, then promote `backlog → todo` |
357
377
 
358
378
  Default for refine stays `standard` so ordinary `refineall --auto` remains cheap. Use `ready` for
359
- multi-package / multi-agent handoffs and flaky-pipeline features where a wrong implement is costly.
379
+ multi-package / multi-agent handoffs, flaky-pipeline features where a wrong implement is costly, and
380
+ evaluating/correcting an existing task (review-triage filing, stale backlog) whose claims may no
381
+ longer hold.
360
382
  Full checklist: [dev-operations.md](dev-operations.md) § refine (depth ready).
361
383
 
362
384
  ### `--bdd` — use BDD scenarios as the verification lens
@@ -20,7 +20,9 @@ recommendation is mandatory, stakes in plain English, and approve/reject is the
20
20
  re-author the report.
21
21
 
22
22
  **Sidecar rule:** The enhanced idea does **not** overwrite `vars.idea`. Feature-create reads both
23
- the original idea and this report.
23
+ the original idea and this report. The operator's verbatim ask of record is the run's
24
+ `.spur/run/<run-id>-idea-input.md` (persisted by the pipeline `start` state / the inline driver
25
+ before any processing); the `## Requirement inventory` items trace back to it.
24
26
 
25
27
  ## Template
26
28
 
@@ -30,6 +32,13 @@ the original idea and this report.
30
32
  ## Enhanced Idea
31
33
  <one-paragraph refined statement of what the idea actually requires — the "real requirement" after discovery sharpens the vague input>
32
34
 
35
+ ## Requirement inventory
36
+ <mandatory — the coverage gate (idea-coverage-check) parses this section, so keep the exact `- I<n> — ` item form>
37
+ - I1 — <requirement stated as an ask, quoting or paraphrasing the source line from the run's idea-input artifact> (source: "<quoted fragment from the operator's idea>")
38
+ - I2 — <next requirement>
39
+ - I<n> — <optional: a requirement explicitly out of scope> [deferred: <reason>]
40
+ - I<n> — <optional: an ambiguous ask> [unclear: <why it is ambiguous>] — an unclear marker does not exempt the item; it still needs coverage or an explicit deferral
41
+
33
42
  ## Scores
34
43
 
35
44
  | Dimension | Score (0–5) | Rationale |
@@ -74,6 +83,7 @@ Stakes: <plain-English cost of proceeding vs not; reversibility; blast radius>
74
83
  |------|--------|
75
84
  | Filled instance path | `.spur/run/idea-eval-report.md` |
76
85
  | Template home | this file |
86
+ | Requirement inventory | mandatory `## Requirement inventory` section (0887 R3); consumed by `idea-coverage-check` (R4) |
77
87
  | HITL state | `idea-eval` in `idea-pipeline.yaml` |
78
88
  | Approve | continue → `feature-create` |
79
89
  | Reject / cancel | → `cancelled` (no feature) |
@@ -160,6 +160,35 @@ the human/native presentation layer — labels are display addresses only, never
160
160
  This is required for the normal `testing → done` provenance guard. Planning pipelines have no
161
161
  task lifecycle link and skip this task-specific action.
162
162
 
163
+ ### Idea-pipeline quick start (0887 R1/R2)
164
+
165
+ Minimum files to read for `/sp:dev-idea` inline runs — then drive `idea-pipeline.yaml`:
166
+
167
+ - `.spur/workflows/idea-pipeline.yaml` (the machine) and this driver.
168
+ - `plugins/sp/skills/spur-dev/references/idea-evaluation.md` (report template incl. the mandatory
169
+ `## Requirement inventory`) and `references/ac-style-guide.md` (scenario `# covers:` form).
170
+ - `references/dev-operations.md` § idea for the stage-by-stage surface.
171
+
172
+ **Persist the verbatim idea FIRST.** Before executing the `start` state, write the operator's
173
+ idea argument (or `--from-file` contents) **unmodified** to
174
+ `.spur/run/<run-id>-idea-input.md`; the run precheck fails when that file is empty or missing,
175
+ and every model-bearing stage prompt treats it as the authoritative ask. `--from-file` and the
176
+ positional idea are mutually exclusive — exactly one must be present.
177
+
178
+ Expected artifacts per stage (all run-scoped under `.spur/run/<run-id>-*`):
179
+
180
+ | Stage | Artifacts |
181
+ | ----- | --------- |
182
+ | start | `-idea-input.md` (verbatim idea), `-idea-precheck-doctor.status` |
183
+ | discovery | `-idea-eval-report.md` (with `## Requirement inventory`), `-idea-needs-design.json` |
184
+ | feature-create | `-idea-feature-id.txt`, `-idea-goal.md`, `-idea-scope.md` |
185
+ | ac-generate | `-idea-ac-content.md`, `-idea-ac-check.status`, `-idea-coverage.status` |
186
+ | system-design | `-idea-design-review.md`, `-idea-design-check.status` |
187
+ | decompose | `-idea-task-batch.json`, `-idea-task-order.json` |
188
+ | batch-create-run | `-idea-batch-create-result.json`, `-idea-batch-create.done`/`.failed` |
189
+ | ready-prepare | `-idea-ready.json` |
190
+ | handoff-finalize | `-idea-handoff.md` |
191
+
163
192
  ## Comprehensive-check retention and evidence (R7/R8)
164
193
 
165
194
  **R7 — comprehensive checks stay at their owning boundaries.** Quick readiness and plan projection are
@@ -428,7 +457,10 @@ The driver reaches it through the existing run delegate (`$SETUP_SCRIPT`,
428
457
  under its declared error policy and `failed` otherwise; `--duration-ms` is the wall clock the
429
458
  driver measured around the action. This writes the `action_runs` row (node, kind, status, `ok`,
430
459
  `duration_ms`, `run_id`) the engine would have written, so the run's rows are queryable by run id
431
- (`spur workflow progress <run-id>`, `ActionRunDao`) without reading the text log.
460
+ (`spur workflow progress <run-id>`, `ActionRunDao`) without reading the text log. The writer
461
+ back-dates the row's `started_at` from its own `completed_at` minus the measured duration
462
+ (0887 R8), so `completed_at − started_at == duration_ms` exactly; a back-date failure is
463
+ recorded (`action.backdate`) and never affects the run.
432
464
 
433
465
  - **At the run's declared terminal state** — before the driver reports the run complete, close the
434
466
  row so a successful inline run is never left non-terminal for `spur workflow clean` to reap as
@@ -301,10 +301,12 @@ the handoff degrades to refineall.
301
301
  (and preferably Plan/AC) so tasks land **content-ready**. **`--skip-design`:** leave `design`
302
302
  empty — headings only.
303
303
 
304
- **Refine is the fallback**, not the primary Design author:
304
+ **Refine is the fallback** Design author, not the primary one. It is also the audit path for
305
+ tasks that already have content:
305
306
 
306
307
  ```text
307
308
  /sp:dev-refine <wbs> # single task — fills blank Design/AC/Plan if L3 gaps
309
+ /sp:dev-refine <wbs> --depth ready # evaluate + correct an existing/filed task, then promote to todo
308
310
  /sp:dev-refineall --feature X --auto
309
311
  /sp:dev-refineall --feature X --auto --depth ready # implement-ready freeze (no L3-only SKIP)
310
312
  ```
@@ -314,7 +316,9 @@ Under `--auto` + **`--depth standard`** (default), refine **SKIP**s when target
314
316
  placeholder, synthesis runs (standard tier by default; escalates only on gate-fail). Under
315
317
  **`--depth ready`**, do not SKIP on L3-clean alone — run the implement-ready checklist in
316
318
  [dev-operations.md](dev-operations.md) § refine (frozen APIs, anti-patterns, file targets, handoffs)
317
- so another agent can implement without inventing design.
319
+ so another agent can implement without inventing design. Ready depth also audits every existing
320
+ claim against the current tree, records corrections in Background, and promotes a passing task
321
+ `backlog → todo`.
318
322
 
319
323
  **Check the variant before you write.** Which sections a task carries is decided by its `template:`
320
324
  frontmatter against `.spur/tasks/section-matrix.yaml` — NOT a fixed list. Before authoring any
@@ -344,16 +348,16 @@ feature filled before a runall, use `/sp:dev-refineall --feature <id> --auto` (b
344
348
  of `/sp:dev-refine`). It reuses the same per-task refine operation, freezes the set, topo-sorts by
345
349
  `dependencies[]`, and emits a batch report — see [dev-operations.md](dev-operations.md) § refineall.
346
350
  This does **not** replace just-in-time refine before each implement; it is a bulk pre-pass when the
347
- feature's tasks are still `backlog`/`todo` placeholders. Prefer `--auto` for batch scale; avoid
348
- `--next` on large features (that chains each task into run).
351
+ feature's tasks are still `backlog`/`todo` placeholders. Prefer `--auto` for batch scale.
352
+ `/sp:dev-refineall` takes no `--next`; chain with `/sp:dev-runall --feature <id>` afterwards.
349
353
 
350
354
  **Refine arguments** (defined on the `/sp:dev-refine` entry point, passed through verbatim; also
351
355
  shared flags on `/sp:dev-refineall`):
352
356
 
353
357
  | Argument | Effect |
354
358
  |----------|--------|
355
- | `--focus <mode>` | Narrows the gap analysis to a subset of domain hints. See the `sp:dev-refine` skill for the full value table (`all`, `requirements`, `background`, `constraints`, `acceptance`, `quick`). Default `all`. |
356
- | `--depth <standard\|ready>` | Spec depth bar. `standard` (default) = L3 structural completeness + L3 SKIP under `--auto`. `ready` = implement-ready freeze (never L3-only SKIP). See [flag-glossary.md](flag-glossary.md#flag-depth). |
359
+ | `--focus <mode>` | Narrows the gap analysis to a subset of domain hints. Values `all`, `requirements`, `background`, `constraints`, `acceptance`, `quick` — hint table in [dev-operations.md](dev-operations.md) § refine. Default `all`. Under `--depth ready` it only orders the work. |
360
+ | `--depth <standard\|ready>` | Spec depth bar. `standard` (default) = L3 structural completeness + L3 SKIP under `--auto`. `ready` = audit + implement-ready freeze + promote to `todo` (never L3-only SKIP). See [flag-glossary.md](flag-glossary.md#flag-depth). |
357
361
  | `--auto` | Skip interactive Q&A — synthesize improvements from the task content alone. Use for well-scoped tasks where the agent can fill gaps without operator input. **Required for practical batch use** via `dev-refineall`. |
358
362
 
359
363
  **Pre-synthesis skip gate (under `--auto` + `--depth standard`).** Before synthesizing, run `spur task check <wbs> --json`. When the **refine target sections** show no L3 findings, emit a structured SKIP instead of calling the synthesis agent. **Not applied when `--depth ready`.**
@@ -1,6 +1,18 @@
1
1
  ---
2
2
  name: taste-refactoring-api
3
3
  description: Design, review, and refactor REST/HTTP, RPC/gRPC, GraphQL, and event API contracts safely.
4
+ license: Apache-2.0
5
+ metadata:
6
+ author: spur
7
+ version: "1.0"
8
+ platforms: "claude-code,codex,openclaw,opencode,antigravity"
9
+ category: execution
10
+ interactions:
11
+ - technique
12
+ operations:
13
+ - refactor-apis
14
+ openclaw:
15
+ emoji: "🔌"
4
16
  ---
5
17
 
6
18
  # taste-refactoring-api
@@ -174,7 +186,7 @@ The goal is to eliminate repeated low-level API decisions, not to create bureauc
174
186
 
175
187
  ## Protocol-specific review
176
188
 
177
- After classifying the API, read the matching REST/HTTP, RPC/gRPC, GraphQL, or event/webhook section in [references/protocol-modes.md](references/protocol-modes.md). Apply that checklist before continuing with the refactoring strategy.
189
+ After classifying the API, read the matching REST/HTTP, RPC/gRPC, GraphQL, or event/webhook section in [references/protocol-modes.md](references/protocol-modes.md). Apply that checklist before continuing with the refactoring strategy. For command-line surface work (designing or reviewing a CLI), read the [CLI mode](references/protocol-modes.md) section instead — a CLI is a contract surface with the same compatibility duties as any wire protocol.
178
190
 
179
191
  ## Refactoring strategy
180
192
 
@@ -332,3 +344,34 @@ Specify the tests/checks needed: contract tests, schema validation, compatibilit
332
344
  ## Decision rule
333
345
 
334
346
  A “better” API is not the one with the prettiest route names. It is the one that makes common client code obvious, predictable, safe under failure, compatible over time, secure by default, and operable in production.
347
+
348
+ ## Spur contract
349
+
350
+ Machine-facing adapter for `sp:code-refactoring` dispatch (feature H13). Everything above is the
351
+ lens's native output and is unchanged; this section only maps it to the shared finding schema.
352
+
353
+ **Inputs received:** a scope path, an optional change description, and the `api` focus. Read
354
+ first: the classification and protocol section above (plus [references/protocol-modes.md](references/protocol-modes.md)),
355
+ then `../code-refactoring/references/finding-schema.md` for shared field semantics.
356
+
357
+ **Native finding → schema mapping:** each “Highest-impact refactor” item → `id` as
358
+ `RF-api-<nnn>`; the item's compatibility label → `rung` verbatim (`additive` | `risky` |
359
+ `breaking`); consumer impact → `title` and `proposal` (imperative); cited routes/lines →
360
+ `evidence` as `{file, line}` entries inside scope; `Verification` items → `verify`; the
361
+ “Proposed contract” snippets stay in `proposal`; new findings start at `status: open`.
362
+
363
+ **Severity mapping (design §5):** a `breaking` change already shipped, or contract ambiguity that
364
+ corrupts data → `P1`; a `risky` inconsistency across ≥2 endpoints → `P2`; `additive` cleanups →
365
+ `P3`; naming/docs → `P4`.
366
+
367
+ **Preserved-behavior inventory (required before proposals):** emit the consumer contract —
368
+ audience (public/partner/internal/service-to-service), endpoints/operations in scope, existing
369
+ clients that must remain compatible, and the compatibility promise — before the first finding.
370
+
371
+ **Preservation class (reuses the native classification):** `additive` → `preserving`; `risky`
372
+ and `breaking` → `breaking`; removal of an endpoint, field, or code path with callers →
373
+ `cutting` even when the change reads additive to remaining consumers.
374
+
375
+ **Stop rules:** an established style guide or public compatibility promise is a constraint, not
376
+ suggestion; `cutting`/`breaking` is never below `P2` and never `fix_eligibility: auto`; risky and
377
+ breaking changes carry a migration strategy; anything outside the scope path is not a finding.
@@ -77,3 +77,34 @@ For mutation races, prefer explicit optimistic concurrency such as ETags / `If-M
77
77
  - Make consumers tolerant of additive fields.
78
78
  - Do not use events as disguised synchronous RPC responses when the caller needs an immediate result.
79
79
 
80
+ ## CLI mode
81
+
82
+ A command-line interface is a contract surface with consumers (scripts, other agents, CI), not a
83
+ collection of convenience shortcuts. Review it with the same compatibility discipline as REST or
84
+ gRPC. Spur's own `apps/cli` is the first target.
85
+
86
+ **Noun/verb grammar:** keep one noun per domain and verbs per action (`spur task show`, not
87
+ `spur showTaskForTask`). Do not introduce a second grammar for the same concept — one spelling per
88
+ noun, one verb per operation, consistent object order.
89
+
90
+ **Flag vocabulary consistency:** shared flags (`--json`, `--scope`, `--agent`, `--fix`) keep the
91
+ same name, arity, and semantics everywhere they appear. A flag that means something new per
92
+ command is a defect; declare a new flag instead.
93
+
94
+ **Exit codes:** `0` = success, nonzero = failure, deterministic and machine-checkable. Never
95
+ swallow failures into `0`; never return nonzero for advisory output. Validation errors and
96
+ runtime failures should be distinguishable from output where practical.
97
+
98
+ **`--json` envelope stability:** `--json` output is a public schema. Add fields additively; never
99
+ remove or rename existing fields, never change a field's type, and emit no human decoration
100
+ (banners, progress text) on the JSON stream. Machine consumers parse the documented envelope only.
101
+
102
+ **Help-text parity:** every accepted flag appears in `--help` with its real arity and default;
103
+ every documented example runs as printed. A flag that works but is not documented, or documented
104
+ but rejected, is a parity break.
105
+
106
+ **Additive vs breaking:** adding a noun, verb, flag, or output field is additive. Renaming or
107
+ removing any of them, re-purposing a flag, changing a default, or altering existing output shape
108
+ is breaking — it requires a migration path and a deprecation window, and it never rides in a
109
+ patch release.
110
+
@@ -1,6 +1,18 @@
1
1
  ---
2
2
  name: taste-refactoring-architect
3
3
  description: Review, simplify, and refactor system architecture toward minimum sufficient architecture while preserving required capabilities, quality attributes, delivery safety, and data invariants. Backs architectural refactoring and boundary reviews.
4
+ license: Apache-2.0
5
+ metadata:
6
+ author: spur
7
+ version: "1.0"
8
+ platforms: "claude-code,codex,openclaw,opencode,antigravity"
9
+ category: execution
10
+ interactions:
11
+ - technique
12
+ operations:
13
+ - refactor-architecture
14
+ openclaw:
15
+ emoji: "🏛️"
4
16
  ---
5
17
 
6
18
  # taste-refactoring-architect
@@ -469,3 +481,34 @@ Do not block on perfect documentation. Produce:
469
481
  - decisions that can safely proceed now.
470
482
 
471
483
  Use A7 DEFER/OBSERVE when evidence is too weak for a structural change.
484
+
485
+ ## Spur contract
486
+
487
+ Machine-facing adapter for `sp:code-refactoring` dispatch (feature H13). Everything above is the
488
+ lens's native output and is unchanged; this section only maps it to the shared finding schema.
489
+
490
+ **Inputs received:** a scope path, an optional change description, and the `architect` focus.
491
+ Read first: the intervention ladder and workflow above, then
492
+ `../code-refactoring/references/finding-schema.md` for shared field semantics.
493
+
494
+ **Native finding → schema mapping:** `ID` → `id` as `RF-architect-<nnn>`; `Action` (A0–A7) →
495
+ `rung` verbatim; `Observation`/`Why it matters` → `title` plus `proposal` (imperative); `Evidence`
496
+ → `evidence` as `{file, line}` entries inside scope; `Fitness functions` → `verify`; new findings
497
+ start at `status: open`. ADR candidates and migration plans stay prose findings with
498
+ `fix_eligibility: suggest`.
499
+
500
+ **Severity mapping (design §5):** any axis ≤1 with a correctness/safety consequence → `P1`; axis
501
+ ≤2 or A5–A7 seam problems → `P2`; A3–A4 → `P3`; A0–A2, ADR candidates, deferred questions → `P4`.
502
+
503
+ **Preserved-behavior inventory (required before proposals):** emit the Preservation Contract
504
+ table — capabilities, quality attributes, data invariants, and their thresholds — for everything
505
+ in scope before the first finding.
506
+
507
+ **Preservation class:** A0/A3/A4 mechanical consolidations with identical behavior →
508
+ `preserving`; A1/A2 removal of a live service, endpoint, queue, or code path with callers →
509
+ `cutting`; A5–A7 seam or behavior changes → `breaking`; migration plans and ADR candidates →
510
+ `preserving` prose with `fix_eligibility: suggest`.
511
+
512
+ **Stop rules:** no removal without dependency/traffic/contract evidence; `cutting`/`breaking` is
513
+ never below `P2` and never `fix_eligibility: auto`; multi-task migrations are `suggest`, applied
514
+ only after an explicit operator answer; anything outside the scope path is not a finding.