@gobing-ai/spur 0.3.63 → 0.3.65

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/config/corpus-baseline.json +2995 -6832
  3. package/config/rules/boundary/sp-runtime-path.yaml +45 -35
  4. package/config/rules/surface/check-cli-surface.yaml +7 -5
  5. package/config/workflows/history-anatomy.yaml +39 -8
  6. package/package.json +9 -9
  7. package/plugins/sp/README.md +17 -10
  8. package/plugins/sp/agents/expert-spur.md +20 -4
  9. package/plugins/sp/commands/dev-find-issue.md +5 -2
  10. package/plugins/sp/commands/dev-gitmsg.md +12 -6
  11. package/plugins/sp/commands/dev-gtd.md +8 -19
  12. package/plugins/sp/commands/dev-idea.md +1 -1
  13. package/plugins/sp/commands/dev-plan.md +1 -1
  14. package/plugins/sp/commands/dev-review-session.md +36 -0
  15. package/plugins/sp/commands/dev-run.md +2 -2
  16. package/plugins/sp/commands/dev-runall.md +2 -2
  17. package/plugins/sp/commands/dev-wrap.md +5 -6
  18. package/plugins/sp/commands/dev-wrapall.md +5 -7
  19. package/plugins/sp/plugin.json +1 -1
  20. package/plugins/sp/references/environment-lens.md +4 -2
  21. package/plugins/sp/references/roles.md +8 -5
  22. package/plugins/sp/scripts/history-anatomy-cache.mjs +3 -2
  23. package/plugins/sp/scripts/history-anatomy-cache.ts +7 -2
  24. package/plugins/sp/skills/dogfood-testing/SKILL.md +22 -1
  25. package/plugins/sp/skills/history-anatomy/references/report-contract.md +8 -0
  26. package/plugins/sp/skills/next-router/SKILL.md +4 -4
  27. package/plugins/sp/skills/pr-reviewing/SKILL.md +2 -3
  28. package/plugins/sp/skills/redesign-web-ui/SKILL.md +184 -0
  29. package/plugins/sp/skills/redesign-web-ui/references/audit-checklist.md +121 -0
  30. package/plugins/sp/skills/redesign-web-ui/references/upgrade-techniques.md +66 -0
  31. package/plugins/sp/skills/session-review/SKILL.md +106 -0
  32. package/plugins/sp/skills/spur-cli/references/agent.md +1 -1
  33. package/plugins/sp/skills/spur-cli/references/workflows/authoring-workflows.md +5 -0
  34. package/plugins/sp/skills/spur-cli/references/workflows/operations.md +20 -5
  35. package/plugins/sp/skills/spur-cli/references/workflows/workflow-fit-and-tuning.md +230 -0
  36. package/plugins/sp/skills/spur-cli/references/workflows.md +27 -5
  37. package/plugins/sp/skills/spur-dev/references/cross-cutting.md +22 -30
  38. package/plugins/sp/skills/spur-dev/references/dev-operations.md +73 -28
  39. package/plugins/sp/skills/spur-dev/references/execution-workflow.md +1 -1
  40. package/plugins/sp/skills/spur-dev/references/flag-glossary.md +18 -8
  41. package/plugins/sp/skills/spur-dev/references/inline-pipeline-driver.md +15 -12
  42. package/spur.js +1152 -816
  43. package/web/_astro/{BoardApp.CKolAjUz.js → BoardApp.BEtcJqde.js} +62 -62
  44. package/web/_astro/BoardApp.DBEin4N5.js +1 -0
  45. package/web/_astro/{TaskDetail.Bre7G4gC.js → TaskDetail.ClAbCXom.js} +1 -1
  46. package/web/_astro/arc.CCvf51_y.js +1 -0
  47. package/web/_astro/architectureDiagram-3BPJPVTR.C0cb0J5M.js +36 -0
  48. package/web/_astro/{blockDiagram-GPEHLZMM.CCGHeRVi.js → blockDiagram-GPEHLZMM.CIyjqoCE.js} +4 -4
  49. package/web/_astro/{c4Diagram-AAUBKEIU.CpqewGmd.js → c4Diagram-AAUBKEIU.fs14IuFs.js} +1 -1
  50. package/web/_astro/channel.BGn_DUCD.js +1 -0
  51. package/web/_astro/chunk-2J33WTMH.CaBKv4ZO.js +1 -0
  52. package/web/_astro/{chunk-4BX2VUAB.ifGXoUA3.js → chunk-4BX2VUAB.BOllTPto.js} +1 -1
  53. package/web/_astro/chunk-55IACEB6.ChEof0O4.js +1 -0
  54. package/web/_astro/{chunk-727SXJPM.DzPE41OS.js → chunk-727SXJPM.Co2kdjD8.js} +2 -2
  55. package/web/_astro/{chunk-AQP2D5EJ.UF2QRXYF.js → chunk-AQP2D5EJ.SWmfcnog.js} +2 -2
  56. package/web/_astro/{chunk-FMBD7UC4.D2zXKa1R.js → chunk-FMBD7UC4.rDAFifF3.js} +1 -1
  57. package/web/_astro/chunk-ND2GUHAM.BCnoXKCw.js +1 -0
  58. package/web/_astro/{chunk-QZHKN3VN.nUKFBiLD.js → chunk-QZHKN3VN.RSmy2hDO.js} +1 -1
  59. package/web/_astro/{classDiagram-4FO5ZUOK.DZg9K9mO.js → classDiagram-4FO5ZUOK.Be7PEfrX.js} +1 -1
  60. package/web/_astro/{classDiagram-v2-Q7XG4LA2.DZg9K9mO.js → classDiagram-v2-Q7XG4LA2.Be7PEfrX.js} +1 -1
  61. package/web/_astro/{client.CdFpTatq.js → client.yhYJvxCU.js} +1 -1
  62. package/web/_astro/cose-bilkent-S5V4N54A.BkUp2aSK.js +1 -0
  63. package/web/_astro/cynefin-OW5HDTMX.BegGGlUV.js +166 -0
  64. package/web/_astro/cytoscape.esm.DzSz-X2X.js +321 -0
  65. package/web/_astro/dagre-BM42HDAG.BkUdjsaC.js +4 -0
  66. package/web/_astro/{defaultLocale.CrowFXzY.js → defaultLocale.DX6XiGOO.js} +1 -1
  67. package/web/_astro/diagram-2AECGRRQ.E9vugt3-.js +43 -0
  68. package/web/_astro/diagram-5GNKFQAL.Dj4yeHXB.js +10 -0
  69. package/web/_astro/diagram-KO2AKTUF.Buaquwli.js +3 -0
  70. package/web/_astro/diagram-LMA3HP47.BV3dgGgm.js +24 -0
  71. package/web/_astro/diagram-OG6HWLK6.Cnx3s-tc.js +24 -0
  72. package/web/_astro/{erDiagram-TEJ5UH35.CF2U-pQZ.js → erDiagram-TEJ5UH35.DKK_abu4.js} +3 -3
  73. package/web/_astro/{flowDiagram-I6XJVG4X.BJK4M3in.js → flowDiagram-I6XJVG4X.BNuu9fbm.js} +4 -4
  74. package/web/_astro/ganttDiagram-6RSMTGT7.b16KUMjy.js +292 -0
  75. package/web/_astro/gitGraphDiagram-PVQCEYII.Kh41lbG5.js +106 -0
  76. package/web/_astro/{graph.D2o_JWn5.js → graph.-OzhPTMs.js} +1 -1
  77. package/web/_astro/{index.A0eX93qW.js → index.De90oHcH.js} +1 -1
  78. package/web/_astro/infoDiagram-5YYISTIA.DEWBXkp-.js +2 -0
  79. package/web/_astro/{ishikawaDiagram-YF4QCWOH.BmWDZtwF.js → ishikawaDiagram-YF4QCWOH.DiAdmcL6.js} +4 -4
  80. package/web/_astro/{journeyDiagram-JHISSGLW.CpB1YWDP.js → journeyDiagram-JHISSGLW.D1Ki7IRm.js} +1 -1
  81. package/web/_astro/{kanban-definition-UN3LZRKU.k-fukQX9.js → kanban-definition-UN3LZRKU.CWUhrQpc.js} +21 -21
  82. package/web/_astro/layout.owoKPs3z.js +1 -0
  83. package/web/_astro/linear.BaFsgcCe.js +1 -0
  84. package/web/_astro/{mermaid.core.DnpzzuPU.js → mermaid.core.CHw_AsGy.js} +5 -5
  85. package/web/_astro/{mindmap-definition-RKZ34NQL.D9NnlLBu.js → mindmap-definition-RKZ34NQL.UIhghgmN.js} +8 -8
  86. package/web/_astro/ordinal.DBvzRdQf.js +1 -0
  87. package/web/_astro/pieDiagram-4H26LBE5.D05l3JUA.js +30 -0
  88. package/web/_astro/{quadrantDiagram-W4KKPZXB.BkAygRlm.js → quadrantDiagram-W4KKPZXB.BcWIhIcE.js} +1 -1
  89. package/web/_astro/{requirementDiagram-4Y6WPE33.CX8ibmwc.js → requirementDiagram-4Y6WPE33.B1rYvKGn.js} +1 -1
  90. package/web/_astro/sankeyDiagram-5OEKKPKP.CKylVRC4.js +40 -0
  91. package/web/_astro/{sequenceDiagram-3UESZ5HK.veO8c2tk.js → sequenceDiagram-3UESZ5HK.Dm3uA_s4.js} +3 -3
  92. package/web/_astro/stateDiagram-AJRCARHV.Bgca_BLe.js +1 -0
  93. package/web/_astro/stateDiagram-v2-BHNVJYJU.C1T7YFrG.js +1 -0
  94. package/web/_astro/{timeline-definition-PNZ67QCA.DayoPp_2.js → timeline-definition-PNZ67QCA.ZOHJn3Sn.js} +4 -4
  95. package/web/_astro/vennDiagram-CIIHVFJN.DegZitjD.js +34 -0
  96. package/web/_astro/{wardleyDiagram-YWT4CUSO.lIAjSkZJ.js → wardleyDiagram-YWT4CUSO.BDsC115d.js} +1 -1
  97. package/web/_astro/{xychartDiagram-2RQKCTM6.D8_2K6U1.js → xychartDiagram-2RQKCTM6.D0MO70ea.js} +4 -4
  98. package/web/index.html +1 -1
  99. package/web/_astro/BoardApp.DXD--ybM.js +0 -1
  100. package/web/_astro/arc.7luwOGiC.js +0 -1
  101. package/web/_astro/architectureDiagram-3BPJPVTR.F6KaHXp-.js +0 -36
  102. package/web/_astro/channel.DxfOFf1l.js +0 -1
  103. package/web/_astro/chunk-2J33WTMH.CB9vKa5F.js +0 -1
  104. package/web/_astro/chunk-55IACEB6.VIaRo7l8.js +0 -1
  105. package/web/_astro/chunk-ND2GUHAM.Cb9bDyvx.js +0 -1
  106. package/web/_astro/cose-bilkent-S5V4N54A.D9STo90d.js +0 -1
  107. package/web/_astro/cytoscape.esm.D3_iZ_3b.js +0 -321
  108. package/web/_astro/dagre-BM42HDAG.D3IbwhHz.js +0 -4
  109. package/web/_astro/diagram-2AECGRRQ.BWTDxBe9.js +0 -43
  110. package/web/_astro/diagram-5GNKFQAL.XnXlonHG.js +0 -10
  111. package/web/_astro/diagram-KO2AKTUF.CywOngCO.js +0 -3
  112. package/web/_astro/diagram-LMA3HP47.DY1D21Iu.js +0 -24
  113. package/web/_astro/diagram-OG6HWLK6.DQxb59KA.js +0 -24
  114. package/web/_astro/ganttDiagram-6RSMTGT7.BSniMzdB.js +0 -292
  115. package/web/_astro/gitGraphDiagram-PVQCEYII.Dxg-yRov.js +0 -106
  116. package/web/_astro/infoDiagram-5YYISTIA.BY4CgO_n.js +0 -2
  117. package/web/_astro/layout.DNLMvjEt.js +0 -1
  118. package/web/_astro/linear.BNNCobvI.js +0 -1
  119. package/web/_astro/ordinal.BYWQX77i.js +0 -1
  120. package/web/_astro/pieDiagram-4H26LBE5.CKhoMiyC.js +0 -30
  121. package/web/_astro/sankeyDiagram-5OEKKPKP.Dvurpa0Y.js +0 -40
  122. package/web/_astro/stateDiagram-AJRCARHV.DpMr4CO3.js +0 -1
  123. package/web/_astro/stateDiagram-v2-BHNVJYJU.CKjso86_.js +0 -1
  124. package/web/_astro/vennDiagram-CIIHVFJN.XNHf04O9.js +0 -34
  125. package/web/_astro/wardley-L42UT6IY.CpM_031g.js +0 -161
@@ -0,0 +1,66 @@
1
+ # Upgrade techniques
2
+
3
+ Optional moves for `sp:redesign-web-ui` Plan/Apply. Load this file only when a Diagnose finding
4
+ needs a technique from it.
5
+
6
+ Every technique here is **polish**. Accessibility and token authority always win. Honor
7
+ `prefers-reduced-motion: reduce` by disabling or replacing motion with an instant state change.
8
+
9
+ Product/app chrome (dashboards, settings, editors) stays restrained. Marketing/landing pages can
10
+ spend more on one signature moment. Spend boldness in **one** place.
11
+
12
+ ---
13
+
14
+ ## When a technique is in play
15
+
16
+ Use a technique when:
17
+
18
+ - The finding is `polish`, and
19
+ - The page type supports it (landing/marketing, or a single product moment), and
20
+ - The authority layer (`DESIGN.md` / theme) does not forbid it.
21
+
22
+ Skip the technique when:
23
+
24
+ - It requires a new animation library or a CSS-framework migration.
25
+ - It hijacks scroll (inertia/custom scrollbar physics) on a product UI.
26
+ - It would be the second signature moment on the same page.
27
+
28
+ ---
29
+
30
+ ## Typography
31
+
32
+ - **Variable-font weight/width** on hover or a short scroll range — one word or one heading, not every line.
33
+ - **Outlined-to-fill** on a display line that is the page's thesis.
34
+ - **Text as mask** only when a real, owned video/image sits behind it.
35
+
36
+ ## Layout
37
+
38
+ - **Broken grid / overlap** — one element bleeds or overlaps on purpose; the rest stay on the grid.
39
+ - **Whitespace maximization** — one block gets aggressive negative space so it is the only focus.
40
+ - **Sticky stack** — sections pin and stack on scroll on a marketing long-scroll, not inside app chrome.
41
+ - **Split-screen scroll** — two panes moving in opposition; marketing only, and never the only way to reach content.
42
+
43
+ ## Motion
44
+
45
+ - **Staggered entry** — small Y + opacity cascade on first paint of a group (40–80ms steps).
46
+ - **Spring on press** — interactive controls, not page load.
47
+ - **Scroll-driven reveal** — mask, wipe, or SVG draw tied to scroll *progress*, with a reduced-motion static end-state.
48
+
49
+ Do not add smooth-scroll inertia, scrolljacking, or a custom scrollbar on product UI. Native
50
+ `scroll-behavior: smooth` plus a reduced-motion instant fallback is the ceiling unless the operator
51
+ asks for more.
52
+
53
+ ## Surfaces
54
+
55
+ - **Glass** — `backdrop-filter` plus a 1px inner border and a faint inner shadow; only over content
56
+ that remains readable.
57
+ - **Spotlight border** — cursor-tracking edge light on one featured card.
58
+ - **Grain overlay** — `pointer-events: none`, fixed, very low contrast; skip on dense data tables.
59
+ - **Tinted shadows** — shadow hue matches the surface, not generic black.
60
+
61
+ ---
62
+
63
+ ## Signature test
64
+
65
+ After picking techniques, keep one memorable moment. If two techniques compete, drop the weaker
66
+ one. The surrounding UI stays quiet so the signature can read.
@@ -0,0 +1,106 @@
1
+ ---
2
+ name: session-review
3
+ description: "Review the active coding-agent session, distinguish resolved and open issues with evidence, and propose bounded improvements. Triggers: review this session, session wrap-up, immediate retrospective, what happened, what was resolved."
4
+ license: Apache-2.0
5
+ version: 1.0.0
6
+ metadata:
7
+ author: spur
8
+ platforms: "claude-code,codex,openclaw,opencode,antigravity,pi"
9
+ category: analysis-core
10
+ interactions:
11
+ - reviewer
12
+ see_also:
13
+ - sp:history-anatomy
14
+ - sp:indexed-context
15
+ - sp:code-verification
16
+ ---
17
+
18
+ # sp:session-review — Active Session Review
19
+
20
+ Review the active coding-agent session while its conversation context is still available. Produce a
21
+ compact, evidence-backed report of outcomes, resolved issues, remaining risks, and improvements.
22
+
23
+ ## When to use
24
+
25
+ Use this skill immediately after focused operations when the operator asks what happened, what was
26
+ resolved, or how the session could improve. Use imported-history analysis for ended sessions,
27
+ cross-agent windows, recurrence, trends, or quantitative performance forensics.
28
+
29
+ ## Arguments
30
+
31
+ | Argument | Description | Default |
32
+ | --- | --- | --- |
33
+ | `[focus]` | Question or operation to emphasize. It changes ordering, not evidence collection. | full active session |
34
+
35
+ ## Evidence boundary
36
+
37
+ - Treat the active conversation as the primary evidence plane.
38
+ - Verify each material claim with source evidence from the conversation or a read-only repository
39
+ check, and cite the exact result or path in the report.
40
+ - Use read-only repository checks only when they confirm a material claim; prefer existing tool
41
+ results already visible in the session over rerunning commands.
42
+ - Mark an issue **resolved** only when the session shows the symptom, root cause, applied resolution,
43
+ and verification evidence. Otherwise classify it as open or attempted.
44
+ - Separate observation from inference. Label an unsupported causal explanation as a hypothesis and
45
+ name the confirmation needed.
46
+ - State `not available` when compaction or missing output removed evidence. Never reconstruct it from
47
+ memory or claim a verification that did not run.
48
+
49
+ ## Protocol
50
+
51
+ 1. **Resolve scope.** Review the active session from the operator's initiating request through the
52
+ latest result. Use `[focus]` only to rank relevant material.
53
+ 2. **Inventory outcomes.** List requested outcomes and classify each as completed, partial, blocked,
54
+ or not attempted. Collapse repeated attempts into one outcome.
55
+ 3. **Classify issues.** For every material issue, distinguish resolved, open, or attempted. Record
56
+ root cause only when the evidence boundary supports it.
57
+ 4. **Select improvements.** Keep at most three changes that would prevent meaningful recurrence.
58
+ Apply the placement rule in
59
+ [the environment-improvement mapping](../../references/environment-lens.md): automate with a
60
+ check when possible, place coding standards on the review path, and keep always-loaded steering
61
+ as navigation pointers.
62
+ 5. **Render the report.** Use the exact compact output contract below. Omit empty table rows, not
63
+ headings; write `None observed` when a section has no supported entry.
64
+
65
+ ## Output contract
66
+
67
+ ### Outcome
68
+
69
+ State the overall result in one to three sentences, including partial or blocked scope.
70
+
71
+ ### Resolved issues
72
+
73
+ | Issue | Root cause | Resolution | Evidence |
74
+ | --- | --- | --- | --- |
75
+
76
+ Evidence names the tool result, verification command, or concrete repository state visible in the
77
+ session. Do not list ordinary implementation steps as issues.
78
+
79
+ ### Open issues and risks
80
+
81
+ | Issue or risk | State | Evidence or confirmation needed |
82
+ | --- | --- | --- |
83
+
84
+ ### Process and environment improvements
85
+
86
+ For each supported proposal, name its owner surface, expected impact, verification method, and
87
+ reversibility. Proposals remain report-only: apply no change and create no task.
88
+
89
+ ### Next actions
90
+
91
+ List only actions needed to finish partial scope, confirm a hypothesis, or preserve a demonstrated
92
+ improvement. Use `None` when the session is complete and no follow-up is justified.
93
+
94
+ ## Boundaries
95
+
96
+ - Stay in the active host session. Do not delegate; a fresh context loses the evidence being reviewed.
97
+ - Do not launch a workflow, import history, create or update corpus items, or edit files.
98
+ - Do not append indexed-context memory.
99
+ - Do not perform baseline comparison, recurrence classification, cache publication, or a twelve-section
100
+ forensic report; those belong to imported-history analysis.
101
+ - Do not turn a single low-impact observation into a new policy. Report it as a candidate until it
102
+ recurs or demonstrates a high-impact contract violation.
103
+
104
+ ## Platform notes
105
+
106
+ On platforms without slash commands, invoke this skill directly before ending the active session.
@@ -48,7 +48,7 @@ through a coding agent as an external process, producing a persisted run record
48
48
 
49
49
  | Flag | Purpose |
50
50
  | ------ | --------- |
51
- | `--agent <name>` | Role, executor, agent binary, `auto`, or `inline`. A **role** (`scribe`/`coder`/`reviewer`/`planner`, from `plugins/sp/references/roles.md`) selects the starting tier; an **executor** (an `agent.executors` entry) is a permanent pin; a **bare binary name** works with a one-time warning (transition shim); `auto` uses the declared/default role. **`inline` is host-session-only** (G5 / ADR-047 amendment): `agent run` is a headless subprocess surface, so explicit `inline` is rejected with exit 2 and a stable error messageit never normalizes to `agent.default`. |
51
+ | `--agent <name>` | Role, executor, agent binary, `auto`, or `inline`. A **role** (`scribe`/`coder`/`reviewer`/`planner`, from `plugins/sp/references/roles.md`) selects the starting tier; an **executor** (an `agent.executors` entry) is a permanent pin; a **bare binary name** works with a one-time warning (transition shim); `auto` uses the declared/default role. `inline` is the default selector (0687 / ADR-087): on a host session the work runs in-session; on a headless subprocess surface like `agent run` it substitutes tier resolution and warns once naming the resolved executor no rejection, no `agent.default` normalization. |
52
52
  | `--model <name>` | Agent model argument (e.g. `o3`, `sonnet`). Passed through to the agent's model flag. |
53
53
  | `--mode <mode>` | Agent output mode: `text` or `json`. |
54
54
  | `--continue` | Resume the previous agent session instead of starting fresh. |
@@ -218,9 +218,14 @@ verifying shape, or stand in a `note` action until the path is proven, then swap
218
218
 
219
219
  ## Authoring checklist
220
220
 
221
+ - [ ] Fit gate cleared before the mode gate — replay + machine-checkable branch + durable record
222
+ ([workflow-fit-and-tuning.md](workflow-fit-and-tuning.md#1-fit-gate--workflow-or-prose)).
221
223
  - [ ] Mode chosen deliberately (loop → state-machine; pipeline → transition-flow); recorded the reason.
222
224
  - [ ] `kind: transition-flow` set for flows; `$schema` quoted.
223
225
  - [ ] Initial + terminal states/nodes declared; every transition/edge target exists.
224
226
  - [ ] Guards/conditions ordered so the specific case precedes the fallback.
225
227
  - [ ] `iterationBound` set for any loop; `env.allow` lists every `${env.X}` used.
228
+ - [ ] Every node inside the simplicity budget — `shell` at or under 5 non-comment units,
229
+ `agent.run` input referencing a command rather than a raw prompt, guards a single predicate
230
+ ([workflow-fit-and-tuning.md](workflow-fit-and-tuning.md#3-node-simplicity-budget)).
226
231
  - [ ] Validates clean AND dry-run reaches the expected terminal state.
@@ -161,22 +161,31 @@ Turn a described process into a validated, dry-run-verified workflow in the righ
161
161
  1. **Clarify intent** — restate the process as one or two sentences: the steps, the success terminal,
162
162
  the loop/branch points. If ambiguous (which step retries? what ends it?), state the interpretation
163
163
  taken.
164
- 2. **Mode-selection gate** — run [mode-selection gate](#sub-procedure-mode-selection-gate). This is
164
+ 2. **Fit gate** — run the three-part test in
165
+ [workflow-fit-and-tuning.md](workflow-fit-and-tuning.md#the-three-part-test) *before* the mode
166
+ gate: the process earns a workflow only if it replays, branches on a machine-checkable predicate,
167
+ and needs a durable per-run record. Fewer than three → recommend a descriptive procedure /
168
+ checklist instead and stop; do not author YAML the process will not use.
169
+ 3. **Mode-selection gate** — run [mode-selection gate](#sub-procedure-mode-selection-gate). This is
165
170
  mandatory and gating: surface the recommended mode with its reason and the rejected alternative,
166
171
  and **confirm before authoring**. Honor an explicit `--kind`.
167
- 3. **Reconcile against existing workflows** — run [find-existing-workflow](#sub-procedure-find-existing-workflow).
172
+ 4. **Reconcile against existing workflows** — run [find-existing-workflow](#sub-procedure-find-existing-workflow).
168
173
  If the process is already covered, **stop and hand to refine (or extend the existing flow) on
169
174
  confirmation** — do not author a redundant workflow. Only the "no real match" / "add-new" branch
170
175
  proceeds to author below.
171
- 4. **Author the YAML** — use the **real schema shape** for the chosen mode
176
+ 5. **Author the YAML** — use the **real schema shape** for the chosen mode
172
177
  ([authoring-workflows.md → Per-mode shapes](authoring-workflows.md#per-mode-shapes)). Set `name`,
173
178
  `description` (the WHY), the initial + terminal states/nodes, the steps with their actions, the
174
179
  transitions/edges with guards/conditions in the right declaration order, `iterationBound` for any
175
180
  loop, `env.allow` for any `${env.X}`, and a quoted `$schema`. For transition-flow, set
176
181
  `kind: transition-flow`.
177
- 5. **Place the file** default `.spur/workflows/<name>.yaml` (a `--file` arg overrides), named for
182
+ Keep each node inside the simplicity budget
183
+ ([workflow-fit-and-tuning.md](workflow-fit-and-tuning.md#3-node-simplicity-budget)): `shell`
184
+ commands at or under 5 non-comment units, `agent.run` inputs referencing a slash command rather
185
+ than carrying a raw prompt, guards a single predicate.
186
+ 6. **Place the file** — default `.spur/workflows/<name>.yaml` (a `--file` arg overrides), named for
178
187
  what the workflow does.
179
- 6. **Verify** — run the [validate-and-dry-run core](#sub-procedure-validate-and-dry-run) with the
188
+ 7. **Verify** — run the [validate-and-dry-run core](#sub-procedure-validate-and-dry-run) with the
180
189
  expected terminal state. Not done until the definition validates AND the dry-run reaches it.
181
190
 
182
191
  Output contract: YAML workflow content + chosen mode + reason + destination path + validate result +
@@ -197,6 +206,12 @@ Adjust an existing workflow with the smallest change that meets the intent. Proc
197
206
  - missing step → add a state/node + its transition/edge in the correct declaration order
198
207
  - runaway loop → set or raise `iterationBound`
199
208
  - missing variable/env → add to `vars` or `env.allow`
209
+ - too slow / unreadable trace (not a correctness bug) → the
210
+ [optimize procedure](workflow-fit-and-tuning.md#optimize--refine-an-accepted-workflow-in-place),
211
+ backed by a before/after `spur workflow trace <run-id> --json` pair
212
+ - the whole file is the wrong surface (all nodes raw `agent.run`, no branching, edited more than
213
+ run) → not a refine: the
214
+ [demote procedure](workflow-fit-and-tuning.md#demote--workflow--descriptive-procedure)
200
215
  See [authoring-workflows.md](authoring-workflows.md) for each mechanism's real shape. **Do not
201
216
  switch mode** in a refine — a mode change is a rewrite; hand it back to `add`.
202
217
  3. **Apply the smallest change.** Preserve declaration order semantics (the first passing
@@ -0,0 +1,230 @@
1
+ ---
2
+ name: workflow-fit-and-tuning
3
+ description: Decide whether a process should be a spur workflow at all, tune an accepted one for latency and observability, keep its nodes inside the simplicity budget, and refactor across the boundary — promote a descriptive procedure into YAML, demote a YAML back to prose, or optimize one in place.
4
+ see_also:
5
+ - spur-cli
6
+ - operations
7
+ - authoring-workflows
8
+ ---
9
+
10
+ # Workflow Fit & Tuning
11
+
12
+ Three questions that sit **before and around** the mode gate:
13
+
14
+ 1. **Fit** — should this be a `spur workflow` at all, or a descriptive procedure / checklist?
15
+ 2. **Tuning** — given it is one, how is it configured for low latency and legible traces?
16
+ 3. **Refactor** — how do you move a process across that boundary in either direction?
17
+
18
+ **Polaris:** get things done with high efficiency, low latency, and observability. A workflow that
19
+ does not beat prose on all three is prose with a YAML tax.
20
+
21
+ ---
22
+
23
+ ## 1. Fit gate — workflow or prose?
24
+
25
+ Run this **before** the [mode-selection gate](operations.md#sub-procedure-mode-selection-gate). The
26
+ mode gate answers *which kind of workflow*; it presumes an answer to *whether* — and that
27
+ presumption is the more expensive one to get wrong.
28
+
29
+ ### What each side actually buys
30
+
31
+ | | `spur workflow` YAML | Descriptive procedure (skill reference, slash command, checklist) |
32
+ | --- | --- | --- |
33
+ | Buys you | Durable run record, resumable HITL pause, bounded retry loops, machine-readable terminal status, `trace`/`--follow`, `cancel`/`clean`, unattended `--async` | Judgment at every step, zero authoring ceremony, edits are one-line, no subprocess per step |
34
+ | Costs you | Authoring + `validate`/`dry-run` upkeep, one subprocess per action node, indirection an agent must read through, a definition that rots when the surface moves | Nothing is replayable, nothing is recorded, "what happened on that run" is unanswerable |
35
+ | Fails at | Steps whose outcome only a reader can judge | Anything that must run the same way twice, unattended |
36
+
37
+ ### The three-part test
38
+
39
+ A process earns a workflow only when **all three** hold:
40
+
41
+ - **Replay** — it runs repeatedly, unattended, or across different operators.
42
+ - **Branch** — at least one step routes on a *machine-checkable* predicate (exit code, JSON field,
43
+ file present), and the run may retry or gate on it.
44
+ - **Record** — someone will later need to answer "what happened on run X" from persisted evidence.
45
+
46
+ Three yes → author the workflow. Two → borderline: prefer prose plus one shell script, and revisit
47
+ when the run count justifies it. One or zero → **descriptive procedure**; say so and stop.
48
+
49
+ ### Signals
50
+
51
+ | Signal in the described process | Read as |
52
+ | --- | --- |
53
+ | "Retry until the gate passes", "loop until clean" | workflow — bounded loop is the engine's job |
54
+ | "Pause for approval, continue later / in another session" | workflow — HITL resume needs a run record |
55
+ | "Every night", "for each task in the batch", unattended | workflow — replay + record |
56
+ | "Then check whether it looks right and decide" | prose — the predicate is judgment, not an exit code |
57
+ | "It depends what the diff says" | prose — the branch has no machine-checkable condition |
58
+ | Runs once, then the situation has changed | prose — a checklist, not a definition |
59
+ | One operator, in-session, watching each step | prose — the trace has no second reader |
60
+
61
+ ### The two anti-patterns
62
+
63
+ - **Prose wearing YAML.** Every node is an `agent.run` with a raw prompt, edges are all
64
+ unconditional. The engine adds a process spawn per node and contributes no branching, no gate, no
65
+ retry. This is the ADR-069 R2 measure firing on *every* node — the advisory is telling you the
66
+ whole file is misplaced, not that eight prompts need polish. Demote it.
67
+ - **YAML wearing prose.** A skill reference that says "then run the checks, and if they fail fix and
68
+ re-run, up to three times" — a bounded retry loop written as a paragraph an agent re-interprets
69
+ every session, with no record of how many attempts actually happened. Promote it.
70
+
71
+ ### Hybrid is the usual answer
72
+
73
+ The boundary runs *through* most processes, not around them. Keep judgment in the skill or slash
74
+ command; put the deterministic gate loop in the workflow; let the workflow's judgment steps call the
75
+ command by name. That is the ADR-043 preference restated as an architecture: **the workflow selects
76
+ and orders capabilities, it does not contain them** (ADR-069).
77
+
78
+ ---
79
+
80
+ ## 2. Tuning — latency, efficiency, observability
81
+
82
+ ### Cost model — know what you are spending
83
+
84
+ | Node / element | Real cost | Consequence for design |
85
+ | --- | --- | --- |
86
+ | `agent.run` action | A full agent session: seconds to minutes, plus tokens | **The dominant cost.** Count them; every one you remove is the largest single win available |
87
+ | `shell` action | One `sh -c` subprocess, milliseconds | Cheap individually — but each node is also a persisted transition round-trip |
88
+ | Guard (`kind: shell`) | One subprocess **per evaluation**, re-run on every loop iteration | A loop with a bound of 5 pays its guards 5 times |
89
+ | Loop `iterationBound` | Worst case = bound x per-iteration cost | It is a **latency ceiling**, not only a runaway safety net |
90
+ | HITL pause | Unbounded wall-clock (waits on a human) | Never put one inside a loop body |
91
+
92
+ ### Latency checklist
93
+
94
+ - [ ] **Fewest `agent.run` nodes that still do the work.** Two adjacent judgment steps with no gate
95
+ between them are one judgment step.
96
+ - [ ] **Soft status-file probe over repeated probing.** Run the expensive check once in an action
97
+ that always exits 0 and writes its verdict to a run-scoped file; branch with ordered cheap
98
+ guards that read that file. One subprocess instead of one per branch — the `basic.yaml` and
99
+ `task-pipeline.yaml` quality-gate idiom.
100
+ - [ ] **Order guards cheapest-discriminating-first.** The first passing guard wins, so a `test -f`
101
+ ahead of a `spur … --json` parse skips the expensive call on the common path.
102
+ - [ ] **No guard recomputes what a prior node already wrote to disk.**
103
+ - [ ] **`iterationBound` from the budget, not from optimism.** Set it to the observed maximum plus
104
+ one; a bound of 10 on a 40-second gate is a seven-minute worst case nobody chose.
105
+ - [ ] **Fan out only genuinely independent branches** (`type: parallel`, transition-flow). Serial
106
+ nodes with no data dependency are pure dead latency.
107
+ - [ ] **`--async` when the caller does not need the terminal state inline**, then follow with
108
+ `spur workflow trace <run-id> --follow`.
109
+
110
+ ### Observability is an authoring decision, not a run flag
111
+
112
+ The flags (`--detail`, `--verbose`, `--trace-file`, `--follow`, `--output`) are catalogued in
113
+ [../workflows.md](../workflows.md); they only expose what the definition already made legible.
114
+
115
+ - [ ] **Name states/nodes after the outcome they establish, not the tool they invoke.**
116
+ `quality-gate-passed` reads at 3am; `run-script-2` does not.
117
+ - [ ] **`description` carries the WHY** of the workflow; a one-line comment carries the why of any
118
+ non-obvious guard order or `iterationBound`.
119
+ - [ ] **Every gate writes a machine-readable verdict to a run-scoped file.** It is the guard's input
120
+ *and* the post-mortem's evidence — one artifact, two consumers.
121
+ - [ ] **Declare `failureStates`** so a failed run reports `status: 'failed'` instead of finalizing as
122
+ a `done` run that landed somewhere bad (the 0425 reader contract).
123
+ - [ ] **`env.allow` lists exactly what is used** — no more; an over-broad allowlist is an
124
+ undocumented dependency.
125
+ - [ ] **While tuning:** `--detail full` plus `--trace-file`, and compare traces. **In production:**
126
+ default detail, retained run log.
127
+
128
+ ---
129
+
130
+ ## 3. Node simplicity budget
131
+
132
+ Simplicity is the operating constraint, and it is already measurable — `spur workflow validate`
133
+ reports it. Do not invent a second threshold; author to the one that is frozen (ADR-069, task 0614).
134
+
135
+ | Element | Budget | What breaching it means |
136
+ | --- | --- | --- |
137
+ | `shell` action `command` | **<= 5** non-comment units (split on newline and `;`) | >= 6 flags the composition advisory: the program holds reusable behavior that wants an owner |
138
+ | `agent.run` action `input` | A **slash command or skill invocation** | A raw prose prompt flags: the operation belongs behind a centralized command (ADR-043). Prompt length sets severity only |
139
+ | Transition guard | **One** boolean predicate | Guards are exempt from the shell measure by design. A guard needing five lines is a probe node in disguise — make it one |
140
+ | Node count | Every node earns its transition round-trip | A node that always runs immediately after another, with no guard between them, is one node |
141
+
142
+ **When a node breaches the budget, do not reformat to dodge the measure.** Joining five lines with
143
+ `&&` moves the complexity, not the ownership. Pick an owner from the five recorded options in
144
+ `docs/design/workflow-shell-ownership.md`: public `spur` verb (consent-gated), application service,
145
+ least-privilege built-in action kind, workflow-relative external extension, or a recorded
146
+ stays-shell exception with its reason in `config/workflow-composition-baseline.json`.
147
+
148
+ **Advisory posture is binding.** Composition findings never block a run, never change a `validate`
149
+ exit status, and are never a reason to hot-edit an executing pipeline. Surface them; fix on operator
150
+ acceptance.
151
+
152
+ ---
153
+
154
+ ## 4. Refactor — moving across the boundary
155
+
156
+ Three named directions. All three end in the shared
157
+ [validate-and-dry-run core](operations.md#sub-procedure-validate-and-dry-run) when a YAML definition
158
+ survives the change.
159
+
160
+ ### promote — descriptive procedure → workflow
161
+
162
+ Use only when the [fit gate](#the-three-part-test) clears all three parts. Steps:
163
+
164
+ 1. **Split the prose into spine and judgment.** The spine is every step whose outcome is
165
+ machine-checkable. Everything else stays judgment and does *not* become a node's inline prompt.
166
+ 2. **Name the terminals** — the success terminal, and every failure terminal that deserves its own
167
+ name. Declare them in `failureStates`.
168
+ 3. **Run the [mode-selection gate](operations.md#sub-procedure-mode-selection-gate)** — the spine's
169
+ shape decides it: retry loop → state-machine; forward pipeline → transition-flow.
170
+ 4. **Author one node per spine step.** Each judgment step becomes an `agent.run` whose `input`
171
+ *references the existing slash command or skill* — never a copy of the prose. If no such command
172
+ exists, create it first; a promotion that inlines prompts has produced anti-pattern one.
173
+ 5. **Verify** with validate-and-dry-run against the expected terminal.
174
+ 6. **Rewrite the descriptive doc as the entry point** — it now explains the WHY and delegates to
175
+ `spur workflow run`, rather than restating the steps. Two copies of the procedure is the drift
176
+ the promotion was supposed to end.
177
+
178
+ ### demote — workflow → descriptive procedure
179
+
180
+ Triggers (any two are sufficient; the first alone is sufficient):
181
+
182
+ - Every node is an `agent.run` with a raw prompt and every edge is unconditional.
183
+ - The definition has been edited more often than it has been run.
184
+ - The graph is a straight line — no guard, no gate, no loop.
185
+ - The run count over its lifetime is in the single digits and not growing.
186
+ - Its `iterationBound` has never been reached because nothing ever loops.
187
+
188
+ Steps:
189
+
190
+ 1. **Check for live dependents** — `spur workflow trace --workflow <name> --last 20 --json` for
191
+ recent runs, and grep the shipped surfaces for the filename. A definition another command invokes
192
+ is not demoted unilaterally.
193
+ 2. **Return each node's work to its owner** — judgment nodes to the command or skill they should
194
+ have been calling; a genuinely useful shell sequence to one script under its owning surface.
195
+ 3. **Rewrite the entry surface as the procedure**, in the order the graph ran.
196
+ 4. **Delete the YAML and its `config/workflow-composition-baseline.json` entries** in the same
197
+ change. A baseline entry whose action no longer exists fails the two-sided check.
198
+ 5. **Record the demotion** and its trigger, so the next author does not re-promote it by reflex.
199
+
200
+ ### optimize — refine an accepted workflow in place
201
+
202
+ A [refine](operations.md#refine) whose `--intent` is latency or legibility rather than correctness.
203
+ Dimensions, highest leverage first:
204
+
205
+ 1. **Remove or merge `agent.run` nodes** — the dominant cost, always the first pass.
206
+ 2. **Collapse probe-then-branch into a soft status-file probe** with ordered guards.
207
+ 3. **Reorder guards** cheapest-discriminating-first.
208
+ 4. **Lower `iterationBound`** to the observed maximum plus one.
209
+ 5. **Merge nodes that always run together** with no guard between them.
210
+ 6. **Move any `>= 6`-unit shell program to a recorded owner** (never by reformatting).
211
+ 7. **Rename states/nodes to outcomes** and add the missing `failureStates`.
212
+
213
+ **Measure, do not estimate.** Capture `spur workflow trace <run-id> --json` before and after and
214
+ compare wall-clock per state and the transition sequence. An optimization with no trace pair behind
215
+ it is a preference. Rules that still bind: the smallest change that meets the intent, no mode
216
+ switch inside a refine (that is a rewrite — hand it to `add`), and never edit a workflow that is
217
+ currently executing.
218
+
219
+ ---
220
+
221
+ ## Checklist
222
+
223
+ - [ ] Fit gate run **before** the mode gate; the prose-vs-workflow verdict stated with its reason.
224
+ - [ ] All three of replay / branch / record hold, or the answer was a descriptive procedure.
225
+ - [ ] Judgment steps reference an existing command; none inline a raw prompt.
226
+ - [ ] `agent.run` node count is the minimum the branching requires.
227
+ - [ ] `iterationBound` chosen from a latency budget; guards ordered cheapest-first.
228
+ - [ ] States/nodes named for outcomes; `failureStates` declared; gates write a verdict file.
229
+ - [ ] Every `shell` action within the <= 5-unit budget, or carrying a recorded owner/disposition.
230
+ - [ ] Refactors verified through validate-and-dry-run; optimizations backed by a before/after trace.
@@ -9,12 +9,24 @@ see_also:
9
9
 
10
10
  `spur workflow` runs declarative YAML workflows (powered by `@gobing-ai/ts-dual-workflow-engine`) that
11
11
  orchestrate a multi-step process — an implement→check→fix loop, an import→validate→transform→write
12
- pipeline, an approval gate. The engine executes **two distinct workflow kinds**, and the first act of
13
- any new workflow is choosing between them. Get the mode right and the rest of authoring follows the
12
+ pipeline, an approval gate. The engine executes **two distinct workflow kinds**, and choosing between
13
+ them is the first act of *authoring*. Get the mode right and the rest of authoring follows the
14
14
  schema; get it wrong and you fight the engine.
15
15
 
16
- Operating a workflow well is a full lifecycle — choose the mode, author the YAML, validate it, run it,
17
- read the trace, and refine. This skill covers all of it.
16
+ Operating a workflow well is a full lifecycle — decide it should be a workflow at all, choose the mode,
17
+ author the YAML, validate it, run it, read the trace, tune it, and refine. This skill covers all of it.
18
+
19
+ ## Decide it is a workflow at all — before the mode
20
+
21
+ The mode gate presumes the process belongs in YAML. That presumption is the more expensive one to get
22
+ wrong: a workflow whose nodes are all raw `agent.run` prompts is a descriptive procedure paying a
23
+ process spawn per step, and a bounded retry loop written as a paragraph in a skill reference is a
24
+ workflow nobody authored. Run the **fit gate** first — a process earns a workflow only when it
25
+ replays, branches on a machine-checkable predicate, **and** needs a durable per-run record.
26
+
27
+ Fit gate, cost model, latency/observability tuning, the node simplicity budget, and the
28
+ promote / demote / optimize refactor procedures:
29
+ [workflows/workflow-fit-and-tuning.md](workflows/workflow-fit-and-tuning.md).
18
30
 
19
31
  ## Choose the execution mode first
20
32
 
@@ -62,6 +74,9 @@ before authoring. Full procedure: [workflows/authoring-workflows.md](workflows/a
62
74
 
63
75
  Use this skill to:
64
76
 
77
+ - **Decide fit** — judge whether a process should be a workflow at all or stay a descriptive
78
+ procedure / checklist, and tune an accepted one for latency and legible traces.
79
+ → workflows/workflow-fit-and-tuning.md
65
80
  - **Author a workflow** — turn a described process into a validated, dry-run-verified YAML definition
66
81
  in the right mode. → authoring-workflows.md
67
82
  - **Validate before trusting** — schema + semantic-check a workflow file (references, terminal
@@ -70,6 +85,9 @@ Use this skill to:
70
85
  taken, terminal status).
71
86
  - **Refine an existing workflow** — fix a stuck guard, add a state/node, retune `iterationBound`,
72
87
  re-scope variables / `env.allow`, with the smallest change. → workflows/operations.md
88
+ - **Refactor across the prose/YAML boundary** — promote a descriptive procedure into a workflow,
89
+ demote a workflow back to prose, or optimize an accepted one for latency.
90
+ → workflows/workflow-fit-and-tuning.md
73
91
  - **Extend the engine** — register a custom action/guard runner or a trust-gated extension module when
74
92
  the built-ins (`note`, `shell`, `always`, `action-ok`) fall short. → workflows/validation-and-extension.md
75
93
 
@@ -237,7 +255,7 @@ the operator accepts it, and never hot-edit a running workflow's shell in place.
237
255
 
238
256
  ```
239
257
  spur workflow validate <file> [--no-schema] [--json]
240
- spur workflow show <file>
258
+ spur workflow show <file> [--format <mermaid|todo>] [--json]
241
259
  spur workflow run <file> [--run-id <id>] [--vars <json>] [--dry-run] [--async] [--no-plan] [--quiet/--silent/--verbose] [--detail <level>] [--trace-file] [--no-log] [--steer] [--json]
242
260
  spur workflow continue [run-id] [--yes] [--answer <yes|no|cancel>] [--json]
243
261
  spur workflow cancel <run-id> [--json]
@@ -362,6 +380,10 @@ the workflow's actions (`shell`, custom runners) do that; this skill builds and
362
380
 
363
381
  ## Additional Resources
364
382
 
383
+ - [workflows/workflow-fit-and-tuning.md](workflows/workflow-fit-and-tuning.md) — the fit gate
384
+ (workflow vs. descriptive procedure), the per-node cost model, latency and observability tuning,
385
+ the node simplicity budget tied to the frozen ADR-069 measures, and the promote / demote /
386
+ optimize refactor procedures. Read before `add` and before any performance-motivated `refine`.
365
387
  - [workflows/operations.md](workflows/operations.md) — the operation procedures (validate/run/list/add/refine),
366
388
  the shared find-existing-workflow and validate-and-dry-run cores, and the mode-selection gate. The
367
389
  entry point for slash-command delegation.