@gobing-ai/spur 0.3.61 → 0.3.62

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/config/config.global.yaml +64 -57
  3. package/config/corpus-baseline.json +409 -49
  4. package/config/plugin-scripts.json +2 -2
  5. package/config/rules/boundary/config-loading-ownership.yaml +21 -0
  6. package/config/workflows/history-anatomy.yaml +366 -0
  7. package/package.json +1 -1
  8. package/plugins/sp/README.md +19 -6
  9. package/plugins/sp/commands/dev-find-issue.md +25 -46
  10. package/plugins/sp/commands/dev-run.md +22 -3
  11. package/plugins/sp/lib/artifact-digest.generated.d.mts +7 -0
  12. package/plugins/sp/lib/artifact-digest.generated.mjs +48 -0
  13. package/plugins/sp/plugin.json +1 -1
  14. package/plugins/sp/references/roles.md +5 -5
  15. package/plugins/sp/scripts/history-anatomy-cache.mjs +669 -0
  16. package/plugins/sp/scripts/history-anatomy-cache.ts +818 -0
  17. package/plugins/sp/skills/history-anatomy/SKILL.md +67 -0
  18. package/plugins/sp/skills/history-anatomy/references/modes.md +82 -0
  19. package/plugins/sp/skills/history-anatomy/references/operations.md +67 -0
  20. package/plugins/sp/skills/history-anatomy/references/report-contract.md +133 -0
  21. package/plugins/sp/skills/spur-dev/references/dev-operations.md +2 -2
  22. package/plugins/sp/skills/spur-dev/references/execution-batch.md +14 -1
  23. package/plugins/sp/skills/spur-dev/references/execution-workflow.md +7 -0
  24. package/plugins/sp/skills/spur-dev/references/flag-glossary.md +25 -13
  25. package/spur.js +1091 -823
  26. package/web/_astro/BoardApp.Z9jMwI9G.js +1 -0
  27. package/web/_astro/BoardApp.rkWGVwwK.js +178 -0
  28. package/web/_astro/{TaskDetail.D2N60cfE.js → TaskDetail.B7ODt_bA.js} +1 -1
  29. package/web/_astro/{arc.D-EfJJwf.js → arc.C5QRz6AQ.js} +1 -1
  30. package/web/_astro/{architectureDiagram-3BPJPVTR.CgvTTzqp.js → architectureDiagram-3BPJPVTR.CHptn8gZ.js} +1 -1
  31. package/web/_astro/{blockDiagram-GPEHLZMM.CnOohvvn.js → blockDiagram-GPEHLZMM.b1nEDkDw.js} +1 -1
  32. package/web/_astro/{c4Diagram-AAUBKEIU.BFcBGUy_.js → c4Diagram-AAUBKEIU.BWj8S_Qq.js} +1 -1
  33. package/web/_astro/channel.BJtn6CGV.js +1 -0
  34. package/web/_astro/{chunk-2J33WTMH.C-6MR-XY.js → chunk-2J33WTMH.q0BBuE1n.js} +1 -1
  35. package/web/_astro/{chunk-4BX2VUAB.BN4LP5AR.js → chunk-4BX2VUAB.DP2KDkAU.js} +1 -1
  36. package/web/_astro/{chunk-55IACEB6.tUJ_CTtZ.js → chunk-55IACEB6.CkCId72p.js} +1 -1
  37. package/web/_astro/{chunk-727SXJPM.BNCk-sxv.js → chunk-727SXJPM.oHWzxMot.js} +1 -1
  38. package/web/_astro/{chunk-AQP2D5EJ.BkpFIbNa.js → chunk-AQP2D5EJ.DkkgxfMP.js} +1 -1
  39. package/web/_astro/{chunk-FMBD7UC4.DcSZ87PN.js → chunk-FMBD7UC4.lmFJDvRK.js} +1 -1
  40. package/web/_astro/{chunk-ND2GUHAM.BkLvHFFf.js → chunk-ND2GUHAM.52jt0eHK.js} +1 -1
  41. package/web/_astro/{chunk-QZHKN3VN.HOdgszax.js → chunk-QZHKN3VN.Cg8KEFoD.js} +1 -1
  42. package/web/_astro/{classDiagram-4FO5ZUOK.CWfRiY6b.js → classDiagram-4FO5ZUOK.BuOUhxcD.js} +1 -1
  43. package/web/_astro/{classDiagram-v2-Q7XG4LA2.CWfRiY6b.js → classDiagram-v2-Q7XG4LA2.BuOUhxcD.js} +1 -1
  44. package/web/_astro/{cose-bilkent-S5V4N54A.C6j4PxoN.js → cose-bilkent-S5V4N54A.BRUU8E_0.js} +1 -1
  45. package/web/_astro/{dagre-BM42HDAG.BATZG1II.js → dagre-BM42HDAG.GDbfFYVV.js} +1 -1
  46. package/web/_astro/{diagram-2AECGRRQ.B0f5yY6x.js → diagram-2AECGRRQ.CvIDBeJF.js} +1 -1
  47. package/web/_astro/{diagram-5GNKFQAL.Bewupqv0.js → diagram-5GNKFQAL.DMgJjWOX.js} +1 -1
  48. package/web/_astro/{diagram-KO2AKTUF.C23r1tB2.js → diagram-KO2AKTUF.njjl-0AP.js} +1 -1
  49. package/web/_astro/{diagram-LMA3HP47.DUErdBOR.js → diagram-LMA3HP47.CcCqgP8M.js} +1 -1
  50. package/web/_astro/{diagram-OG6HWLK6.BubmwAjb.js → diagram-OG6HWLK6.Btqd-YVE.js} +1 -1
  51. package/web/_astro/{erDiagram-TEJ5UH35.BE_Qs5mv.js → erDiagram-TEJ5UH35.BbhML_Xo.js} +1 -1
  52. package/web/_astro/{flowDiagram-I6XJVG4X.COWdbzon.js → flowDiagram-I6XJVG4X.CkvsIgY8.js} +1 -1
  53. package/web/_astro/{ganttDiagram-6RSMTGT7.5EGYC4MK.js → ganttDiagram-6RSMTGT7.m8IWD_wW.js} +1 -1
  54. package/web/_astro/{gitGraphDiagram-PVQCEYII.bUUV7iEw.js → gitGraphDiagram-PVQCEYII.5z87HXO-.js} +1 -1
  55. package/web/_astro/{index.Cestp9nh.css → index.nWve6EHS.css} +1 -1
  56. package/web/_astro/{infoDiagram-5YYISTIA.CbiInFvz.js → infoDiagram-5YYISTIA.Bvyvsd7Q.js} +1 -1
  57. package/web/_astro/{ishikawaDiagram-YF4QCWOH.CB7lrSsJ.js → ishikawaDiagram-YF4QCWOH.DdWVYO87.js} +1 -1
  58. package/web/_astro/{journeyDiagram-JHISSGLW.rhGZgWt8.js → journeyDiagram-JHISSGLW.C3kOlYH1.js} +1 -1
  59. package/web/_astro/{kanban-definition-UN3LZRKU.DPeZD_lP.js → kanban-definition-UN3LZRKU.CfKTUHVC.js} +1 -1
  60. package/web/_astro/{linear.l60Nyp5b.js → linear.CMDHnzgX.js} +1 -1
  61. package/web/_astro/{mermaid.core.DmtMJcmL.js → mermaid.core.DN7-WrsP.js} +4 -4
  62. package/web/_astro/{mindmap-definition-RKZ34NQL.Cc47AcCn.js → mindmap-definition-RKZ34NQL.DcQxLvqG.js} +1 -1
  63. package/web/_astro/{pieDiagram-4H26LBE5.B8cN-S9g.js → pieDiagram-4H26LBE5.Nu-Inbw6.js} +1 -1
  64. package/web/_astro/{quadrantDiagram-W4KKPZXB.DY1FN89L.js → quadrantDiagram-W4KKPZXB.Dck2EChs.js} +1 -1
  65. package/web/_astro/{requirementDiagram-4Y6WPE33.CnM6VbV8.js → requirementDiagram-4Y6WPE33.C--E5XuW.js} +1 -1
  66. package/web/_astro/{sankeyDiagram-5OEKKPKP.B9DcGfNV.js → sankeyDiagram-5OEKKPKP.B_0iRNna.js} +1 -1
  67. package/web/_astro/{sequenceDiagram-3UESZ5HK.ymxaNlAY.js → sequenceDiagram-3UESZ5HK.PKuKr9mk.js} +1 -1
  68. package/web/_astro/{stateDiagram-AJRCARHV.DDOJ3d8P.js → stateDiagram-AJRCARHV.ClhGHb-O.js} +1 -1
  69. package/web/_astro/{stateDiagram-v2-BHNVJYJU.Ck0Nb_KY.js → stateDiagram-v2-BHNVJYJU.Bv6BBnvg.js} +1 -1
  70. package/web/_astro/{timeline-definition-PNZ67QCA.BNOvZwAN.js → timeline-definition-PNZ67QCA.BZm2ZvAH.js} +1 -1
  71. package/web/_astro/{vennDiagram-CIIHVFJN.D3k5ivSF.js → vennDiagram-CIIHVFJN.Dt6yZXry.js} +1 -1
  72. package/web/_astro/{wardley-L42UT6IY.BylaxVqn.js → wardley-L42UT6IY.rUs-E00M.js} +1 -1
  73. package/web/_astro/{wardleyDiagram-YWT4CUSO.HLNtFUm9.js → wardleyDiagram-YWT4CUSO.BcWgv4cA.js} +1 -1
  74. package/web/_astro/{xychartDiagram-2RQKCTM6.B-WooKSU.js → xychartDiagram-2RQKCTM6.DH-M-cX9.js} +1 -1
  75. package/web/index.html +2 -2
  76. package/plugins/sp/commands/dev-history-load.md +0 -63
  77. package/plugins/sp/scripts/history-load.mjs +0 -268
  78. package/plugins/sp/scripts/history-load.ts +0 -400
  79. package/web/_astro/BoardApp.BtRfVADq.js +0 -1
  80. package/web/_astro/BoardApp.CBIzcvqi.js +0 -178
  81. package/web/_astro/channel.kIu33Gui.js +0 -1
@@ -0,0 +1,67 @@
1
+ ---
2
+ name: history-anatomy
3
+ description: "Independent owner of diagnostic interpretation over already-imported history — the daily/ad-hoc mode contract, a closed finding taxonomy, the eleven-section report contract, and the enrich/validate rubrics. Triggers: history-anatomy, run the daily report, ad-hoc diagnosis, find issues over history."
4
+ license: Apache-2.0
5
+ version: 1.0.0
6
+ metadata:
7
+ author: spur
8
+ platforms: "claude-code,codex,openclaw,opencode,antigravity,pi"
9
+ category: analysis-core
10
+ interactions:
11
+ - pipeline
12
+ - inversion
13
+ pipeline_steps:
14
+ - report
15
+ - identify
16
+ - propose
17
+ - generate
18
+ openclaw:
19
+ emoji: "🩺"
20
+ see_also:
21
+ - sp:issue-finding
22
+ - sp:spur-cli
23
+ - sp:spur-dev
24
+ ---
25
+
26
+ # sp:history-anatomy
27
+
28
+ Owns **interpretation only** over already-imported history. It decides which arguments are legal
29
+ in which mode, what counts as evidence, how findings are keyed and graded, and what a published
30
+ report must contain, plus the rubrics its `enrich` and `validate` operations follow. It never
31
+ launches a workflow, never reprocesses raw history files, and never mutates the corpus, docs, or sources —
32
+ all orchestration lives in `history-anatomy.yaml` (0660), not here.
33
+
34
+ This body routes to the three references; the procedure lives there (the BODY_BUDGET shape — a
35
+ fresh skill cannot be added to the BASELINE exemption map).
36
+
37
+ **Choose the mode first.** The single entry is `--mode daily|ad-hoc` (default `daily`):
38
+ `references/modes.md`.
39
+
40
+ **Then hold the report to the contract.** Every report must contain the eleven sections and every
41
+ finding the full field set, with evidence rules and an explicit comparability verdict:
42
+ `references/report-contract.md`.
43
+
44
+ **Then apply the operations.** The workflow calls `enrich` to author the model half, and
45
+ `validate` to gate a candidate report; neither operation launches a workflow:
46
+ `references/operations.md`.
47
+
48
+ ## Primary directive
49
+
50
+ - The artifact is the only evidence plane. A dimension the forensics artifact cannot support is a
51
+ **telemetry gap** reported as `not available` — never a raw history-file fallback.
52
+ - Every causal claim needs **two independent signals** (or one labelled hypothesis with a
53
+ confirmation path). Every finding carries at least one evidence anchor. No anchor, no finding.
54
+ - The report never states a trend, delta, or percentage it cannot support — an insufficient or
55
+ materially different baseline renders `not comparable`, with no fabricated comparison.
56
+ - Findings are keyed on a **stable key**, never the prose title, so rewording never reclassifies
57
+ recurrence.
58
+ - Remediation options are **proposals only**: an owner surface, expected impact, a verification
59
+ method, and reversibility. No applied change, diff, or command the report claims to have run.
60
+
61
+ ## Read the references
62
+
63
+ | Reference | Owns |
64
+ | --- | --- |
65
+ | [`references/modes.md`](references/modes.md) | The daily/ad-hoc mode matrix, bounds normalization, the DST-aware calendar-day rule, and the fail-loud message shape. |
66
+ | [`references/report-contract.md`](references/report-contract.md) | The eleven sections (in order), the per-finding field set, the closed category vocabulary, the stable-key grammar, the evidence rules, comparison semantics, recurrence classes, and the positive-pattern / remediation standards. |
67
+ | [`references/operations.md`](references/operations.md) | The `enrich` and `validate` operation rubrics; neither launches a workflow. |
@@ -0,0 +1,82 @@
1
+ # Mode contract — `sp:history-anatomy` (HA-S1, 0658)
2
+
3
+ The skill resolves exactly two modes. Everything else fails loud. This matrix is the enforcement
4
+ surface the workflow (0660) and the skill share; keep the vocabulary frozen.
5
+
6
+ ## Mode vocabulary (frozen)
7
+
8
+ - Modes: `daily` | `ad-hoc`.
9
+ - Unsupported value literal: `not available`.
10
+
11
+ ## Resolution rule
12
+
13
+ `--mode <value>` selects the mode. If `--mode` is omitted, the resolved mode is **`daily`** and the
14
+ resolved window is the **current local calendar day**.
15
+
16
+ ## `--mode daily` (default)
17
+
18
+ | Allowed | Behavior |
19
+ | --- | --- |
20
+ | (none) | Default window = the current local calendar day. |
21
+ | `--date <YYYY-MM-DD>` | Selects that local calendar day as the window. |
22
+
23
+ Rejected arguments (each fails loud, naming the offending argument):
24
+
25
+ | Argument | Why rejected |
26
+ | --- | --- |
27
+ | focus text (positional) | Daily mode has no focus string. |
28
+ | `--since` / `--until` | Daily always uses the calendar-day window. |
29
+ | `--output` | Daily always writes to the run directory (see 0660). |
30
+
31
+ A daily invocation must print the normalized **inclusive ISO bounds** and the timezone used, so the
32
+ wall-clock window is auditable.
33
+
34
+ ## `--mode ad-hoc`
35
+
36
+ Requires a **non-empty focus** and **two ordered inclusive bounds**.
37
+
38
+ | Argument | Rule |
39
+ | --- | --- |
40
+ | focus (positional) | Required; a missing or empty focus fails loud. |
41
+ | `--since <iso>` | Required; the inclusive lower bound. |
42
+ | `--until <iso>` | Required; must be present with `--since`; must not be earlier than `--since`. |
43
+ | `--output <path>` | Optional; when present, writes to that explicit path. When absent, writes to the run directory. |
44
+
45
+ Rejected arguments (each fails loud, naming the offending argument):
46
+
47
+ | Argument | Why rejected |
48
+ | --- | --- |
49
+ | `--date` | Ad-hoc windows are explicit bounds, not a single date. |
50
+ | `--recompute` | Ad-hoc never recomputes a cache (0660 owns the cache branch). |
51
+
52
+ ## Bounds normalization
53
+
54
+ Normalize both bounds to inclusive ISO-8601 instants. The report prints the normalized bounds and
55
+ the timezone used.
56
+
57
+ ## DST-aware calendar-day rule
58
+
59
+ `--date <YYYY-MM-DD>` must span the **full local calendar day including any DST shift** — it is
60
+ never a fixed 24-hour offset from local midnight.
61
+
62
+ - Resolve the local timezone (the operator's configured zone; state it in the report).
63
+ - The interval runs from the day's first local instant to the day's last local instant.
64
+ - On a 23-hour or 25-hour daylight-saving day, the interval is that actual length, not 24 hours.
65
+ - The bounds are NOT computed as `midnight + 24h`.
66
+
67
+ This is what makes a daily report's window reproducible across DST transitions.
68
+
69
+ ## Fail-loud message shape
70
+
71
+ Every rejection names the offending argument and the rule it violated, e.g.:
72
+
73
+ ```
74
+ --mode daily rejects "focus text" (daily mode has no focus string)
75
+ --mode ad-hoc requires a non-empty focus
76
+ --mode ad-hoc: --until is missing --since
77
+ --mode ad-hoc: --until (2026-08-01T00:00:00Z) is earlier than --since (2026-08-02T00:00:00Z)
78
+ --mode ad-hoc rejects --date (ad-hoc uses explicit --since/--until bounds)
79
+ ```
80
+
81
+ A combined conflict (e.g. `--mode daily --since X`) reports **each** offending argument it can
82
+ attribute; at minimum it names the first conflicting one.
@@ -0,0 +1,67 @@
1
+ # Operations — `sp:history-anatomy` (HA-S1, 0658)
2
+
3
+ The skill exposes exactly two operations, invoked by `history-anatomy.yaml` (0660):
4
+
5
+ - `enrich` — the model is given the rendered forensics artifacts (current + baseline) and authors
6
+ the model half of the report.
7
+ - `validate` — the model is given a candidate report and independently checks its evidence claims
8
+ against the artifacts.
9
+
10
+ Both are **skill operations**, not workflows. **Neither operation launches a workflow** — that is
11
+ the recursion guard. This is why the rubric lives here, single-sourced, rather than being
12
+ duplicated into the YAML: the rubric cannot recurse.
13
+
14
+ ## Invocation contract
15
+
16
+ | Operation | Input | Output |
17
+ | --- | --- | --- |
18
+ | `enrich` | current forensics artifact + baseline artifact (plain `HistoryArtifact` JSON) | the model-authored report sections (Baseline comparison, Findings, Recurrence ledger, Remediation options, Performance analysis, Workflow and process improvements, Positive patterns) meeting the report contract |
19
+ | `validate` | a candidate report + the current/baseline artifacts | a PASS / FAIL verdict with per-finding and per-section evidence checks, naming any failing rule |
20
+
21
+ Freeze these operation names and this input/output contract — 0660 consumes them verbatim.
22
+
23
+ ## `enrich` rubric
24
+
25
+ Given the current and baseline artifacts, author the model half of the report. Apply, in order:
26
+
27
+ 1. **Scope and provenance** + **Executive summary** + **Evidence ledger** are populated from the
28
+ artifact and rendered deterministically (or stated as the artifact's own data). The model
29
+ authors only the sections listed as model-authored.
30
+ 2. **Baseline comparison** — apply the comparison semantics: daily → immediately preceding local
31
+ calendar day; ad-hoc → immediately preceding equal-duration window. If baseline coverage is
32
+ insufficient or materially different, emit `not comparable` and **no** trend/delta/percentage.
33
+ 3. **Findings** — each with the full per-finding field set (key, category, impact, trend,
34
+ observation, inference, confidence, contradictions, evidenceAnchor). Categories from the closed
35
+ vocabulary; stable keys of the form `<category>:<owner-surface>:<signal>`.
36
+ 4. **Recurrence ledger** — classify every finding against the baseline on the **stable key**.
37
+ 5. **Telemetry gaps** — every dimension the artifact cannot support, rendered `not available`.
38
+ 6. **Remediation options** — proposals only (owner surface, expected impact, verification method,
39
+ reversibility). No applied change/diff/command.
40
+ 7. **Positive patterns** — same evidence standard as problems.
41
+
42
+ Never fabricate a value, a trend, an anchor, or an applied fix. Never launch a workflow.
43
+
44
+ ## `validate` rubric
45
+
46
+ Given a candidate report and the artifacts, independently verify:
47
+
48
+ - **Eleven-section completeness and order** — all eleven section names present in the frozen
49
+ order; none renamed/omitted.
50
+ - **Per-finding fields** — every finding (problem and positive) carries key, category, impact,
51
+ trend, observation, inference, confidence, contradictions, evidenceAnchor; category in the closed
52
+ vocabulary; stable-key grammar.
53
+ - **Evidence anchors** — every finding has at least one verifiable anchor; no anchor → FAIL.
54
+ - **Causality gate** — a causal claim with one signal must be labelled a hypothesis with a
55
+ confirmation path; otherwise FAIL.
56
+ - **Inference names observations** — an inference that does not name its supporting observations
57
+ FAILs.
58
+ - **Comparability** — a `not comparable` baseline must state no trend/delta/percentage; a stated
59
+ trend must be supported by a comparable baseline.
60
+ - **Recurrence integrity** — classification is consistent with the stable keys (a rewording must
61
+ not flip a recurring finding to new).
62
+ - **Positive patterns** — held to the same standard; an anchor-less entry FAILs.
63
+ - **Remediation** — proposals only; any applied change, diff, or command the report claims to have
64
+ run FAILs.
65
+
66
+ Emit `PASS` only when every rule holds. On FAIL, name the section, the finding key, and the rule
67
+ violated. Never launch a workflow.
@@ -0,0 +1,133 @@
1
+ # Report contract — `sp:history-anatomy` (HA-S1, 0658)
2
+
3
+ The published report is the skill's contract. This reference owns the eleven sections (in order),
4
+ the per-finding field set, the closed category vocabulary, the stable-key grammar, the evidence
5
+ rules, comparison semantics, recurrence classes, and the positive-pattern / remediation standards.
6
+
7
+ ## Eleven required sections (in order, frozen)
8
+
9
+ 1. Scope and provenance
10
+ 2. Executive summary
11
+ 3. Baseline comparison
12
+ 4. Findings
13
+ 5. Recurrence ledger
14
+ 6. Telemetry gaps
15
+ 7. Remediation options
16
+ 8. Performance analysis
17
+ 9. Workflow and process improvements
18
+ 10. Positive patterns
19
+ 11. Evidence ledger
20
+
21
+ These names are consumed verbatim by 0659's structure gate and 0660's validation stage. Do not
22
+ rename, reorder, omit, or restate them.
23
+
24
+ ## Closed category vocabulary (frozen)
25
+
26
+ - Categories: `reliability` | `repetition` | `workflow` | `performance` | `coverage` |
27
+ `telemetry` | `positive`.
28
+
29
+ Every finding's category is drawn from this closed set. No category is invented.
30
+
31
+ ## Stable finding key (frozen)
32
+
33
+ Each finding carries a stable key of the form:
34
+
35
+ ```
36
+ <category>:<owner-surface>:<signal>
37
+ ```
38
+
39
+ - `category` — one of the closed vocabulary above.
40
+ - `owner-surface` — the surface that would own a change (e.g. a module, command, or skill; free
41
+ text is allowed but must be stable across runs).
42
+ - `signal` — a concise, stable machine-readable identifier of the finding.
43
+
44
+ The key is the recurrence identity. Rewording a finding's title must **never** reclassify it.
45
+
46
+ ## Per-finding field set
47
+
48
+ Every finding — problem and positive alike — carries the full field set:
49
+
50
+ | Field | Requirement |
51
+ | --- | --- |
52
+ | `key` | Stable `<category>:<owner-surface>:<signal>` key. |
53
+ | `category` | From the closed vocabulary, matching the key's first segment. |
54
+ | `impact` | What the situation costs or enables (qualitative, or a concrete number where supported). |
55
+ | `trend` | `new` / `recurring` / `regressed` / `improved` / `resolved` / `not-comparable` (see recurrence). |
56
+ | `observation` | What the artifacts show — evidence, not interpretation. |
57
+ | `inference` | What the observation is reasoned to mean; names its supporting observations. |
58
+ | `confidence` | Per-finding: `high` / `medium` / `low`. Never one blanket report-level score. |
59
+ | `contradictions` | Any contradicting signal shown beside the finding, not silently reconciled. |
60
+ | `evidenceAnchor` | At least one anchor to the forensics artifact or cited `file:line`. An entry with no anchor is invalid. |
61
+
62
+ ## Evidence rules
63
+
64
+ Enforcement rail for every claim:
65
+
66
+ 1. **Causality needs two independent signals.** A causal claim supported by two or more
67
+ independent signals passes. Exactly one signal is not causation — it must be labelled a
68
+ **hypothesis** with a stated confirmation path.
69
+ 2. **A process/workflow change needs recurrence** across two independent sessions, or a single
70
+ **high-impact contract violation cited at `file:line`**.
71
+ 3. **Unsupported dimensions read `not available`** and are mirrored into the telemetry-gaps
72
+ section. Never a fabricated value, never a raw history-file fallback.
73
+ 4. **Focus biases ranking, not collection.** A focus string changes finding ranking and emphasis;
74
+ it never suppresses material off-topic findings within the window.
75
+ 5. **Every inference names its supporting observations.** An inference that does not name its
76
+ observations fails validation.
77
+ 6. **Every finding has an evidence anchor.** No anchor, no finding — no row may be dropped because
78
+ it "lacks evidence" while still appearing in a summary.
79
+
80
+ ## Comparison semantics
81
+
82
+ The baseline comparison states an explicit comparability verdict.
83
+
84
+ - **Daily** compares against the **immediately preceding local calendar day**.
85
+ - **Ad-hoc** compares against the **immediately preceding equal-duration window**.
86
+ - **Insufficient or materially different coverage** renders **`not comparable`**.
87
+ - A `not comparable` baseline states no trend, no delta, and no percentage. An unsupported
88
+ comparison is never fabricated.
89
+
90
+ ## Recurrence ledger
91
+
92
+ Every finding is classified against the baseline using the **stable key**:
93
+
94
+ - `new` — not present in the baseline.
95
+ - `recurring` — present in the baseline with comparable severity.
96
+ - `regressed` — present but worse (severity, count, or impact increased).
97
+ - `improved` — present but better.
98
+ - `resolved` — present in the baseline but absent now.
99
+ - `not-comparable` — the baseline was `not comparable`.
100
+
101
+ Matching is on the stable key, never the prose title. Rewording a title between two runs must not
102
+ reclassify a recurring finding as new.
103
+
104
+ ## Positive patterns
105
+
106
+ Positive entries are held to the **same evidence standard as problems**: they carry observation,
107
+ inference, confidence, and at least one evidence anchor. An entry without an anchor is invalid. The
108
+ section renders successful workflows, resolved past issues that stay resolved, and healthy,
109
+ repeating behavior worth keeping.
110
+
111
+ ## Remediation options (proposals only)
112
+
113
+ Each option is a **proposal** that names:
114
+
115
+ - **owner surface** — who would apply it.
116
+ - **expected impact** — what it is expected to change.
117
+ - **verification method** — how the change would be confirmed.
118
+ - **reversibility** — whether and how it can be rolled back.
119
+
120
+ The report must contain **no applied change, no diff, and no command it claims to have run**. The
121
+ skill never applies a fix; it proposes one.
122
+
123
+ ## Evidence ledger
124
+
125
+ The final section lists, for every finding, the artifact anchor(s) and any cited `file:line`,
126
+ so a reader can verify the report's claims against the evidence plane.
127
+
128
+ ## Truthfulness invariants
129
+
130
+ - `not available` is the true rendering for an unsupported dimension, never a masked gap.
131
+ - `not comparable` is the true rendering for an unsupported comparison, never a computed delta.
132
+ - No bounded leaderboard length is presented as a population total (see 0657 / ADR-080: the
133
+ coverage section reads `artifact.population` and renders `top N of M`).
@@ -68,7 +68,7 @@ each would be scope creep for one-liner procedures.
68
68
  | 2 | review | `dev-review` | `Skill()` | `sp:code-verification` (`review`) + `sp:functional-review` + `sp:code-improvement` | `[<wbs\|path>] [--agent <inline\|auto\|name>] [--focus <dims>] [--fix (deprecated)]` |
69
69
  | 3 | verify | `dev-verify` | `Skill()` | `sp:code-verification` (`verify`) | `<wbs> [--agent <inline\|auto\|name>] [--fix <none\|blockers-first\|all>] [--focus <lens>] [--bdd] [--auto] [--force] [--next] [--skip-shippable]` |
70
70
  | 3a | verifyall | `dev-verifyall` | `Skill()` → agent | `sp:spur-dev` (`verifyall`) | `--tasks <selector> [--feature <id>] [--agent <inline\|auto\|name>] [--fix <none\|blockers-first\|all>] [--focus <lens>] [--bdd] [--auto] [--force] [--next] [--json] [--skip-shippable] [--worktree [<name>]]` |
71
- | 4 | run | `dev-run` | `Skill()` | `sp:spur-dev` (`run` / `implement`) | `<wbs> [--mode <full\|implement>] [--agent <inline\|auto\|name>] [--auto] [--next] [--wrap] [--continue]` |
71
+ | 4 | run | `dev-run` | `Skill()` | `sp:spur-dev` (`run` / `implement`) | `<wbs> [--mode <full\|implement>] [--agent <inline\|auto\|name>] [--auto] [--next] [--wrap] [--continue] [--worktree [<name>]]` |
72
72
  | 5 | refine | `dev-refine` | `Skill()` | `sp:spur-dev` (`refine`) | `<wbs> [--focus <mode>] [--description <text>] [--depth <standard\|ready>] [--agent <inline\|auto\|name>] [--auto] [--next]` |
73
73
  | 5a | refineall | `dev-refineall` | `Skill()` | `sp:spur-dev` (`refineall`) | `--feature <id> \| --tasks <selector> [--focus <mode>] [--description <text>] [--depth <standard\|ready>] [--agent <inline\|auto\|name>] [--auto] [--keep-going] [--status <s>] [--json] [--worktree [<name>]]` |
74
74
  | 6 | plan | `dev-plan` | `Skill()` | `sp:spur-dev` (`plan`) | `"<description>" [--feature <id>] [--parent <feature-id>] [--agent <inline\|auto\|name>] [--skip-design] [--auto] [--approve-taste]` |
@@ -140,7 +140,7 @@ must not be changed without updating the backing skill.
140
140
  ### 4. run
141
141
 
142
142
  - **Purpose:** Run a task through the execution pipeline (full) or execute a single pipeline step (implement).
143
- - **Inputs:** `<wbs>` (required). `--mode <full|implement>` selects the execution mode. `implement` invokes `sp:code-implementation` inline by default. Interactive `full` with omitted `--agent` or explicit `--agent inline` reads `task-pipeline.yaml` and drives its actions/guards in the host session — host-controlled and non-subprocess; **omitted** `--agent` keeps 0508 eligibility (eligible `agent.run` stages dispatch once to a native subagent, host fallback), while explicit `--agent inline` is the zero-dispatch carve-out (every stage executes in the invoking session). `--agent auto`, a name, or headless invocation launches the workflow subprocess. `--agent <inline|auto|name>` selects the execution surface (see [SSOT](cross-cutting.md#inline-default-execution-surface)). `--auto` skips the HITL approve gate / confirmations and propagates down the `--next` chain. `--next` controls chaining only and never changes the mode; a pipeline implement stage must invoke `/sp:dev-run <wbs> --mode implement`. On implement success with `--next`, transition `todo → wip → testing` through the FSM (guards honored — no `--no-lifecycle`) + chain to `/sp:dev-verify <wbs> --auto --next`. On a guard failure, stop as review-pending. **Partial-deliverable rule:** if the task ships only part of its requirements (e.g. an R1/R2 split with the rest in a follow-up task), the `## Solution` section must state that explicitly and the verify verdict will record the scope. `--wrap` hands off to `/sp:dev-wrap <wbs>` after the main step; the `--agent` selector is preserved into that handoff when supplied (omission remains omission), and the wrap hop reports its own trigger-3 subprocess override per the wrap contract.
143
+ - **Inputs:** `<wbs>` (required). `--mode <full|implement>` selects the execution mode. `implement` invokes `sp:code-implementation` inline by default. Interactive `full` with omitted `--agent` or explicit `--agent inline` reads `task-pipeline.yaml` and drives its actions/guards in the host session — host-controlled and non-subprocess; **omitted** `--agent` keeps 0508 eligibility (eligible `agent.run` stages dispatch once to a native subagent, host fallback), while explicit `--agent inline` is the zero-dispatch carve-out (every stage executes in the invoking session). `--agent auto`, a name, or headless invocation launches the workflow subprocess. `--agent <inline|auto|name>` selects the execution surface (see [SSOT](cross-cutting.md#inline-default-execution-surface)). `--auto` skips the HITL approve gate / confirmations and propagates down the `--next` chain. `--next` controls chaining only and never changes the mode; a pipeline implement stage must invoke `/sp:dev-run <wbs> --mode implement`. On implement success with `--next`, transition `todo → wip → testing` through the FSM (guards honored — no `--no-lifecycle`) + chain to `/sp:dev-verify <wbs> --auto --next`. On a guard failure, stop as review-pending. **Partial-deliverable rule:** if the task ships only part of its requirements (e.g. an R1/R2 split with the rest in a follow-up task), the `## Solution` section must state that explicitly and the verify verdict will record the scope. `--worktree [<name>]` runs the full pipeline inside an isolated git worktree (create or reuse; FF-merge on success, retain on failure) — the batch lifecycle in [execution-batch.md § Worktree isolation](execution-batch.md#worktree-isolation---worktree-name) applied to a batch of one; rejected with `--mode implement`. `--wrap` hands off to `/sp:dev-wrap <wbs>` after the main step; the `--agent` selector is preserved into that handoff when supplied (omission remains omission), and the wrap hop reports its own trigger-3 subprocess override per the wrap contract.
144
144
  - **Backing:** `sp:spur-dev` skill — `run` operation for the full pipeline (the spine drives it); `sp:code-implementation` competency skill for the implement step (the spine dispatches to it).
145
145
  - **Modes:**
146
146
  - **`full`** (default): Drive the full pipeline — precheck → implement → test → review → approve(HITL) → verify → record → done. Interactive omit/inline uses [inline-pipeline-driver.md](inline-pipeline-driver.md) (host-controlled; eligible stages may use a native subagent); explicit/headless executor selection invokes `spur workflow run task-pipeline.yaml --vars '{"wbs":"<wbs>"}'` (with `profile: auto` when `--auto`). Both monitor/surface HITL and preserve the YAML gates. `--next` never changes this mode.
@@ -421,6 +421,15 @@ isolated git worktree instead of the operator's working directory. This section
421
421
  lifecycle for the sequential batch loop. Per-task worktrees and `--mode parallel` isolation stay out
422
422
  of scope (task 0142 Slice A); `--worktree --mode parallel` is rejected.
423
423
 
424
+ **Single-task `dev-run` (batch of one).** `/sp:dev-run <wbs> --worktree [<name>]` runs this same
425
+ lifecycle with a one-task loop: WT-1…WT-6 apply unchanged, the marker's `command` is `dev-run` and
426
+ its `selector` is the `<wbs>` (so WT-6's command+selector fallback resolves the resume), and the
427
+ derived branch/directory slug is the WBS — `sp/run-<wbs>-<short-id>`. The WT-4 success condition
428
+ "no failed task" reads as "the task reached terminal `done` with no failed stage"; a failing gate, a
429
+ non-PASS verify verdict, or a HITL pause that ends the run take the WT-5 retention path. Only the
430
+ full pipeline is eligible — `--worktree --mode implement` is rejected (WT-7), because that mode is
431
+ the pipeline's implement stage and already runs in the driver's tree.
432
+
424
433
  One flag, two modes (see the glossary entry for the ownership rule). Bare `--worktree` is **create
425
434
  mode** (cut a fresh branch + sibling tree). `--worktree <name>` is **reuse mode** (attach to a tree
426
435
  that already exists); name resolution (§ WT-2 below) runs before WT-1. The deltas each mode applies
@@ -714,9 +723,13 @@ fallback, because `<name>` was explicit and unambiguous intent.
714
723
  ### WT-7 — Exclusions (R8)
715
724
 
716
725
  - **`dev-next`** does not get `--worktree` — it dispatches a single step; per-step isolation is not
717
- worth the worktree cost.
726
+ worth the worktree cost. `dev-run` is different: it drives a whole task pipeline, so it does get
727
+ the flag.
718
728
  - **`--mode parallel`** is rejected when combined with `--worktree` — per-task worktrees and
719
729
  parallel isolation remain task 0142 Slice A.
730
+ - **`--mode implement`** is rejected when combined with `--worktree` on `dev-run` — that mode *is*
731
+ the pipeline's implement stage (bug-742) and runs in whatever tree the driver set up; a second
732
+ worktree would split one task's evidence across two trees.
720
733
  - **No** create-with-name (`--worktree <name>` never creates; an unresolvable name is an error),
721
734
  no `--worktree-keep` variant, no auto-cleanup of stale worktrees or markers from prior runs.
722
735
 
@@ -100,6 +100,13 @@ cache-conservation discipline (`plugins/sp/skills/dogfood-testing/references/mon
100
100
  > - Without `--auto`: warn the operator before calling `spur workflow run` and prompt for confirmation or a plan-item override via `--vars '{"maxImplementPlanItems":"<count>"}'`.
101
101
  > - With `--auto`: automatically append `"maxImplementPlanItems": "<count>"` to `--vars` and log a single-line notice (e.g. `Notice: task <wbs> has N plan items (>8 default cap); injecting maxImplementPlanItems override`).
102
102
 
103
+ **`--worktree [<name>]` wraps Step 2, on either surface.** When `/sp:dev-run --mode full` carries
104
+ [`--worktree`](flag-glossary.md#flag-worktree), create or adopt the worktree *before* launching the
105
+ pipeline, `cd` into it (`spur workflow run` and the inline driver both resolve cwd from the process),
106
+ and merge-or-retain after the run reports. The lifecycle is
107
+ [execution-batch.md § Worktree isolation](execution-batch.md#worktree-isolation---worktree-name)
108
+ applied to a batch of one. Rejected with `--mode implement`.
109
+
103
110
  **Choose the surface before execution.** In an interactive `/sp:dev-run --mode full` invocation,
104
111
  omit/`--agent inline` selects the [inline pipeline driver](inline-pipeline-driver.md). Read the YAML
105
112
  at invocation time, allocate the inline run id/session provenance, record `task run-link`, and walk
@@ -200,16 +200,25 @@ restrictor. On `dev-refineall` it is instead one of a required pair — supply e
200
200
 
201
201
  Select an execution mode: `full|implement` on `dev-run` (full pipeline vs implement-only),
202
202
  `sequential|parallel` on `dev-runall` (serial vs fanned-out-independent-subset),
203
- `fan-out|review-panel|investigation` on `dev-parallel`, and the reconstruction depth
204
- `briefing|structure|architecture|design|full` on `dev-reverse`. Mode selection is explicit and orthogonal
203
+ `fan-out|review-panel|investigation` on `dev-parallel`, the reconstruction depth
204
+ `briefing|structure|architecture|design|full` on `dev-reverse`, and `daily|ad-hoc` on
205
+ `dev-find-issue` (the history-anatomy report mode). Mode selection is explicit and orthogonal
205
206
  to `--next`.
206
207
 
208
+ ### `--date <YYYY-MM-DD>` — local calendar day selection
209
+
210
+ **Anchor:** `#flag-date`.
211
+
212
+ Select a local calendar day on `dev-find-issue` (daily mode) and `dev-daily` (report date). DST-aware:
213
+ `--date <that-date>` spans the full local calendar day including any daylight-saving shift, never a
214
+ fixed 24-hour offset.
215
+
207
216
  ### `--task <wbs>` — task work or task narrowing
208
217
 
209
218
  **Anchor:** `#flag-task`.
210
219
 
211
220
  Connect the current command's result to task work, or narrow a history analysis to one task
212
- (`dev-brainstorm`, `dev-debug`, `dev-dogfood`, `dev-find-next`, `dev-history-load`). The value and
221
+ (`dev-brainstorm`, `dev-debug`, `dev-dogfood`, `dev-find-next`). The value and
213
222
  the effect are per-command — this flag is a family, not one behavior:
214
223
 
215
224
  - `dev-brainstorm` `[<feature-id>]` — **creates** one task from the chosen approach, landing at
@@ -221,7 +230,6 @@ the effect are per-command — this flag is a family, not one behavior:
221
230
  id names the target instead of offering rank 1.
222
231
  - `dev-debug` `[<wbs>]` — **attaches** findings to an existing task. Optional WBS names it.
223
232
  - `dev-dogfood` (no value) — **records** run outcomes against the task under test.
224
- - `dev-history-load` `<wbs>` — narrows the `analyze` step to that task's messages.
225
233
 
226
234
  ### `--since <ref>` — lower bound on a range
227
235
 
@@ -250,10 +258,10 @@ Upper bound on a range: a git ref on `dev-changelog` (defaults to `HEAD`), or an
250
258
 
251
259
  **Anchor:** `#flag-source`.
252
260
 
253
- Scope the operation to one agent source (`dev-find-issue`, `dev-history-load`): one of
254
- `pi|claude|codex|gemini|opencode|antigravity|openclaw|omp|grok|agy` (or `all`). On
255
- `dev-history-load` the value is forwarded to **both** `spur history import` and
256
- `spur history analyze`; on `dev-find-issue` it narrows the report scan to that source's sessions.
261
+ _(Removed 2026-08-24, HA-S1 0661.)_ The only two commands that consumed this flag;
262
+ `dev-find-issue` (now a `sp:history-anatomy` forwarder) and `dev-history-load` (deleted) no
263
+ longer declare it, so the entry is dead and removed. The underlying `spur history` surfaces
264
+ retain their own `--source` handling.
257
265
 
258
266
  ### `--status <s>` — filter by task status
259
267
 
@@ -280,7 +288,8 @@ Omit the design package (system-design satellite + task `### Design`) on plannin
280
288
 
281
289
  **Anchor:** `#flag-output`.
282
290
 
283
- Write the command's result to a file path (`dev-daily`, `dev-reverse`) instead of stdout.
291
+ Write the command's result to a file path (`dev-daily`, `dev-find-issue`, `dev-reverse`) instead of
292
+ stdout. On `dev-find-issue` (ad-hoc mode) an explicit path replaces the default run-directory write.
284
293
 
285
294
  ### `--merge` — trigger branch cleanup
286
295
 
@@ -359,8 +368,9 @@ flag sets both.
359
368
 
360
369
  **Anchor:** `#flag-worktree`.
361
370
 
362
- Batch commands only (`dev-refineall`, `dev-runall`, `dev-verifyall`): run the entire driver loop
363
- inside an isolated git worktree instead of the operator's working directory. One flag, two modes:
371
+ Batch commands plus single-task `dev-run` (`dev-refineall`, `dev-runall`, `dev-verifyall`,
372
+ `dev-run`): run the entire driver loop inside an isolated git worktree instead of the operator's
373
+ working directory. One flag, two modes:
364
374
 
365
375
  - **Create mode** — bare `--worktree` (no value). Cut a fresh branch from the current HEAD's ref,
366
376
  create a sibling worktree with a derived name, run the batch there. On a fully successful batch
@@ -385,8 +395,10 @@ merges but never removes. This keeps the continue-the-work loop stable — after
385
395
  **Value binding.** The following token is consumed as `<name>` **only when it does not begin with
386
396
  `-`**, so `--worktree --auto` is the bare create form and `--agent`/`--feature`/etc. are never
387
397
  swallowed as the name. `--worktree=<name>` is the unambiguous spelling. `/sp:dev-next` does not get
388
- the flag (single step; not worth the worktree cost), and `--worktree --mode parallel` is rejected
389
- (per-task parallel isolation stays task 0142). The full lifecycle — name resolution, dirty-tree
398
+ the flag (single *step*; not worth the worktree cost unlike `dev-run`, which isolates a whole
399
+ task pipeline), `--worktree --mode parallel` is rejected (per-task parallel isolation stays task
400
+ 0142), and `--worktree --mode implement` is rejected on `dev-run` (that mode *is* the pipeline's
401
+ implement stage and runs in the driver's tree). The full lifecycle — name resolution, dirty-tree
390
402
  precheck, creation or adoption, crash-safe marker, merge-or-retain, and `--continue` re-entry — is
391
403
  specified in [execution-batch.md § Worktree isolation](execution-batch.md#worktree-isolation---worktree-name).
392
404
  Portable `git worktree` commands only; the git mechanics are reused from