@gobing-ai/spur 0.3.75 → 0.3.77
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/config/config.example.yaml +6 -0
- package/config/plugin-scripts.json +4 -0
- package/config/rules/README.md +1 -1
- package/config/rules/strict/runtime-boundaries.yaml +2 -0
- package/config/rules/typescript/no-eslint-suppressions.yaml +30 -0
- package/config/workflows/task-pipeline.yaml +19 -6
- package/package.json +9 -9
- package/plugins/sp/commands/dev-refineall.md +2 -2
- package/plugins/sp/plugin.json +1 -1
- package/plugins/sp/scripts/inline-run-setup.ts +198 -0
- package/plugins/sp/scripts/verify-answer-lint.ts +134 -26
- package/plugins/sp/skills/code-verification/SKILL.md +4 -2
- package/plugins/sp/skills/code-verification/references/verdict-schema.md +11 -2
- package/plugins/sp/skills/dogfood-testing/SKILL.md +48 -40
- package/plugins/sp/skills/dogfood-testing/references/monitor-ledger.md +34 -10
- package/plugins/sp/skills/dogfood-testing/references/report-template.md +10 -6
- package/plugins/sp/skills/spur-cli/references/tasks/section-editing.md +9 -3
- package/plugins/sp/skills/spur-cli/references/tasks/verbs.md +5 -3
- package/plugins/sp/skills/spur-cli/references/workflows.md +3 -3
- package/plugins/sp/skills/spur-dev/references/dev-operations.md +2 -2
- package/plugins/sp/skills/spur-dev/references/inline-pipeline-driver.md +35 -7
- package/schemas/spur-config.schema.json +4 -0
- package/spur.js +10741 -9079
- package/web/_astro/{BoardApp.BOW1815F.js → BoardApp.CHQ1lycZ.js} +83 -83
- package/web/_astro/BoardApp.DV9kx0wo.js +1 -0
- package/web/_astro/{TaskDetail.CX6C7dSJ.js → TaskDetail.GKfQJ60c.js} +1 -1
- package/web/_astro/{arc.CwSvH1ji.js → arc.DWEtA3Tx.js} +1 -1
- package/web/_astro/{architectureDiagram-3BPJPVTR.Dxne1RvP.js → architectureDiagram-3BPJPVTR.DB42oWmP.js} +1 -1
- package/web/_astro/{blockDiagram-GPEHLZMM.D9Fb14rC.js → blockDiagram-GPEHLZMM.rhv-zNQV.js} +1 -1
- package/web/_astro/{c4Diagram-AAUBKEIU.DmvWSFvv.js → c4Diagram-AAUBKEIU.Ci4-4VvY.js} +1 -1
- package/web/_astro/channel.BAI6xLeV.js +1 -0
- package/web/_astro/{chunk-2J33WTMH.BCCktIrE.js → chunk-2J33WTMH.Cc9veUgf.js} +1 -1
- package/web/_astro/{chunk-4BX2VUAB.BzCk9q27.js → chunk-4BX2VUAB.Bec9c4eI.js} +1 -1
- package/web/_astro/{chunk-55IACEB6.CZXVgTk4.js → chunk-55IACEB6.DoV8S1iB.js} +1 -1
- package/web/_astro/{chunk-727SXJPM.CJL8UIpX.js → chunk-727SXJPM.DwR-Qlyj.js} +1 -1
- package/web/_astro/{chunk-AQP2D5EJ.Dz8MDEdw.js → chunk-AQP2D5EJ.ND_a81WY.js} +1 -1
- package/web/_astro/{chunk-FMBD7UC4.CkQeYYUw.js → chunk-FMBD7UC4.Wv_jwG48.js} +1 -1
- package/web/_astro/{chunk-ND2GUHAM.KiC1QgzH.js → chunk-ND2GUHAM.CXKXCMmp.js} +1 -1
- package/web/_astro/{chunk-QZHKN3VN.AJ08mw2e.js → chunk-QZHKN3VN.nkaoNYQq.js} +1 -1
- package/web/_astro/{classDiagram-4FO5ZUOK.DRWRDKzK.js → classDiagram-4FO5ZUOK.cMQcVlQu.js} +1 -1
- package/web/_astro/{classDiagram-v2-Q7XG4LA2.DRWRDKzK.js → classDiagram-v2-Q7XG4LA2.cMQcVlQu.js} +1 -1
- package/web/_astro/{cose-bilkent-S5V4N54A.CavydfLP.js → cose-bilkent-S5V4N54A.OaDJ7Mr2.js} +1 -1
- package/web/_astro/{cynefin-OW5HDTMX.D_7o_a0B.js → cynefin-OW5HDTMX.Chi8IphF.js} +1 -1
- package/web/_astro/{dagre-BM42HDAG.LWk2dKgg.js → dagre-BM42HDAG.CzK2t_Fp.js} +1 -1
- package/web/_astro/{diagram-2AECGRRQ.82lbq6aB.js → diagram-2AECGRRQ.DRvxlVS7.js} +1 -1
- package/web/_astro/{diagram-5GNKFQAL.hyNL1pwY.js → diagram-5GNKFQAL.CnYvNdwA.js} +1 -1
- package/web/_astro/{diagram-KO2AKTUF.BaLqVf-b.js → diagram-KO2AKTUF.CpLpMw5R.js} +1 -1
- package/web/_astro/{diagram-LMA3HP47.DQJxj0oz.js → diagram-LMA3HP47.JTb78qUA.js} +1 -1
- package/web/_astro/{diagram-OG6HWLK6.DhlL8cG2.js → diagram-OG6HWLK6.Bk-1jDIb.js} +1 -1
- package/web/_astro/{erDiagram-TEJ5UH35.HoTnXwkF.js → erDiagram-TEJ5UH35.D8hN9GZq.js} +1 -1
- package/web/_astro/{flowDiagram-I6XJVG4X.DkFKlZIZ.js → flowDiagram-I6XJVG4X.-6zQr6m5.js} +1 -1
- package/web/_astro/{ganttDiagram-6RSMTGT7.9qPfOTDb.js → ganttDiagram-6RSMTGT7.DboLQ9ca.js} +1 -1
- package/web/_astro/{gitGraphDiagram-PVQCEYII.B4G18Dwc.js → gitGraphDiagram-PVQCEYII.4tYvJKGR.js} +1 -1
- package/web/_astro/{infoDiagram-5YYISTIA.e3KJkXAM.js → infoDiagram-5YYISTIA.Bd9rXpsB.js} +1 -1
- package/web/_astro/{ishikawaDiagram-YF4QCWOH.CY3yddhD.js → ishikawaDiagram-YF4QCWOH.CvMoaf67.js} +1 -1
- package/web/_astro/{journeyDiagram-JHISSGLW.DaAKIO1t.js → journeyDiagram-JHISSGLW.Ccy1CA7y.js} +1 -1
- package/web/_astro/{kanban-definition-UN3LZRKU.DGkDddtc.js → kanban-definition-UN3LZRKU.0MaMqHNS.js} +1 -1
- package/web/_astro/{linear.DnDPcpd1.js → linear.CHXgcIbN.js} +1 -1
- package/web/_astro/{mermaid.core.1uBmxa9t.js → mermaid.core.Ca-kcelG.js} +4 -4
- package/web/_astro/{mindmap-definition-RKZ34NQL.XFJVayxw.js → mindmap-definition-RKZ34NQL.BUIDlHa0.js} +1 -1
- package/web/_astro/{pieDiagram-4H26LBE5.BXm2OgHh.js → pieDiagram-4H26LBE5.2dX3CU1s.js} +1 -1
- package/web/_astro/{quadrantDiagram-W4KKPZXB.DgB2p5fc.js → quadrantDiagram-W4KKPZXB.B3LBlRiv.js} +1 -1
- package/web/_astro/{requirementDiagram-4Y6WPE33.jBR39bnS.js → requirementDiagram-4Y6WPE33.X12I2uNx.js} +1 -1
- package/web/_astro/{sankeyDiagram-5OEKKPKP.N3u9OhE3.js → sankeyDiagram-5OEKKPKP.BXohIHqx.js} +1 -1
- package/web/_astro/{sequenceDiagram-3UESZ5HK.CZhqZXNL.js → sequenceDiagram-3UESZ5HK.C37ZIUzg.js} +1 -1
- package/web/_astro/{stateDiagram-AJRCARHV.DlNu1VEa.js → stateDiagram-AJRCARHV.BRgz317z.js} +1 -1
- package/web/_astro/{stateDiagram-v2-BHNVJYJU.BEqi8NQV.js → stateDiagram-v2-BHNVJYJU.7VYSXN9-.js} +1 -1
- package/web/_astro/{timeline-definition-PNZ67QCA.BPcexclc.js → timeline-definition-PNZ67QCA.BVNz_HiN.js} +1 -1
- package/web/_astro/{vennDiagram-CIIHVFJN.BZalxKGQ.js → vennDiagram-CIIHVFJN.CHVDkPX4.js} +1 -1
- package/web/_astro/{wardleyDiagram-YWT4CUSO.BqOljod9.js → wardleyDiagram-YWT4CUSO.EQQ_qT9v.js} +1 -1
- package/web/_astro/{xychartDiagram-2RQKCTM6.tkO4ppN5.js → xychartDiagram-2RQKCTM6.DrAT9WoP.js} +1 -1
- package/web/index.html +1 -1
- package/web/_astro/BoardApp.C4zimv1U.js +0 -1
- package/web/_astro/channel.CQHdDVp9.js +0 -1
|
@@ -110,12 +110,21 @@ For answer files, emit a matching parseable table:
|
|
|
110
110
|
`Verdict: PARTIAL` first, append one complete row at a time, and replace the first verdict line only
|
|
111
111
|
after every row is certified. `verify-answer-lint.ts` gates the file before `spur task verdict
|
|
112
112
|
--from-answer` and rejects, with row-level diagnostics: missing/duplicate/unknown requirement IDs,
|
|
113
|
-
AC ids that do not
|
|
114
|
-
linked feature scenario title,
|
|
113
|
+
AC ids that do not resolve to one accepted identity — a task AC checklist label or its declared
|
|
114
|
+
`AC-N`/checklist-token alias, or a linked feature scenario title, in the ac-style-guide forms
|
|
115
|
+
(exact/bare title, `Scenario:` prefix, bracket tags, `AC-N`); paraphrases and ambiguous aliases
|
|
116
|
+
fail — invalid status (`MET | PARTIAL | UNMET` for requirements;
|
|
115
117
|
`N/A` additionally allowed for AC), invalid evidence type (`test | command | static-ref |
|
|
116
118
|
manual-review | llm-judge | n/a`, or a `+` compound), and empty evidence. Interrupted runs keep the
|
|
117
119
|
rows that pass the lint and complete only the missing IDs on retry.
|
|
118
120
|
|
|
121
|
+
**Concrete anchors only (task 0804 R9).** An evidence anchor must be a concrete existing `file:line`
|
|
122
|
+
(or `file:start-end`) path. A glob or directory summary (`src/services/*.ts`, `the retry
|
|
123
|
+
classifiers in task-pipeline.yaml`) is not an anchor: expand it into the specific cited files/ranges
|
|
124
|
+
the run actually verified. Since 0804 R9 the checker ignores complete parsed citation spans before
|
|
125
|
+
scanning for subjects, so a citation's filename (including snake_case paths) can never become a
|
|
126
|
+
false subject — a real absent symbol, nonexistent file, or invalid range still reports.
|
|
127
|
+
|
|
119
128
|
## Checks evidence
|
|
120
129
|
|
|
121
130
|
Wave C verification can emit the following additive `checks[]` rows:
|
|
@@ -39,7 +39,8 @@ sinks; this skill owns the protocol.
|
|
|
39
39
|
testee (a /sp:... command, Skill(...), or shell CLI invocation)
|
|
40
40
|
→ PLAN classify + derive steps + open dual artifacts (live + docs/dogfood) with status:running
|
|
41
41
|
→ EXECUTE run each step as a user; on failure, bounded diagnose→fix→re-run (or observe-only)
|
|
42
|
-
→ MONITOR
|
|
42
|
+
→ MONITOR live-ledger row to disk every step resolve; mirror frozen inside a proof window —
|
|
43
|
+
never reconstruct from memory
|
|
43
44
|
→ REPORT finalize-or-abort (non-skippable): status complete|aborted, Cost block, both paths, footer
|
|
44
45
|
```
|
|
45
46
|
|
|
@@ -75,8 +76,11 @@ The command forwards these via `$ARGUMENTS`:
|
|
|
75
76
|
> independent mutation sources is present:
|
|
76
77
|
>
|
|
77
78
|
> - **Pipeline-driving testees** — tokens
|
|
78
|
-
>
|
|
79
|
-
> `runall`, `wrapall`, `run`, `wrap`, `idea
|
|
79
|
+
> <!-- pipeline-tokens:start -->
|
|
80
|
+
> [`--next`, `dev-runall`, `dev-wrapall`, `dev-refineall`, `dev-verifyall`, `dev-run`, `dev-wrap`, `dev-idea`,
|
|
81
|
+
> `refineall`, `verifyall`, `runall`, `wrapall`, `run`, `wrap`, `idea`]
|
|
82
|
+
> <!-- pipeline-tokens:end -->
|
|
83
|
+
> matched as a **distinct hyphen-word**
|
|
80
84
|
> (machine-checked by
|
|
81
85
|
> [`detectPipelineDriving`](../../scripts/dogfood-testing/detect-pipeline-driving.ts);
|
|
82
86
|
> see [§Pipeline-driving word-boundary contract](#pipeline-driving-word-boundary-contract))
|
|
@@ -102,9 +106,8 @@ The command forwards these via `$ARGUMENTS`:
|
|
|
102
106
|
```
|
|
103
107
|
|
|
104
108
|
- Exit **2** → print the stdout refuse line and **stop** (do not plan). The CLI refuses on either
|
|
105
|
-
of two independent mutation sources (task 0293);
|
|
106
|
-
|
|
107
|
-
- mutating `--fix`: `⚠ mutating --fix mode detected (--fix all | --fix blockers-first); pass --max-retry 0 (observe-only for the driver; the testee still mutates the tree) or --max-retry N (fix mode, driver + testee both mutate)`.
|
|
109
|
+
of two independent mutation sources (task 0293); its two refuse lines are the ones quoted
|
|
110
|
+
verbatim in the repo-mutation warning above.
|
|
108
111
|
- Exit **0** → proceed. Do not auto-substitute `--max-retry 0`.
|
|
109
112
|
- The matcher contract is unit-checked by `tests/dogfood-testing/pipeline-detect.test.ts`.
|
|
110
113
|
See [§Pipeline-driving word-boundary contract](#pipeline-driving-word-boundary-contract) and
|
|
@@ -183,17 +186,18 @@ report — the report is assembled from the files, not from memory.
|
|
|
183
186
|
On **every** step resolve:
|
|
184
187
|
|
|
185
188
|
1. Append/update the ledger row on the **live** file first.
|
|
186
|
-
2. Mirror the same row to the **report** path under `docs/dogfood
|
|
189
|
+
2. Mirror the same row to the **report** path under `docs/dogfood/` — unconditional **outside** a
|
|
190
|
+
pipeline proof window; **inside** one the mirror stays **frozen** (live rows only) until the
|
|
191
|
+
window closes, then sync/validate with live-based recovery (task 0804 R2 —
|
|
192
|
+
[monitor-ledger.md](references/monitor-ledger.md) live-ledger rule 3).
|
|
187
193
|
3. Do **not** batch rows until Phase 4.
|
|
188
194
|
|
|
189
|
-
The final report MUST include a `### 3. Monitor Ledger` section containing those rows
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
**[monitor-ledger.md](references/monitor-ledger.md)**. Apply the conservation discipline while
|
|
196
|
-
monitoring — low cache% is usually the driver re-fetching data it already holds.
|
|
195
|
+
The final report MUST include a `### 3. Monitor Ledger` section containing those rows (cardinality:
|
|
196
|
+
row count == the declared executed steps). Cardinality, full methodology, column contract,
|
|
197
|
+
token/cache estimation, multi-source Cost honesty, the cache-health finding rule, and the
|
|
198
|
+
**cache-conservation discipline** live in
|
|
199
|
+
**[monitor-ledger.md](references/monitor-ledger.md)** — apply conservation while monitoring; low
|
|
200
|
+
cache% is usually the driver re-fetching data it already holds.
|
|
197
201
|
|
|
198
202
|
## Phase 4 — Report (finalize-or-abort — non-skippable)
|
|
199
203
|
|
|
@@ -211,8 +215,9 @@ contract violation**.
|
|
|
211
215
|
declared in §2 (N/A steps documented explicitly as rows). A mismatch refuses `complete`.
|
|
212
216
|
4. Write the **Cost** block under §2 (ledger `~estimate` + Method + confidence; `Meter: n/a` or
|
|
213
217
|
optional ccusage/agent usage when real). For any `chained:<step>` ledger row whose meter is not
|
|
214
|
-
observable, Fresh/Cached MUST be `~unknown`
|
|
215
|
-
|
|
218
|
+
observable, Fresh/Cached MUST be `~unknown` — excluded from the cache% aggregate (or surfaced as
|
|
219
|
+
a separate unknown bucket), never counted as `Cached ~0` — **and** emit finding
|
|
220
|
+
`P3 — chained-step cost not observable`; never invent chained totals.
|
|
216
221
|
5. **R2 drift check at finalize.** If a workspace fingerprint was recorded in Phase 1, re-take
|
|
217
222
|
the snapshot and diff against baseline minus the run's own touched files. If drift is detected,
|
|
218
223
|
append a `drift:external` warning row to the ledger and emit a mandatory P2 report finding
|
|
@@ -289,13 +294,15 @@ Do **not** use this skill for:
|
|
|
289
294
|
mutating pipeline — pass `--max-retry 0` first and inspect the findings before letting it apply
|
|
290
295
|
fixes.
|
|
291
296
|
2. **The ledger is live on disk, not reconstructed.** Honest fixed-vs-unresolved accounting depends
|
|
292
|
-
on
|
|
293
|
-
fiction. Working-memory-only ledgers are a contract
|
|
297
|
+
on writing each step *as it happens* to the live file (mirror per Phase 3 — frozen inside a proof
|
|
298
|
+
window). Reconstructing at the end produces fiction. Working-memory-only ledgers are a contract
|
|
299
|
+
violation.
|
|
294
300
|
3. **A hiding fix is a finding.** If "fixing" a step would mask the bug, log it as a finding and
|
|
295
301
|
leave the step unresolved.
|
|
296
302
|
4. **Token numbers are estimates, but cache math is not free-form.** A skill cannot read its own
|
|
297
303
|
exact token meter, so label numbers `~estimate` and put Method + confidence in the Cost block;
|
|
298
|
-
however, cache% must be recomputable from Monitor Ledger row sums
|
|
304
|
+
however, cache% must be recomputable from Monitor Ledger row sums of observable rows (`~unknown`
|
|
305
|
+
rows are excluded from the aggregate, never counted as `~0` cached). Never invent or reuse a fixed
|
|
299
306
|
percentage. Optional meters (`ccusage`, agent usage) are session/day scope — never fake per-step.
|
|
300
307
|
5. **Testee-scoped `--agent`.** Don't confuse the driver agent (always current) with the testee
|
|
301
308
|
agent (the forwarded value).
|
|
@@ -335,8 +342,8 @@ node "$(superskill script path sp dogfood-testing/detect-pipeline-driving.mjs)"
|
|
|
335
342
|
|
|
336
343
|
| Token shape | Examples | Matches | Rejects |
|
|
337
344
|
|-------------|----------|---------|---------|
|
|
338
|
-
| Flag / complete |
|
|
339
|
-
| Bare noun |
|
|
345
|
+
| Flag / complete | <!-- pipeline-tokens:start -->`--next`, `dev-run`, `dev-runall`, `dev-wrap`, `dev-wrapall`, `dev-idea`, `dev-refineall`, `dev-verifyall`<!-- pipeline-tokens:end --> | `/sp:dev-run 0125`, bare `--next` | `--next-gen`, `dev-runner` |
|
|
346
|
+
| Bare noun | <!-- pipeline-tokens:start -->`run`, `runall`, `wrap`, `wrapall`, `idea`, `refineall`, `verifyall`<!-- pipeline-tokens:end --> | `task run 0042` | `runaway`, `wrapper`, `idealist` |
|
|
340
347
|
|
|
341
348
|
`-` is a **word character** for boundaries: a token must be a distinct hyphen-word. Contract tests:
|
|
342
349
|
`plugins/sp/tests/dogfood-testing/pipeline-detect.test.ts`. Helpers:
|
|
@@ -428,11 +435,10 @@ baseline is drift.
|
|
|
428
435
|
### Worktree advisory (mutating dogfoods)
|
|
429
436
|
|
|
430
437
|
For fix-mode dogfoods of **pipeline-driving** or **mutating-`--fix`** testees (the two refuse-gate
|
|
431
|
-
cases above),
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
435
|
-
preferred setup for mutating dogfoods where the operator cares about clean attribution.
|
|
438
|
+
cases above), run in an **isolated `git worktree`** — **advisory, not a hard gate**. Phase 1 must
|
|
439
|
+
print this advisory whenever the driver or testee may mutate and the tree is dirty, **or** the
|
|
440
|
+
testee drives a pipeline (incl. observe-only over a mutating testee). Rationale, trigger matrix
|
|
441
|
+
and wording: `references/monitor-ledger.md`.
|
|
436
442
|
|
|
437
443
|
## Step-splitting recipe (implement-heavy pipeline dogfoods)
|
|
438
444
|
|
|
@@ -483,8 +489,9 @@ Rules:
|
|
|
483
489
|
2. When the chained step ran in a subagent or session whose usage data the driver cannot read, label
|
|
484
490
|
the chained row `~unknown` and emit a **P3** finding: "chained-step cost not observable — candidate
|
|
485
491
|
for surfacing subagent usage in the driver context." Do not invent a number.
|
|
486
|
-
3.
|
|
487
|
-
|
|
492
|
+
3. An observable chained row counts toward the aggregate cache%. A `~unknown` chained row is
|
|
493
|
+
**excluded** from the aggregate (or surfaced as a separate unknown bucket) — never folded in as
|
|
494
|
+
`Cached = ~0` — per the anti-fiction rule in
|
|
488
495
|
[monitor-ledger.md](references/monitor-ledger.md).
|
|
489
496
|
|
|
490
497
|
## `--next` chain stop-at-testing
|
|
@@ -515,18 +522,17 @@ Do NOT:
|
|
|
515
522
|
driver permission to read the chained leg's named artifacts (`.spur/run/<wbs>-verdict.json`,
|
|
516
523
|
task-file section diffs, review tables) after the leg completes and attribute normally. The flag
|
|
517
524
|
licenses **reading** chained-leg evidence that already exists — it does NOT license the driver to
|
|
518
|
-
execute the chained leg itself. The legacy "operator may direct" prose direction
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
[§Arguments](#arguments).
|
|
525
|
+
execute the chained leg itself. The legacy "operator may direct" prose direction stays honored for
|
|
526
|
+
back-compat; omitting the flag keeps stop-at-testing as the **default**. The flag is a driver
|
|
527
|
+
attribute only — it does not change `detect-pipeline-driving` gate semantics (it is not a testee
|
|
528
|
+
mutation source). See [§Arguments](#arguments).
|
|
523
529
|
|
|
524
530
|
## Additional Resources
|
|
525
531
|
|
|
526
|
-
- [references/report-template.md](references/report-template.md) —
|
|
527
|
-
mandatory
|
|
528
|
-
- [references/monitor-ledger.md](references/monitor-ledger.md) —
|
|
529
|
-
token/cache estimation
|
|
532
|
+
- [references/report-template.md](references/report-template.md) — report section contract,
|
|
533
|
+
mandatory footer, task-sink L3 rule.
|
|
534
|
+
- [references/monitor-ledger.md](references/monitor-ledger.md) — live-ledger column contract,
|
|
535
|
+
token/cache estimation, cache-health finding rule.
|
|
530
536
|
|
|
531
537
|
## Platform Notes
|
|
532
538
|
|
|
@@ -562,8 +568,10 @@ shape just because `report-template.md` wasn't auto-loaded.
|
|
|
562
568
|
- Report: `docs/dogfood/YYYY-MM-DD-<testee-slug>-dogfood.md`
|
|
563
569
|
|
|
564
570
|
Both start with YAML frontmatter including `status: running | aborted | complete`, `run_id`,
|
|
565
|
-
`protocol: sp:dogfood-testing@1.2`, and paths.
|
|
566
|
-
resolve
|
|
571
|
+
`protocol: sp:dogfood-testing@1.2`, and paths. Write each ledger row to the live file on every step
|
|
572
|
+
resolve; mirror it to the report only outside a proof window — frozen inside one, synced at
|
|
573
|
+
finalize (Phase 3). On stop, set `status` to `complete` or
|
|
574
|
+
`aborted` (finalize-or-abort — non-skippable).
|
|
567
575
|
|
|
568
576
|
**The six mandatory section headings** (in order, each report MUST contain all six):
|
|
569
577
|
|
|
@@ -28,9 +28,14 @@ artifacts):
|
|
|
28
28
|
1. **Open both artifacts in Phase 1**, before the first step runs (frontmatter `status: running` +
|
|
29
29
|
empty ledger table in each).
|
|
30
30
|
2. **Write a row the instant a step resolves** (pass, fixed, unresolved, or N/A) — not after the run.
|
|
31
|
-
3. **Dual-write every step
|
|
32
|
-
**
|
|
33
|
-
|
|
31
|
+
3. **Dual-write every step — EXCEPT inside a proof window (task 0804 R2).** Append/update the row
|
|
32
|
+
on the **live** file first; the live file is SSOT. While a pipeline proof window is open (from
|
|
33
|
+
the first proof capture until the final proof-sensitive action, including done/provenance
|
|
34
|
+
checks), the tracked report mirror stays **frozen**: append each observation to the live ledger
|
|
35
|
+
only, and sync/validate the mirror after the window closes (or after abort). Tracked reports are
|
|
36
|
+
proof-input fingerprint inputs — writing them mid-window churns the fingerprint and voids the
|
|
37
|
+
proof. If the report write fails at finalize, recover by recreating the mirror from the valid
|
|
38
|
+
live content and re-validate; missing live evidence cannot manufacture complete.
|
|
34
39
|
4. **The report reads the on-disk ledger, not your memory.** Every number in the report traces to a
|
|
35
40
|
ledger row on disk. If it is not in the ledger file, it does not go in the report.
|
|
36
41
|
5. **Cardinality (@1.2).** The ledger's data-row count MUST equal the `**Steps:** N derived, N executed` declared
|
|
@@ -44,6 +49,19 @@ artifacts):
|
|
|
44
49
|
a `FIXED` / `PASS` outcome — it is purely documentary. Cache columns carry `—` (not estimated).
|
|
45
50
|
See [SKILL.md §Workspace-drift guard](../SKILL.md#workspace-drift-guard-r2--task-0296).
|
|
46
51
|
|
|
52
|
+
### Worktree advisory — planning-time surfacing (task 0804 R5)
|
|
53
|
+
|
|
54
|
+
The worktree advisory (SKILL.md §Worktree advisory) is surfaced at planning time, not only when
|
|
55
|
+
drift is detected. Phase 1 prints it whenever the **driver or the testee may mutate** and the tree
|
|
56
|
+
is dirty (`git status --porcelain` non-empty) **or** the testee itself drives a pipeline —
|
|
57
|
+
including observe-only driver mode over a mutating testee. The advisory is **not a hard gate** (the
|
|
58
|
+
refuse-gate semantics from task 0293 are unchanged); the isolated checkout is simply the preferred
|
|
59
|
+
setup for mutating dogfoods because no concurrent writer can reach it, so attribution stays clean
|
|
60
|
+
— which is why the §Mutating `--fix` mode contract recommends it for those two refuse-gate cases.
|
|
61
|
+
Dirtiness alone is not proof of a concurrent writer and must not be reported as one; known
|
|
62
|
+
concurrent writes still follow the project's one-writer rule. A clean, read-only run adds no
|
|
63
|
+
warning.
|
|
64
|
+
|
|
47
65
|
### Fast-run exemption (task 0294 R6a)
|
|
48
66
|
|
|
49
67
|
The per-step live-write mandate (rules 1–4) exists to bound information loss when a mid-run crash
|
|
@@ -81,7 +99,7 @@ in the report's §6 Findings (no exemption applies).
|
|
|
81
99
|
| `Finding` | One-line finding surfaced at this step, or `—`. A finding does **not** change `Outcome`. |
|
|
82
100
|
| `Fresh Tokens` | Estimated fresh context for the step. Prefix with `~`. |
|
|
83
101
|
| `Cached Tokens` | Estimated reused context for the step. Prefix with `~`. |
|
|
84
|
-
| `Cache %` | `Cached Tokens / (Fresh Tokens + Cached Tokens)`, rounded to the nearest whole percent. |
|
|
102
|
+
| `Cache %` | `Cached Tokens / (Fresh Tokens + Cached Tokens)`, rounded to the nearest whole percent. An `~unknown` row carries `—`, never `0%` — unknown basis is not an observed zero. |
|
|
85
103
|
| `Basis` | Observable basis for the estimate: command output, prior file read reused, generated text, etc. |
|
|
86
104
|
| `Wall-clock` | Elapsed time for the step. |
|
|
87
105
|
|
|
@@ -98,8 +116,10 @@ A skill **cannot read its own exact token meter** — derive an estimate and lab
|
|
|
98
116
|
reused by reference in this step. Use the same `ceil(characters / 4)` basis and round to the
|
|
99
117
|
nearest 100. Do not count fresh command output, newly read files, or regenerated scaffolding as
|
|
100
118
|
cached.
|
|
101
|
-
3. Compute each row: `Cache % = round(Cached Tokens / (Fresh Tokens + Cached Tokens) * 100)`.
|
|
102
|
-
|
|
119
|
+
3. Compute each row: `Cache % = round(Cached Tokens / (Fresh Tokens + Cached Tokens) * 100)`. A row
|
|
120
|
+
whose basis is unknown carries `—`, never `0%`.
|
|
121
|
+
4. Compute the report aggregate from **observable rows only** — `~unknown` rows are excluded from
|
|
122
|
+
both sums (or surfaced as a separate unknown bucket), never folded in as `Cached ~0`:
|
|
103
123
|
`aggregate cache% = round(sum(Cached Tokens) / sum(Fresh Tokens + Cached Tokens) * 100)`.
|
|
104
124
|
|
|
105
125
|
The **trend across runs** is the signal, not the absolute value: rising cache% = the testee is
|
|
@@ -132,8 +152,10 @@ step stays on the driver's own row.
|
|
|
132
152
|
- Observable chained usage (subagent output in driver context, or the operator explicitly provided
|
|
133
153
|
the artifact) → estimate Fresh/Cached from that output normally.
|
|
134
154
|
- Unobservable chained usage (subagent ran in a different session, usage data never surfaced) →
|
|
135
|
-
label Fresh `~unknown`, Cached `~
|
|
136
|
-
|
|
155
|
+
label Fresh `~unknown`, Cached `~unknown`, Basis `chained-leg usage not observable from driver`,
|
|
156
|
+
and **exclude the row from the aggregate cache%** (or surface it as a separate unknown bucket) —
|
|
157
|
+
unknown cache use is not an observed zero. **MUST** emit a P3 finding: `P3 — chained-step cost
|
|
158
|
+
not observable` (task 0278 R3). Do not invent totals.
|
|
137
159
|
|
|
138
160
|
Never fold a chained row into the driver's row; the whole point of dogfooding a pipeline-driving
|
|
139
161
|
testee is to see the testee's own cost separately from the driver's monitoring cost. See
|
|
@@ -142,8 +164,10 @@ testee is to see the testee's own cost separately from the driver's monitoring c
|
|
|
142
164
|
## Anti-fiction rule
|
|
143
165
|
|
|
144
166
|
Never reuse a convenient cache percentage such as `45%` because it "feels right." A cache percentage
|
|
145
|
-
is valid only when it can be recomputed from the ledger row sums. If the basis is
|
|
146
|
-
|
|
167
|
+
is valid only when it can be recomputed from the ledger row sums of observable rows. If the basis is
|
|
168
|
+
missing, mark the row `~unknown`, exclude it from the aggregate (or surface it as a separate unknown
|
|
169
|
+
bucket), and explain the missing basis — never fold it in as `Cached Tokens = ~0`: unknown cache use
|
|
170
|
+
is not an observed zero-percent hit rate, and a low-cache diagnosis needs observed data.
|
|
147
171
|
|
|
148
172
|
## Cache-health finding rule
|
|
149
173
|
|
|
@@ -27,7 +27,7 @@ Every dogfood run **always** writes **two** files — with or without `--save`:
|
|
|
27
27
|
| Artifact | Path | Role |
|
|
28
28
|
| ---------- | ------ | ------ |
|
|
29
29
|
| **Live** | `.spur/run/dogfood/<run_id>.md` | Mid-run SSOT; opened in Phase 1; ledger rows appended on every step resolve |
|
|
30
|
-
| **Report** | `docs/dogfood/YYYY-MM-DD-<testee-slug>-dogfood.md` | Operator artifact; same content promoted on open + every step + finalize |
|
|
30
|
+
| **Report** | `docs/dogfood/YYYY-MM-DD-<testee-slug>-dogfood.md` | Operator artifact; same content promoted on open + every step + finalize — **except inside a pipeline proof window (task 0804 R2): the mirror stays frozen (live ledger only) until the window closes, then sync/validate, recovering from live if the write failed** — [monitor-ledger.md](monitor-ledger.md) → live-ledger rule 3 |
|
|
31
31
|
|
|
32
32
|
`--save` is **back-compat no-op** for delivery: it still documents/prints the report path but is
|
|
33
33
|
**not required** to create the file. A run that ends with no file under `docs/dogfood/` (and no live
|
|
@@ -158,14 +158,17 @@ it is the audit trail for step outcomes, fix attempts, findings, and cache math.
|
|
|
158
158
|
|------|----------|---------|-------------|---------|--------------|---------------|---------|-------|------------|
|
|
159
159
|
| resolve | 1 | PASS | — | — | ~800 | ~300 | 27% | 1 command + reused task summary | ~3s |
|
|
160
160
|
|
|
161
|
-
**Cache calculation:** aggregate cache% = round((sum(Cached Tokens) / sum(Fresh Tokens + Cached Tokens)) * 100)
|
|
161
|
+
**Cache calculation:** aggregate cache% = round((sum(Cached Tokens) / sum(Fresh Tokens + Cached Tokens)) * 100),
|
|
162
|
+
computed over **observable rows only** — `~unknown` rows are excluded from both sums (or surfaced as
|
|
163
|
+
a separate unknown bucket), never counted as `~0` cached.
|
|
162
164
|
```
|
|
163
165
|
|
|
164
166
|
Ledger rules:
|
|
165
167
|
|
|
166
168
|
- Every executed step gets exactly one row, recorded when the step resolves (**on disk**, both files).
|
|
167
|
-
- `Fresh Tokens` and `Cached Tokens` must be numbers
|
|
168
|
-
from
|
|
169
|
+
- `Fresh Tokens` and `Cached Tokens` must be `~`-prefixed numbers, or `~unknown` when the basis is
|
|
170
|
+
unobservable (that row is then excluded from the aggregate, never counted as `~0` cached);
|
|
171
|
+
`Cache %` must be computed from those two cells, not guessed.
|
|
169
172
|
- `Basis` is mandatory. It names the observable inputs used for the estimate: command output,
|
|
170
173
|
previously-read file reused from context, generated report text, or similar.
|
|
171
174
|
- The aggregate cache line in `#### Cost` under §2 must equal the ledger formula above. If it
|
|
@@ -173,8 +176,9 @@ Ledger rules:
|
|
|
173
176
|
- **Cardinality (@1.2):** the number of ledger data rows MUST equal the `**Steps:** N derived, N executed`
|
|
174
177
|
declared in §2. Steps marked N/A are documented explicitly as their own rows (`Outcome: N/A`);
|
|
175
178
|
an unaccounted step or an extra row refuses `status: complete` at finalize.
|
|
176
|
-
- If the driver cannot make a defensible estimate for a row, write `~
|
|
177
|
-
|
|
179
|
+
- If the driver cannot make a defensible estimate for a row, write `~unknown`, exclude it from the
|
|
180
|
+
aggregate cache% (or surface it as a separate unknown bucket), and explain the missing basis in
|
|
181
|
+
`Basis`; do not fold it in as `~0` cached or invent a stable percentage.
|
|
178
182
|
|
|
179
183
|
### 4. What We Did
|
|
180
184
|
|
|
@@ -9,8 +9,8 @@ see_also:
|
|
|
9
9
|
|
|
10
10
|
Task bodies are edited section-by-section through `spur task update --section <name> --from-file
|
|
11
11
|
<path>`. The write is **file-wins and crash-safe** (atomic write): the named section's body is
|
|
12
|
-
replaced wholesale from the file you point at
|
|
13
|
-
body in a file first.
|
|
12
|
+
replaced wholesale from the file you point at — with one exception, `Q&A`, which appends (see
|
|
13
|
+
below). There is no inline-body flag — always stage the new body in a file first.
|
|
14
14
|
|
|
15
15
|
For **pipeline output**, section authorship is one-writer-per-section (F92 0593 R1):
|
|
16
16
|
`Testing` comes from `spur task record` (deterministic, from a verify verdict artifact — the
|
|
@@ -23,6 +23,8 @@ for `Plan`, `Acceptance Criteria`, hand-authored `Solution`, and any narrative s
|
|
|
23
23
|
|
|
24
24
|
1. **Assemble the full section body** in a temp file. The body is everything *under* the `###`
|
|
25
25
|
heading — do not include the heading line itself; the CLI owns the heading.
|
|
26
|
+
Sub-headings inside the body MUST be `####` or deeper: a `###` in a body parses as a new
|
|
27
|
+
top-level section and trips `L2.disallowed-section` (task 0787 grew 8 phantom sections this way).
|
|
26
28
|
|
|
27
29
|
```bash
|
|
28
30
|
cat > /tmp/review.md <<'EOF'
|
|
@@ -42,7 +44,11 @@ for `Plan`, `Acceptance Criteria`, hand-authored `Solution`, and any narrative s
|
|
|
42
44
|
|
|
43
45
|
3. The whole `### Review` body is now that file's contents. To amend rather than overwrite, read
|
|
44
46
|
the current body (`spur task show 0040`), edit the temp file to the full desired state, and
|
|
45
|
-
replace again — there is no append mode.
|
|
47
|
+
replace again — there is no append mode for ordinary sections.
|
|
48
|
+
|
|
49
|
+
**`Q&A` is the exception.** `--section "Q&A"` APPENDS a timestamped `#### Q&A entry — <ISO>`
|
|
50
|
+
block rather than replacing the section. Start the body with `<!-- qa:replace -->` to replace it
|
|
51
|
+
wholesale.
|
|
46
52
|
|
|
47
53
|
`--section` **requires** `--from-file` (exit `2` otherwise). Section names match the DD-08 headings
|
|
48
54
|
exactly: `Background`, `Requirements`, `Acceptance Criteria`, `Q&A`, `Design`, `Plan`, `Solution`, `Root Cause`, `Testing`, `Review`, `References`, `History`, `Notes` (universal sections are `History`, `References`, `Notes`; `Root Cause` is carried by the `issue` template variant).
|
|
@@ -61,8 +61,10 @@ frontmatter scalar.
|
|
|
61
61
|
to avoid orphaned nested lifecycle runs). **It is not a guard bypass** — the `wip→testing` and
|
|
62
62
|
`testing→done` `check` gates above still run; the CLI evaluates them inline when the FSM guard
|
|
63
63
|
does not. `--force-done` waives the verify **verdict** only, never the section matrix.
|
|
64
|
-
- **Section** (`--section` **requires** `--from-file`):
|
|
65
|
-
|
|
64
|
+
- **Section** (`--section` **requires** `--from-file`): writes the named section body from the file.
|
|
65
|
+
Most sections replace wholesale. Exception: `--section "Q&A"` APPENDS a timestamped
|
|
66
|
+
`#### Q&A entry — <ISO>` block; start the body with `<!-- qa:replace -->` to replace the section
|
|
67
|
+
wholesale. No inline-body flag. Section names: `Background`, `Requirements`, `Acceptance Criteria`, `Q&A`, `Design`, `Plan`, `Solution`, `Testing`, `Review`, `References`, `History`, `Notes`.
|
|
66
68
|
- **Frontmatter** (`--feature <id>`, `--priority <p>`): sets the scalar frontmatter field on an
|
|
67
69
|
existing task — the only post-create path, allow-listed to `feature_id` / `parent_wbs` / `priority`.
|
|
68
70
|
|
|
@@ -85,7 +87,7 @@ Flags: `--folder <path>`, `--json`. Exit codes: `0` success, `1` error, `2` usag
|
|
|
85
87
|
|
|
86
88
|
## `sections <wbs> <op> [name]`
|
|
87
89
|
|
|
88
|
-
CLI-safe, matrix-enforced task section mutation. Section names are validated against canonical sections (`Background`, `Requirements`, `Acceptance Criteria`, `Q&A`, `Design`, `Plan`, `Solution`, `Root Cause`, `Testing`, `Review`, `References`, `History`, `Notes`). Universal sections (`History`, `References`, `Notes`) are always allowed; `Root Cause` is carried by the `issue` template variant.
|
|
90
|
+
CLI-safe, matrix-enforced task section mutation. Section names are validated against canonical sections (`Background`, `Requirements`, `Acceptance Criteria`, `Q&A`, `Design`, `Plan`, `Solution`, `Root Cause`, `Testing`, `Review`, `References`, `History`, `Notes`). Universal sections (`History`, `References`, `Notes`) are always allowed; `Root Cause` is carried by the `issue` template variant. `Q&A` on `update --section` appends rather than replacing (see the Section bullet above); `<!-- qa:replace -->` forces a wholesale replace.
|
|
89
91
|
|
|
90
92
|
| Op | Usage | Description |
|
|
91
93
|
| --- | --- | --- |
|
|
@@ -209,7 +209,7 @@ spur workflow run ./workflows/approval.yaml --steer # interactive
|
|
|
209
209
|
are exclusive (exit `2`); `--silent` cannot combine with either (exit `2`).
|
|
210
210
|
- **`--detail <level>`** sets human verbosity: `minimal` (state changes only), `invocation` (default;
|
|
211
211
|
per-step headers), `full` (transitions + correlation). `--verbose` is shorthand for `--detail full`.
|
|
212
|
-
- **`--trace-file`** appends a redacted, schema-versioned JSONL trace under `.spur/
|
|
212
|
+
- **`--trace-file`** appends a redacted, schema-versioned JSONL trace under `.spur/workflow/`
|
|
213
213
|
for post-run analysis - independent of human/JSON output.
|
|
214
214
|
- **`--no-log`** opts out of writing the consolidated all-in-one run log (`.spur/run/<RUNID>.log`).
|
|
215
215
|
By default the log is written **and retained** after the run ends; this flag skips it entirely
|
|
@@ -280,7 +280,7 @@ not advertise `--json-envelope` because its JSON projection is a kept-raw docume
|
|
|
280
280
|
| `--silent` | Suppress all routine output; errors still set a non-zero exit status. |
|
|
281
281
|
| `--verbose` | Include transitions and correlation diagnostics in human progress (implies `--detail full`). |
|
|
282
282
|
| `--detail <level>` | Human detail level: `minimal`, `invocation` (default), or `full`. |
|
|
283
|
-
| `--trace-file` | Append a redacted schema-versioned JSONL trace under `.spur/
|
|
283
|
+
| `--trace-file` | Append a redacted schema-versioned JSONL trace under `.spur/workflow/`. |
|
|
284
284
|
| `--no-log` | Opt out of writing the consolidated `.spur/run/<RUNID>.log` (retained by default; propagates to `--async` workers). |
|
|
285
285
|
| `--steer` | Accept in-process steering commands on stdin at declared action boundaries (sync only; incompatible with `--json`/`--async`). |
|
|
286
286
|
|
|
@@ -335,7 +335,7 @@ redirecting `agent.run` stages (ADR-047).
|
|
|
335
335
|
- **Run-log reclamation** (0429): removes retained `.spur/run/<RUNID>.log` files whose mtime is older
|
|
336
336
|
than `workflow.logRetentionDays` in `.spur/config.yaml` (default 30 days; integer days, not minutes).
|
|
337
337
|
Age is the only gate; best-effort deletes never abort the rest. Never touches
|
|
338
|
-
`.spur/
|
|
338
|
+
`.spur/workflow/<RUNID>.jsonl` or `*-partial.md`.
|
|
339
339
|
- **`--logs`** scopes to log reclamation only (skips stale-run finalization). `--dry-run` applies to
|
|
340
340
|
both scopes (lists what would be removed, writes nothing). `--json` returns
|
|
341
341
|
`{ olderThanMinutes, dryRun, cleaned, logs: { retentionDays, dryRun, reclaimed, failures } }` (with
|
|
@@ -237,11 +237,11 @@ must not be changed without updating the backing skill.
|
|
|
237
237
|
- **Inputs:**
|
|
238
238
|
- `--feature <id>` **or** `--tasks <selector>` (required — at least one). `--feature` is sugar for `--tasks feature:<id>` (shared selector grammar: explicit WBS list, `feature:<id>`, `ready`, status pseudo-list — [execution-batch.md](execution-batch.md) Step 1). If both are present, `--tasks` wins (one-line note in the report).
|
|
239
239
|
- Shared refine flags (passed through to each per-task refine): `--focus <mode>`, `--description <text>`, `--depth <standard|ready>`, `--agent <inline|auto|name>`, `--auto`.
|
|
240
|
-
- Batch-only flags: `--keep-going` (continue independents after a failure; default halt), `--status <s>` (filter resolved membership; default **`backlog
|
|
240
|
+
- Batch-only flags: `--keep-going` (continue independents after a failure; default halt), `--status <s>` (filter resolved membership; default **`backlog` + `todo`** — planning-side fill candidates. The filter is applied in-agent against the frozen set; `spur task list --status` takes exactly one canonical status per call (see `execution-batch.md` Step 1).), `--json` (machine-readable batch report).
|
|
241
241
|
- **Backing:** `sp:spur-dev` skill, `refineall` operation (orchestrates; per-task body is the single-task `refine` operation — never a second refine implementation).
|
|
242
242
|
- **Behavior:**
|
|
243
243
|
1. Resolve + **freeze** the set at kickoff (never re-query membership mid-batch).
|
|
244
|
-
2. Apply `--status` filter (default `backlog
|
|
244
|
+
2. Apply `--status` filter (default `backlog` + `todo`; applied in-agent against the frozen set; `spur task list --status` takes exactly one canonical status per call — see `execution-batch.md` Step 1). Tasks already `done`/`cancelled`/`testing` are excluded unless the operator widens `--status`. Report each exclusion with reason.
|
|
245
245
|
3. Topo-sort by `dependencies[]` (Kahn, WBS-ascending tie-break). Cycle → abort entire batch before any refine. Out-of-set deps: `done` → allow; else → block subtree (same as runall).
|
|
246
246
|
4. For each WBS in order: invoke single-task refine with shared flags **including `--depth`**. Under `--auto` + **`--depth standard`** (default), the per-task **L3 pre-synthesis SKIP gate** still applies. Under **`--depth ready`**, each task runs the implement-ready checklist (no L3-only SKIP).
|
|
247
247
|
5. Failure policy: **stop-the-batch** (default) or `--keep-going` (skip in-batch dependents of a failed refine; continue independents).
|
|
@@ -48,11 +48,28 @@ command, skill, script, or second workflow.
|
|
|
48
48
|
chooses the subprocess workflow path.
|
|
49
49
|
2. Allocate a collision-resistant inline run id (`uuidgen`, with a timestamp/pid fallback), create
|
|
50
50
|
`.spur/run/`, and use `.spur/run/<run-id>.log` as the run log.
|
|
51
|
-
3.
|
|
51
|
+
3. **Authoritative run identity (task 0804 R1, fail-closed).** Persist the run row through the
|
|
52
|
+
internal delegate before any stage executes — this is what makes bound `run.artifact` record
|
|
53
|
+
accept the inline run (0785 R3):
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
SETUP_SCRIPT="plugins/sp/scripts/inline-run-setup.ts";
|
|
57
|
+
[ -f "$SETUP_SCRIPT" ] || SETUP_SCRIPT="$(superskill script path sp inline-run-setup.ts 2>/dev/null)";
|
|
58
|
+
[ -n "$SETUP_SCRIPT" ] && [ -f "$SETUP_SCRIPT" ] && \
|
|
59
|
+
bun "$SETUP_SCRIPT" --run-id "$RUN_ID" --file <selected-pipeline-yaml> \
|
|
60
|
+
|| { echo "inline run setup failed closed — checker not found; run 'superskill install sp'" >&2; exit 1; }
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
The delegate resolves the app service from the SPUR_BIN chain, creates-or-attaches the row,
|
|
64
|
+
and writes `.spur/run/<run-id>-inline-setup.json`. Seed `__runId` and `__definitionDigest`
|
|
65
|
+
from that file so proof capture and bound registration verify against the persisted identity.
|
|
66
|
+
A non-zero exit (missing row identity, changed definition, bundle-only install) stops the run —
|
|
67
|
+
never continue unbound and never fabricate a PASS.
|
|
68
|
+
4. Resolve the host session id from `.spur/context/.session.json`, accepting the normalized hook key
|
|
52
69
|
`session` and the Codex key `session_id` (in that order). If neither is available, allocate
|
|
53
70
|
`host-session-<run-id>` and record that fallback in the log; provenance must never be blank or
|
|
54
71
|
guessed from an executor subprocess.
|
|
55
|
-
|
|
72
|
+
5. Render the two-layer plan into the host todo list (task 0596):
|
|
56
73
|
- **Layer 1** = `spur workflow show <pipeline-yaml> --format todo --json` → its `steps[]`: the
|
|
57
74
|
declared state inventory in declaration order with `initial` / `terminal` / `failure` /
|
|
58
75
|
`pause` / `loopBack` / `conditional` markers. Mark the active state. Never re-derive this
|
|
@@ -69,7 +86,7 @@ command, skill, script, or second workflow.
|
|
|
69
86
|
ended 0/11 with precheck and implement still open).
|
|
70
87
|
- **Source of truth** = the CLI projection for layer 1; the YAML parsed in step 1 for layer 2.
|
|
71
88
|
Never hand-copy or hand-derive the state list into the driver, a command, a skill, or a script.
|
|
72
|
-
|
|
89
|
+
6. For task execution only, record lifecycle provenance before entering the FSM:
|
|
73
90
|
|
|
74
91
|
```bash
|
|
75
92
|
spur task run-link <wbs> --source inline-full --run-id <run-id> --json
|
|
@@ -101,6 +118,16 @@ Action semantics come from the YAML and the workflow action contract:
|
|
|
101
118
|
`answerFile`; assert `expectFile`; enforce `requireDiff` against a pre-action git snapshot,
|
|
102
119
|
including the task-scope guard; honor declared error policy. `timeoutMs` is recorded as not
|
|
103
120
|
applicable because the host session has no independent kill boundary.
|
|
121
|
+
- `run.artifact` — the engine's ledger registration has **no inline execution surface** (0808 R4).
|
|
122
|
+
The inline equivalent is a documented **registration-equivalent convention**: before the record
|
|
123
|
+
state mutates the task, the host validates the same refusal conditions inline — the declared
|
|
124
|
+
artifact exists at the resolved path and is canonical-valid for the run's wbs (for
|
|
125
|
+
`verify-verdict`: verdict `PASS`), `proofBinding: current` is honored against a freshly captured
|
|
126
|
+
proof digest, and the run-scoped review-completion marker exists — then appends one provenance
|
|
127
|
+
line to `.spur/run/<run-id>.log` naming the equivalence (artifact kind, path, verdict, digest) and
|
|
128
|
+
proceeds to `spur task record`. A failed validation stops at the state and follows the failure
|
|
129
|
+
contract; the step is never silently skipped. Artifact-provenance consumers read that run-log
|
|
130
|
+
line on the inline path — there is no ledger row.
|
|
104
131
|
|
|
105
132
|
**Native-subagent dispatch (R2 eligibility, evaluated before each action):**
|
|
106
133
|
|
|
@@ -204,10 +231,11 @@ tasks (0617, 0619) because the sections were hand-written **before** the verdict
|
|
|
204
231
|
and `L3.required-section-placeholder` before the transition, not after.
|
|
205
232
|
3. **Solution change-map anchor rule (L4.anchor-subject-mismatch).** A Solution change-map table must
|
|
206
233
|
list **one `file:line` per row**. A ·-joined paragraph makes every anchor's "subject" the other
|
|
207
|
-
anchors and trips the L4 subject check.
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
234
|
+
anchors and trips the L4 subject check. Since 0804 R9, subject extraction ignores complete parsed
|
|
235
|
+
citation spans, so a path's underscores no longer manufacture a subject: an underscore path row
|
|
236
|
+
(`docs/help/cmd_example.md:12`) is checked exactly like any other row — cite an **existing file**
|
|
237
|
+
with a **valid line or line range** whose content names the requirement's subject. A real absent
|
|
238
|
+
symbol, nonexistent file or invalid range still reports; never replace a citable row with prose.
|
|
211
239
|
|
|
212
240
|
## Failure contract
|
|
213
241
|
|
|
@@ -165,6 +165,10 @@
|
|
|
165
165
|
"type": "string",
|
|
166
166
|
"enum": ["cheap", "standard", "capable-1", "capable-2", "capable-3"],
|
|
167
167
|
"description": "Capability tier for stage-registry adaptive model routing (ADR-033, 0343). Live values: cheap | standard | capable-1 | capable-2 | capable-3 (1=low output quality, 3=high within the capable band). A stage starts on the cheapest eligible executor meeting its model_policy min_tier. Bare legacy `capable` is accepted at runtime (zod preprocess \u2192 capable-1) during the deprecation window but is not part of this editor enum."
|
|
168
|
+
},
|
|
169
|
+
"disabled": {
|
|
170
|
+
"type": "boolean",
|
|
171
|
+
"description": "Routing kill-switch (111): a disabled profile never serves a role, team, stage, or explicit selection, and doctor inventories it without probing. Omitted = enabled; only true/false accepted (no string/number/null coercion)."
|
|
168
172
|
}
|
|
169
173
|
}
|
|
170
174
|
}
|