@gobing-ai/spur 0.3.91 → 0.3.92
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/config/pipeline-budgets.json +2 -2
- package/config/plugin-scripts.json +17 -1
- package/config/workflows/feature-lifecycle.yaml +14 -5
- package/config/workflows/feature-verification.yaml +35 -27
- package/config/workflows/history-anatomy.yaml +14 -25
- package/config/workflows/idea-pipeline.yaml +9 -5
- package/config/workflows/task-pipeline.yaml +185 -4
- package/config/workflows/wrapup-pipeline.yaml +47 -5
- package/package.json +9 -9
- package/plugins/sp/agents/super-planner.md +14 -5
- package/plugins/sp/commands/dev-dogfood.md +4 -4
- package/plugins/sp/commands/dev-fixall.md +8 -5
- package/plugins/sp/commands/dev-run.md +6 -0
- package/plugins/sp/commands/dev-runall.md +12 -6
- package/plugins/sp/commands/dev-verify.md +9 -0
- package/plugins/sp/commands/dev-verifyall.md +5 -0
- package/plugins/sp/lib/idea-handoff.generated.mjs +301 -300
- package/plugins/sp/lib/inline-run.generated.d.mts +17 -0
- package/plugins/sp/lib/inline-run.generated.mjs +1460 -0
- package/plugins/sp/plugin.json +1 -1
- package/plugins/sp/references/environment-lens.md +1 -1
- package/plugins/sp/scripts/dogfood-testing/validate-report.mjs +136 -0
- package/plugins/sp/scripts/dogfood-testing/validate-report.ts +196 -2
- package/plugins/sp/scripts/feature-verification-steps.mjs +174 -0
- package/plugins/sp/scripts/feature-verification-steps.ts +275 -0
- package/plugins/sp/scripts/history-anatomy-cache.mjs +104 -4
- package/plugins/sp/scripts/history-anatomy-cache.ts +137 -13
- package/plugins/sp/scripts/inline-pipeline-parity-check.ts +1 -0
- package/plugins/sp/scripts/inline-run-setup.mjs +349 -0
- package/plugins/sp/scripts/inline-run-setup.ts +192 -75
- package/plugins/sp/scripts/record-feature-sync.mjs +63 -0
- package/plugins/sp/scripts/record-feature-sync.ts +84 -0
- package/plugins/sp/scripts/residual-scan.mjs +476 -0
- package/plugins/sp/scripts/residual-scan.ts +614 -0
- package/plugins/sp/scripts/task-evidence-precheck.ts +8 -3
- package/plugins/sp/scripts/task-size-precheck.ts +8 -3
- package/plugins/sp/skills/branch-workflow/SKILL.md +1 -0
- package/plugins/sp/skills/branch-workflow/references/worktree-patterns.md +2 -0
- package/plugins/sp/skills/code-implementation/SKILL.md +17 -0
- package/plugins/sp/skills/code-verification/SKILL.md +21 -0
- package/plugins/sp/skills/code-verification/references/verdict-schema.md +1 -0
- package/plugins/sp/skills/dogfood-testing/SKILL.md +5 -3
- package/plugins/sp/skills/dogfood-testing/references/monitor-ledger.md +63 -26
- package/plugins/sp/skills/dogfood-testing/references/report-template.md +33 -10
- package/plugins/sp/skills/history-anatomy/references/modes.md +5 -3
- package/plugins/sp/skills/next-feature/references/ranking-rubric.md +1 -1
- package/plugins/sp/skills/next-router/references/routing-table.md +7 -0
- package/plugins/sp/skills/parallel-execution/references/dispatch-surface.md +1 -1
- package/plugins/sp/skills/session-review/SKILL.md +12 -2
- package/plugins/sp/skills/spur-cli/references/features.md +1 -1
- package/plugins/sp/skills/spur-cli/references/projects.md +3 -1
- package/plugins/sp/skills/spur-cli/references/workflows.md +39 -19
- package/plugins/sp/skills/spur-dev/SKILL.md +11 -4
- package/plugins/sp/skills/spur-dev/references/cross-cutting.md +4 -1
- package/plugins/sp/skills/spur-dev/references/dev-operations.md +2 -2
- package/plugins/sp/skills/spur-dev/references/execution-batch.md +285 -57
- package/plugins/sp/skills/spur-dev/references/execution-workflow.md +1 -1
- package/plugins/sp/skills/spur-dev/references/flag-glossary.md +19 -4
- package/plugins/sp/skills/spur-dev/references/inline-pipeline-driver.md +43 -16
- package/plugins/sp/skills/spur-dev/references/planning-workflow.md +5 -0
- package/spur.js +21827 -19615
- package/web/_astro/BoardApp.CerSBgis.js +192 -0
- package/web/_astro/BoardApp.eoTz0pZs.js +1 -0
- package/web/_astro/{TaskDetail.CXGltuT_.js → TaskDetail.DCqiC-OZ.js} +1 -1
- package/web/_astro/arc.CPwg6Rw0.js +1 -0
- package/web/_astro/{architectureDiagram-3BPJPVTR.DJ8DHkWE.js → architectureDiagram-3BPJPVTR.DM_vp_hO.js} +1 -1
- package/web/_astro/{blockDiagram-GPEHLZMM.D8DHK3Jl.js → blockDiagram-GPEHLZMM.DXVIiv0p.js} +1 -1
- package/web/_astro/{c4Diagram-AAUBKEIU.BugQbX9u.js → c4Diagram-AAUBKEIU.BbF_zCxW.js} +1 -1
- package/web/_astro/channel.MYZLKNwy.js +1 -0
- package/web/_astro/{chunk-2J33WTMH.Mv26KlVn.js → chunk-2J33WTMH.CAgQHpPC.js} +1 -1
- package/web/_astro/{chunk-4BX2VUAB.CD51JoT_.js → chunk-4BX2VUAB.BN-5tpw4.js} +1 -1
- package/web/_astro/{chunk-55IACEB6.D_PFaIEe.js → chunk-55IACEB6.CnPkEEr0.js} +1 -1
- package/web/_astro/{chunk-727SXJPM.DemMW1ao.js → chunk-727SXJPM.BQzQeMVm.js} +4 -4
- package/web/_astro/{chunk-AQP2D5EJ.Iu2V5-ex.js → chunk-AQP2D5EJ.B6xNyDnL.js} +1 -1
- package/web/_astro/{chunk-FMBD7UC4.MsNgSP-E.js → chunk-FMBD7UC4.C7f9Ih78.js} +1 -1
- package/web/_astro/{chunk-ND2GUHAM.DfbRaAlm.js → chunk-ND2GUHAM.CNV1dFXT.js} +1 -1
- package/web/_astro/{chunk-QZHKN3VN.BbJAQ4h-.js → chunk-QZHKN3VN.Cudn2TkJ.js} +1 -1
- package/web/_astro/{classDiagram-4FO5ZUOK.CPurtiC2.js → classDiagram-4FO5ZUOK.D1NwP50q.js} +1 -1
- package/web/_astro/{classDiagram-v2-Q7XG4LA2.CPurtiC2.js → classDiagram-v2-Q7XG4LA2.D1NwP50q.js} +1 -1
- package/web/_astro/{cose-bilkent-S5V4N54A.CKKdx1bM.js → cose-bilkent-S5V4N54A.B1wSL-Xb.js} +1 -1
- package/web/_astro/{cynefin-OW5HDTMX.CoKMTg-R.js → cynefin-OW5HDTMX.BmK52w8G.js} +1 -1
- package/web/_astro/{dagre-BM42HDAG.D4h4_k56.js → dagre-BM42HDAG.Bfy5CTDT.js} +2 -2
- package/web/_astro/diagram-2AECGRRQ.DhNnvUvX.js +43 -0
- package/web/_astro/diagram-5GNKFQAL.lTX5KwnS.js +10 -0
- package/web/_astro/{diagram-KO2AKTUF.BFoCkiCr.js → diagram-KO2AKTUF.CW_vMJ4z.js} +3 -3
- package/web/_astro/{diagram-LMA3HP47.exHn9OVx.js → diagram-LMA3HP47.B_8ZGF67.js} +1 -1
- package/web/_astro/{diagram-OG6HWLK6.CeqO34nN.js → diagram-OG6HWLK6.BppnHsdS.js} +1 -1
- package/web/_astro/{erDiagram-TEJ5UH35.D_v7HqxR.js → erDiagram-TEJ5UH35.BEuHXcjJ.js} +5 -5
- package/web/_astro/{flowDiagram-I6XJVG4X.EUmrpbwh.js → flowDiagram-I6XJVG4X.CH-UlnGr.js} +4 -4
- package/web/_astro/{ganttDiagram-6RSMTGT7.BOCF5lII.js → ganttDiagram-6RSMTGT7.BO81S85v.js} +1 -1
- package/web/_astro/{gitGraphDiagram-PVQCEYII.Di7otYZD.js → gitGraphDiagram-PVQCEYII.XnPxPPZN.js} +1 -1
- package/web/_astro/index.Hjbr15fG.css +1 -0
- package/web/_astro/{infoDiagram-5YYISTIA.TcBkCAJk.js → infoDiagram-5YYISTIA.JyjYRu_T.js} +1 -1
- package/web/_astro/{ishikawaDiagram-YF4QCWOH.D-2y4M0c.js → ishikawaDiagram-YF4QCWOH.BBRBF-Fo.js} +5 -5
- package/web/_astro/{journeyDiagram-JHISSGLW.DPbJI_n2.js → journeyDiagram-JHISSGLW.C_iymSyp.js} +1 -1
- package/web/_astro/{kanban-definition-UN3LZRKU.EFxhQ9Fj.js → kanban-definition-UN3LZRKU.DdfW-Oqt.js} +7 -7
- package/web/_astro/{linear.DSAsQLzs.js → linear.C2_IkbZT.js} +1 -1
- package/web/_astro/mermaid.core.GAOYeSR0.js +303 -0
- package/web/_astro/{mindmap-definition-RKZ34NQL.CJY1N_7V.js → mindmap-definition-RKZ34NQL.DAZIxQSK.js} +2 -2
- package/web/_astro/{pieDiagram-4H26LBE5.567ZNoL2.js → pieDiagram-4H26LBE5.CN8sIhKM.js} +3 -3
- package/web/_astro/{quadrantDiagram-W4KKPZXB.mqfz9-MY.js → quadrantDiagram-W4KKPZXB.3dGcX5GP.js} +1 -1
- package/web/_astro/{requirementDiagram-4Y6WPE33.Bv1Gv9In.js → requirementDiagram-4Y6WPE33.BV2y4dd6.js} +3 -3
- package/web/_astro/{sankeyDiagram-5OEKKPKP.B6Gs4X4r.js → sankeyDiagram-5OEKKPKP.Cqo15Tvo.js} +4 -4
- package/web/_astro/{sequenceDiagram-3UESZ5HK.BhYj4v-m.js → sequenceDiagram-3UESZ5HK.CROCPMJB.js} +1 -1
- package/web/_astro/{stateDiagram-AJRCARHV.BPbBnkpw.js → stateDiagram-AJRCARHV.RfXZrkFE.js} +1 -1
- package/web/_astro/{stateDiagram-v2-BHNVJYJU.C4squMNK.js → stateDiagram-v2-BHNVJYJU.CPXmbBs9.js} +1 -1
- package/web/_astro/{timeline-definition-PNZ67QCA.C_SwIHgl.js → timeline-definition-PNZ67QCA.DdgKTiO8.js} +3 -3
- package/web/_astro/{vennDiagram-CIIHVFJN.Bz4NZGpQ.js → vennDiagram-CIIHVFJN.CPNVSHF1.js} +5 -5
- package/web/_astro/{wardleyDiagram-YWT4CUSO.CozMVZ3i.js → wardleyDiagram-YWT4CUSO.CQhA0Jyr.js} +3 -3
- package/web/_astro/{xychartDiagram-2RQKCTM6.BwMGBwjB.js → xychartDiagram-2RQKCTM6.n61BWyy4.js} +1 -1
- package/web/index.html +2 -2
- package/web/_astro/BoardApp.BEDWpzsr.js +0 -188
- package/web/_astro/BoardApp.DQG2xfEz.js +0 -1
- package/web/_astro/arc.C0rrflm_.js +0 -1
- package/web/_astro/channel.SRrg1P-w.js +0 -1
- package/web/_astro/diagram-2AECGRRQ.DJ0h9zgw.js +0 -43
- package/web/_astro/diagram-5GNKFQAL.DvPk1jYd.js +0 -10
- package/web/_astro/index.Bx6GY4RH.css +0 -1
- package/web/_astro/mermaid.core.kAZjgJHG.js +0 -301
|
@@ -92,9 +92,15 @@ pick task (spur task list --json)
|
|
|
92
92
|
The pipeline (`kind: state-machine`) runs the work loop:
|
|
93
93
|
|
|
94
94
|
```
|
|
95
|
-
precheck → implement → test → review → approve(HITL) → verify → record → done
|
|
95
|
+
precheck → implement[→escalate(HITL)] → test → review → approve(HITL) → verify → record → done
|
|
96
96
|
```
|
|
97
97
|
|
|
98
|
+
**Escalation contract (0933).** The implement agent may pause on an operator question: it writes
|
|
99
|
+
`.spur/run/<wbs>-question.md` and exits 0; the `escalate` hop surfaces it via HITL and pauses the
|
|
100
|
+
run. Resume with `spur workflow continue --answer-text <answer>` — the guard appends the Q/A to
|
|
101
|
+
`.spur/run/<wbs>-escalation.md` and re-enters implement (transcript via `--escalation-file`).
|
|
102
|
+
`maxEscalations` (default 2) bounds the loop; exhausted or empty answer → `failed`. Implement only.
|
|
103
|
+
|
|
98
104
|
Full procedure: **[references/execution-workflow.md](references/execution-workflow.md)**.
|
|
99
105
|
Host-session procedure: **[references/inline-pipeline-driver.md](references/inline-pipeline-driver.md)**.
|
|
100
106
|
|
|
@@ -162,9 +168,10 @@ CLI does.
|
|
|
162
168
|
- Before accepting a child/watcher result, compare its run ID with the dispatched run ID and check
|
|
163
169
|
current trace state through Spur. If using a run log as evidence, require its mtime to be at least
|
|
164
170
|
the dispatch time. A mismatch or stale timestamp is not completion evidence. Do not scrape terminals.
|
|
165
|
-
Bound each watch invocation to
|
|
166
|
-
|
|
167
|
-
|
|
171
|
+
Bound each watch invocation to `spur workflow trace <run-id> --follow --timeout 600000` — one
|
|
172
|
+
bounded call, no hand-rolled poll loop; a timeout prints one checkpoint line and exits 1 while the
|
|
173
|
+
run continues, so persist the last confirmed identity/state and report the checkpoint before
|
|
174
|
+
continuing. A watcher timeout does not cancel the owned run or authorize launching a replacement.
|
|
168
175
|
- Mark superseded scratch with `SUPERSEDED` and a pointer to the authoritative task Design. Never
|
|
169
176
|
let a scratch instruction override the live task, even when its old run is still readable.
|
|
170
177
|
- Checker-policy changes require one explicit unsuppressed audit (T10); ordinary corpus commit
|
|
@@ -512,7 +512,10 @@ iterating one task).
|
|
|
512
512
|
|
|
513
513
|
**The rule.** When a test fails and you are iterating to green:
|
|
514
514
|
|
|
515
|
-
1. Run the narrow target first: `bun test <file> --test-name-pattern <test>`.
|
|
515
|
+
1. Run the narrow target first: `bun test <file> --test-name-pattern <test>`. Workspace tests need
|
|
516
|
+
the workspace `bunfig.toml` preload — invoke them as a subshell `(cd <workspace> && bun test …)`
|
|
517
|
+
or with absolute paths, because the shell's cwd persists between calls and a bare `cd` breaks
|
|
518
|
+
later relative-path commands (verifyall session friction, 2026-09-23).
|
|
516
519
|
2. Loop on that narrow target until green.
|
|
517
520
|
3. **Then** run the single full `spur-check` (or `bun run check`) as the final gate.
|
|
518
521
|
|
|
@@ -352,8 +352,8 @@ must not be changed without updating the backing skill.
|
|
|
352
352
|
### 13. runall
|
|
353
353
|
|
|
354
354
|
- **Purpose:** Run a batch of tasks through their pipelines in dependency-correct order — resolve a set, topo-sort, run each via `task-pipeline.yaml`, inspect verdicts, apply the failure policy, emit a batch report.
|
|
355
|
-
- **Inputs:** `--tasks <selector>` (required — explicit WBS list, status pseudo-list, `feature:<id>`, or `ready`). `--mode <sequential|parallel>` (default `sequential`; `parallel` fans out a proven-independent subset per `execution-batch.md` § Parallel Execution). `--keep-going` skips a failed task's in-batch dependents and continues independents (default halts on first failure). `--auto` sets `profile=auto` on each per-task run (skips the HITL approve gate). `--feature <id>` is sugar for `--tasks feature:<id>`; when the effective selector is feature-derived, the batch runs `spur feature check <id> --strict --json` **once** before task resolution — a non-zero strict check aborts with verdict `aborted`, zero attempted tasks, and the structured findings (task 0510 R2); scoped: `L4.scenario-unverified` (expected pre-run state of any not-yet-run feature) is reported verbatim but does not abort — any other strict error aborts. Explicit WBS/status/`ready` selectors add no feature check. The orchestrator loop itself continues in this session; interactive sequential omit/`inline` uses the host driver — host-controlled with eligible `agent.run` stages dispatching once to a native subagent and host fallback (task 0508) — while `--agent auto`/a name, parallel mode, and headless invocation keep the isolated per-task workflow subprocess boundary. `--agent <inline|auto|name>` pins the executor for the per-task stages (see [SSOT](cross-cutting.md#inline-default-execution-surface)); the value crosses into per-task `vars.agent`, not the orchestrator. `--json` emits the report as JSON. `--wrap` triggers `wrapup-pipeline.yaml`
|
|
356
|
-
- **Three orthogonal axes (do not confuse):** `--keep-going` = batch failure policy (halt vs skip dependents); `--continue` = resume from checkpoint (
|
|
355
|
+
- **Inputs:** `--tasks <selector>` (required — explicit WBS list, status pseudo-list, `feature:<id>`, or `ready`). `--mode <sequential|parallel>` (default `sequential`; `parallel` fans out a proven-independent subset per `execution-batch.md` § Parallel Execution). `--keep-going` skips a failed task's in-batch dependents and continues independents (default halts on first failure). `--auto` sets `profile=auto` on each per-task run (skips the HITL approve gate). `--feature <id>` is sugar for `--tasks feature:<id>`; when the effective selector is feature-derived, the batch runs `spur feature check <id> --strict --json` **once** before task resolution — a non-zero strict check aborts with verdict `aborted`, zero attempted tasks, and the structured findings (task 0510 R2); scoped: `L4.scenario-unverified` (expected pre-run state of any not-yet-run feature) is reported verbatim but does not abort — any other strict error aborts. Explicit WBS/status/`ready` selectors add no feature check. The orchestrator loop itself continues in this session; interactive sequential omit/`inline` uses the host driver — host-controlled with eligible `agent.run` stages dispatching once to a native subagent and host fallback (task 0508) — while `--agent auto`/a name, parallel mode, and headless invocation keep the isolated per-task workflow subprocess boundary. `--agent <inline|auto|name>` pins the executor for the per-task stages (see [SSOT](cross-cutting.md#inline-default-execution-surface)); the value crosses into per-task `vars.agent`, not the orchestrator. `--json` emits the report as JSON. `--wrap` triggers `wrapup-pipeline.yaml` **once for the batch** over the `done` subset only (execution-batch.md Step 6: `vars.feature` only when every frozen task is `done`/`cancelled`; empty done subset skips with a reason), `--next` chains each task to terminal status then runs the same batch-once wrap, `--continue` resumes from checkpoint.
|
|
356
|
+
- **Three orthogonal axes (do not confuse):** `--keep-going` = batch failure policy (halt vs skip dependents); `--continue` = resume from checkpoint against the batch's original frozen identity (execution-batch.md § Batch continuation, task 0919); `--next` = per-task lifecycle chaining (advance status on a verdict — `dev-verify`/`dev-verifyall` only). `routing-table.md` offers `--continue` and `--next` as competing options for the same situation only when the batch was interrupted mid-run; otherwise they address different problems.
|
|
357
357
|
- **Backing:** `sp:spur-dev` skill, `runall` operation → delegates the driver loop to the **`sp:super-planner`** agent (the batch orchestrator).
|
|
358
358
|
- **Behavior:** The orchestrator reads [execution-batch.md](execution-batch.md) and drives: resolve selector → freeze set → topo-sort by `dependencies[]` (Kahn, WBS-ascending tie-break; cycle aborts) → resolve out-of-set deps by status (done → allow, else → block subtree) → run each task via `spur workflow run task-pipeline.yaml --async` + `spur workflow trace` polling → inspect terminal state + `.spur/run/<wbs>-verdict.json` → stop-the-batch default or `--keep-going` subtree skip → emit batch report. Per-task pipeline is invoked **verbatim** — no new FSM, no step edits. `--auto`/`--agent` are the only flags that cross the orchestrator→pipeline boundary (both into per-task `--vars`).
|
|
359
359
|
- **Delegation:** `Skill(skill="sp:spur-dev", args="runall $ARGUMENTS")` → `sp:super-planner` agent.
|
|
@@ -212,8 +212,15 @@ for wbs in plan: # default sequential mode
|
|
|
212
212
|
inline-pipeline-driver(task-pipeline.yaml, wbs)
|
|
213
213
|
else:
|
|
214
214
|
spur workflow run task-pipeline.yaml --vars <vars> --async --json
|
|
215
|
-
follow trace until terminal
|
|
215
|
+
follow trace until terminal # spur workflow trace "$RUN" --follow --timeout 600000
|
|
216
|
+
# (timeout → one checkpoint, exit 1; run continues — never cancel/relaunch)
|
|
217
|
+
if run paused (escalate HITL ask (0933) or standard-profile approve):
|
|
218
|
+
surface the paused prompt (escalate: .spur/run/<wbs>-question.md via workflow.hitl.ask)
|
|
219
|
+
operator answers: spur workflow continue --answer-text <answer> "$RUN" (escalate) | continue "$RUN" (approve)
|
|
220
|
+
resume the SAME run (never relaunch); maxEscalations (default 2) bounds the ask loop
|
|
216
221
|
inspect terminal state + .spur/run/<wbs>-verdict.json
|
|
222
|
+
# accept only if trace .run.runId == $RUN AND verdict mtime ≥ .run.startedAt
|
|
223
|
+
# (else outcome = stale-evidence, non-done; failure policy applies)
|
|
217
224
|
report += outcome(wbs)
|
|
218
225
|
if terminal == failed OR stuck status:
|
|
219
226
|
recovery = recoveryHint(status, wbs) # Step 3.3b — at most once
|
|
@@ -225,11 +232,12 @@ for wbs in plan: # default sequential mode
|
|
|
225
232
|
emit batch report (per-task outcome + preflight skips + recovery hints + batch verdict)
|
|
226
233
|
```
|
|
227
234
|
|
|
228
|
-
Parallel mode keeps the same lifecycle but swaps the inner loop for the
|
|
229
|
-
|
|
235
|
+
Parallel mode keeps the same lifecycle but swaps the inner loop for the per-task-worktree fan-out
|
|
236
|
+
specified in [§ Parallel isolation](#parallel-isolation---mode-parallel): identify a zero-edge,
|
|
230
237
|
non-overlapping subset; **preflight each** selected task; run each ready task's `task-pipeline.yaml`
|
|
231
|
-
invocation in its own
|
|
232
|
-
**sequential** (one WBS). If any decision-framework check fails, serialize and
|
|
238
|
+
invocation in its own create-mode worktree; integrate by rebase + `--ff-only` as tasks finish;
|
|
239
|
+
recovery stays **sequential** (one WBS). If any decision-framework check fails, serialize and
|
|
240
|
+
record the reason.
|
|
233
241
|
|
|
234
242
|
### 3.1 Per-task execution reuses the pipeline verbatim (R4)
|
|
235
243
|
|
|
@@ -239,15 +247,21 @@ Each task runs through the **standard single-task pipeline** — `task-pipeline.
|
|
|
239
247
|
execution invokes the workflow engine. The batch loop inspects the result and never redefines a
|
|
240
248
|
step.
|
|
241
249
|
|
|
242
|
-
**Explicit/parallel path: launch async
|
|
243
|
-
`agent.run` stages runs for many minutes. Always use `--async` + `spur workflow trace
|
|
250
|
+
**Explicit/parallel path: launch async, then one bounded follow** (per execution-workflow.md §"Step 2"): a pipeline with
|
|
251
|
+
`agent.run` stages runs for many minutes. Always use `--async` + `spur workflow trace`. The watch is a
|
|
252
|
+
single bounded call — `--timeout 600000` (10 minutes) is the one bound; a hand-rolled poll loop (and the
|
|
253
|
+
retired "10 minutes or 20 polls" rule) is not needed:
|
|
244
254
|
|
|
245
255
|
```bash
|
|
246
256
|
RUN=$(spur workflow run task-pipeline.yaml \
|
|
247
257
|
--vars '{"wbs":"<wbs>","profile":"auto","agent":"claude"}' --async --json | jq -r '.runId')
|
|
248
|
-
spur workflow trace "$RUN" --
|
|
258
|
+
spur workflow trace "$RUN" --follow --timeout 600000 # single bounded watch to terminal; checkpoint + exit 1 on timeout
|
|
259
|
+
spur workflow trace "$RUN" --json | jq '.run | {runId, status, startedAt}' # inspect: identity, status, freshness anchor
|
|
249
260
|
```
|
|
250
261
|
|
|
262
|
+
A timed-out follow prints one checkpoint naming the run id and last status and exits 1; the run itself
|
|
263
|
+
continues — never cancel or relaunch it over a watch timeout. Resume the watch with the same command.
|
|
264
|
+
|
|
251
265
|
### 3.2 Flag → `--vars` passthrough (R4.2, R4.3)
|
|
252
266
|
|
|
253
267
|
Only two flags cross the orchestrator→pipeline boundary; both are merged into the per-task
|
|
@@ -273,6 +287,26 @@ Each pipeline run ends in one of two terminal states:
|
|
|
273
287
|
`onEnter` exception). Record `failed` with the blocking reason from the trace. This triggers the
|
|
274
288
|
failure policy.
|
|
275
289
|
|
|
290
|
+
**Verify-answer table contract (0948 R9).** A verify stage's
|
|
291
|
+
`.spur/run/<wbs>-verify-answer.txt` AC table is exactly four columns:
|
|
292
|
+
`| AC | Status | Evidence Type | Evidence |`. The evidence-type token
|
|
293
|
+
(`test`, `command`, `static-ref`, `manual-review`, `llm-judge`, `n/a`, or a `+`
|
|
294
|
+
compound) is isolated in cell 3. A token merged into the evidence cell fails
|
|
295
|
+
`verify-answer-lint`.
|
|
296
|
+
|
|
297
|
+
**Driver acceptance (0930 R3).** The trace row and `.spur/run/<wbs>-verdict.json` are accepted as
|
|
298
|
+
terminal evidence only if BOTH hold:
|
|
299
|
+
|
|
300
|
+
1. **Identity** — the accepted run is the dispatched run: the **trace's** `.run.runId` equals
|
|
301
|
+
`$RUN` (compare against `spur workflow trace "$RUN" --json | jq -r '.run.runId'`). The verdict
|
|
302
|
+
artifact itself carries no runId — it is WBS-keyed, so it binds to this run only through the
|
|
303
|
+
trace identity plus the freshness check below.
|
|
304
|
+
2. **Freshness** — the verdict file's mtime is at least the trace's `.run.startedAt` (a verdict
|
|
305
|
+
written before this run started is a stale child result, not completion evidence).
|
|
306
|
+
|
|
307
|
+
Otherwise record outcome **`stale-evidence`** (non-`done`); the failure policy applies. A stale or
|
|
308
|
+
mismatched artifact never authorizes cancelling or relaunching the run.
|
|
309
|
+
|
|
276
310
|
### 3.3b One-shot recovery (task 0279 — next-router consumer)
|
|
277
311
|
|
|
278
312
|
After a non-PASS terminal state (or when the task status is stuck at `wip`/`testing` without a clean
|
|
@@ -349,16 +383,16 @@ subagents are meant to hold (task 0508's dispatch contract is preserved unchange
|
|
|
349
383
|
dep-status lookups (Step 1), and any other controller-side `task show`. A status-only lookup may
|
|
350
384
|
narrow further (`| jq '.status'`), but never widen.
|
|
351
385
|
|
|
352
|
-
- Green-path trace observation projects to
|
|
386
|
+
- Green-path trace observation projects to `.run | {runId, status, startedAt}` only:
|
|
353
387
|
|
|
354
388
|
```bash
|
|
355
|
-
spur workflow trace "$RUN" --json | jq '{runId, status,
|
|
389
|
+
spur workflow trace "$RUN" --json | jq '.run | {runId, status, startedAt}'
|
|
356
390
|
```
|
|
357
391
|
|
|
358
|
-
The controller decides continue/halt from
|
|
359
|
-
`status === 'done'`, never by string-matching a `finalState` name) plus the
|
|
360
|
-
artifact `.spur/run/<wbs>-verdict.json
|
|
361
|
-
summarize status.
|
|
392
|
+
The controller decides continue/halt from `.run.status` (ADR-044: judge a run by
|
|
393
|
+
`status === 'done'`, never by string-matching a `finalState` name) plus the accepted verdict
|
|
394
|
+
artifact `.spur/run/<wbs>-verdict.json` under the §3.3 acceptance rule (identity + freshness).
|
|
395
|
+
It never streams or re-reads a full trace merely to summarize status.
|
|
362
396
|
|
|
363
397
|
**Failure-path reads are bounded.** On a failed/blocked task, request only the terminal error and
|
|
364
398
|
the minimal anchor set the batch report needs (e.g. the blocking finding line, the unmet-dep WBS,
|
|
@@ -386,9 +420,12 @@ derivable from the same dependency graph built in Step 2.
|
|
|
386
420
|
|
|
387
421
|
## Step 5 — Batch report (R5.2)
|
|
388
422
|
|
|
389
|
-
When the batch finishes — clean (all `done`), halted (default failure policy),
|
|
390
|
-
|
|
391
|
-
|
|
423
|
+
When the batch finishes — clean (all `done`), halted (default failure policy), parked (`paused`:
|
|
424
|
+
one or more tasks sit on an operator question (escalate, 0933) or an approval gate), or aborted
|
|
425
|
+
(cycle / unknown selector) — emit a structured report. The report is the orchestrator's sole
|
|
426
|
+
output; it does not mutate the corpus (the pipeline's `record` step already wrote per-task
|
|
427
|
+
results). A `paused` task is non-terminal — the report marks it `paused` (never `done`) and names
|
|
428
|
+
the resume command.
|
|
392
429
|
|
|
393
430
|
```
|
|
394
431
|
## Batch Report — <selector>
|
|
@@ -407,9 +444,25 @@ not mutate the corpus (the pipeline's `record` step already wrote per-task resul
|
|
|
407
444
|
| 0060 | blocked | unmet out-of-set dep: 0099 is wip |
|
|
408
445
|
|
|
409
446
|
**Next:** <one-line action — pick up halted run / resolve 0099 / all green, feature H1 complete>
|
|
447
|
+
|
|
448
|
+
**Excluded from wrap** (only under `--wrap`/`--next`, when some batch tasks are not `done`):
|
|
449
|
+
|
|
450
|
+
```
|
|
451
|
+
| WBS | Status | Recovery |
|
|
452
|
+
|-----|--------|--------|
|
|
453
|
+
| 0042 | failed | C6 recovery: /sp:dev-run 0042 (residual report: .spur/run/0042-residual-report.md) |
|
|
454
|
+
| 0050 | not-attempted | /sp:dev-next 0050 |
|
|
410
455
|
```
|
|
411
456
|
|
|
412
|
-
|
|
457
|
+
A task with a residual report (failing `residual-sweep` check in its verdict artifact) uses the
|
|
458
|
+
C6 recovery line (re-run the pipeline); every other excluded task uses the next-router A-row
|
|
459
|
+
command for its status (F96 task 0952 R2).
|
|
460
|
+
|
|
461
|
+
The per-task outcome vocabulary: `done` | `failed` | `blocked` | `skipped` | `not-attempted`,
|
|
462
|
+
plus the resume-only `recheck` (stale/mismatched evidence — pipeline re-run), `not-admitted`
|
|
463
|
+
(post-freeze selector match, never executed — task 0919 BC-1/BC-2), and the parallel-only
|
|
464
|
+
`integration-conflict` (rebase conflict retained for manual integration — task 0931 R4,
|
|
465
|
+
[§ Parallel isolation](#parallel-isolation---mode-parallel)).
|
|
413
466
|
The batch verdict: `clean` (all attempted tasks `done`) | `halted` (a failure stopped the batch) |
|
|
414
467
|
`aborted` (cycle or selector error before any run).
|
|
415
468
|
|
|
@@ -421,8 +474,9 @@ written for a batch with nothing to run. The early-exit report carries zero per-
|
|
|
421
474
|
terminal action runs. A contract test pins this
|
|
422
475
|
(`plugins/sp/tests/dogfood-testing/execution-batch-contract.test.ts`).
|
|
423
476
|
|
|
424
|
-
**Evidence persistence (worktree batches — task 0720 R3).**
|
|
425
|
-
|
|
477
|
+
**Evidence persistence (worktree batches — task 0720 R3).** Copy-out is mandatory.
|
|
478
|
+
A worktree batch's Step 5 report, verdict artifacts, and the batch's own
|
|
479
|
+
`.spur/run` records live in the worktree while the batch runs — exactly the tree
|
|
426
480
|
create-mode WT-4 deletes. Before any WT-4 removal, persist them into the **invoking** tree, which
|
|
427
481
|
survives removal:
|
|
428
482
|
|
|
@@ -438,13 +492,44 @@ batch can never destroy its own evidence. Reuse mode retains its operator-owned
|
|
|
438
492
|
persists the Step 5 report under the invoking tree; the reused tree's `.spur/run/` remains the live
|
|
439
493
|
copy while that tree lives on.
|
|
440
494
|
|
|
495
|
+
**Stage records are worktree-local too (0948 R9, E7 Finding 5).** The persisted report and verdict
|
|
496
|
+
JSONs above are the batch's *summary* evidence. Each task's own per-stage run record
|
|
497
|
+
(`.spur/run/<runId>.md` + `.state.json`, plus gate/answer artifacts) is written inside the
|
|
498
|
+
worktree's `.spur/run/` and **is removed with the worktree** in create mode. A merged batch
|
|
499
|
+
therefore leaves no per-stage run record in the invoking tree unless it is copied out. Anything
|
|
500
|
+
auditing "what did this batch actually run?" must copy those records out **before** WT-4 removal —
|
|
501
|
+
the E7 batch lost exactly this evidence this way. The rule is the same one above: copy out first,
|
|
502
|
+
then remove; a copy failure retains the worktree.
|
|
503
|
+
|
|
504
|
+
## Step 6 — Batch wrap (`--wrap` / `--next`) (F96 task 0952 R1)
|
|
505
|
+
|
|
506
|
+
Run **once for the batch**, after the Step 5 report — never per task (dev-operations.md §runall
|
|
507
|
+
agrees; the old "per task" flag-row wording in dev-runall.md was a contradiction, now fixed).
|
|
508
|
+
|
|
509
|
+
The wrap receives only what it would accept:
|
|
510
|
+
|
|
511
|
+
1. `vars.tasks` = the JSON-encoded array of batch WBS whose terminal status is `done`. A task that
|
|
512
|
+
is `failed`/`blocked`/`skipped`/`not-attempted`/`recheck`/`not-admitted` is **excluded** —
|
|
513
|
+
`wrapup task-resolve` refuses non-done members (wrapup-steps.ts task-resolve), so the driver
|
|
514
|
+
simply stops handing the wrap tasks it would refuse.
|
|
515
|
+
2. `vars.feature` = the batch feature **only when every task in the frozen batch is `done` or
|
|
516
|
+
`cancelled`**. Otherwise omit it and print `feature lifecycle not advanced: <n> task(s)
|
|
517
|
+
unfinished` — advancing a feature while some of its batch tasks are unfinished would overstate
|
|
518
|
+
completion. Learnings/metrics capture still happens for the done subset.
|
|
519
|
+
3. When the done subset is **empty**, skip the wrap entirely with the reason (e.g. `batch wrap
|
|
520
|
+
skipped: no done tasks`) instead of invoking wrapup-pipeline on an empty set.
|
|
521
|
+
|
|
522
|
+
Filtering lives here, in the batch driver — no change to wrapup-pipeline.yaml or wrapup-steps.ts;
|
|
523
|
+
the wrap's refusal of non-done tasks remains the hard invariant.
|
|
524
|
+
|
|
441
525
|
## Worktree isolation (`--worktree [<name>]`)
|
|
442
526
|
|
|
443
527
|
When a batch command (`dev-runall`, `dev-refineall`, `dev-verifyall`) is invoked with
|
|
444
528
|
[`--worktree [<name>]`](flag-glossary.md#flag-worktree), the entire driver loop runs inside an
|
|
445
529
|
isolated git worktree instead of the operator's working directory. This section owns the worktree
|
|
446
|
-
lifecycle for the sequential batch loop.
|
|
447
|
-
|
|
530
|
+
lifecycle for the sequential batch loop. `--mode parallel` has its own per-task isolation — see
|
|
531
|
+
[§ Parallel isolation](#parallel-isolation---mode-parallel); `--worktree --mode parallel` is
|
|
532
|
+
rejected there (parallel mode already isolates each task in its own worktree).
|
|
448
533
|
|
|
449
534
|
**Startup ordering (task 0814 R3).** Resolve the selector/status filter and run the quick
|
|
450
535
|
command-aware readiness (the `quickReadiness` contract in `batch-preflight.ts`) **before** creating
|
|
@@ -525,6 +610,14 @@ Create one worktree on a new branch cut from the current HEAD's ref (the **base
|
|
|
525
610
|
`feat/…` branch, not literally `main`). Location follows the sibling-directory convention in
|
|
526
611
|
[worktree-patterns.md](../../branch-workflow/references/worktree-patterns.md):
|
|
527
612
|
|
|
613
|
+
**Location rule (0948 R9) — never under `.spur/`.** The default root is the **sibling** directory
|
|
614
|
+
(`../<repo>-<command>-<selector-slug>-<short-id>`). A worktree nested under `.spur/` breaks Biome's
|
|
615
|
+
vcs-root detection: `bunx biome check .` inside it reports `Checked 0 files`, so lint/format silently
|
|
616
|
+
no-op for the whole run. Verified in both directions — a sibling worktree reports a non-zero file
|
|
617
|
+
count (`Checked 1076 files` at the time of the fix) while a `.spur/`-nested one reports
|
|
618
|
+
`Checked 0 files`. If a project's tooling needs a custom root, keep it **outside** any path that a
|
|
619
|
+
VCS-root-detecting tool treats as ignorable.
|
|
620
|
+
|
|
528
621
|
```bash
|
|
529
622
|
BASE_REF=$(git rev-parse --abbrev-ref HEAD)
|
|
530
623
|
BASE_SHA=$(git rev-parse HEAD)
|
|
@@ -537,6 +630,13 @@ git worktree add "../<repo>-<command>-<selector-slug>-<short-id>" -b "$BRANCH" "
|
|
|
537
630
|
|| { git branch -D "$BRANCH"; false; }
|
|
538
631
|
```
|
|
539
632
|
|
|
633
|
+
**Worktree root is outside `.spur/` (0948 R9).** The default create path is that
|
|
634
|
+
sibling directory, which sits next to the repository and not under it. Do not
|
|
635
|
+
put the default root at `.spur/worktrees` or any other gitignored path: Biome's
|
|
636
|
+
`vcs.useIgnoreFile` then treats the checkout as empty ("Checked 0 files") and
|
|
637
|
+
the quality gate cannot see the tree. Same class as the eval-pipeline worktree
|
|
638
|
+
move off `.spur/tmp/` (task 0610).
|
|
639
|
+
|
|
540
640
|
Branch and directory names are derived (command + selector slug + short id); the create path never
|
|
541
641
|
takes an operator-supplied name (R8.3 — no create-with-name; `--worktree <name>` where `<name>` does
|
|
542
642
|
not resolve is an error, not a create).
|
|
@@ -743,6 +843,8 @@ if [ -n "$FINAL" ]; then
|
|
|
743
843
|
fi # NO prune/remove/branch delete
|
|
744
844
|
git worktree remove "../<worktree-dir>"
|
|
745
845
|
git branch -d "$BRANCH"
|
|
846
|
+
# WT-4c — clean up registry entry in ~/.config/spur/projects.json (task 0924)
|
|
847
|
+
spur projects remove "$WT_PATH" 2>/dev/null || spur projects clean --json 2>/dev/null || true
|
|
746
848
|
# update marker: status = "merged"
|
|
747
849
|
```
|
|
748
850
|
|
|
@@ -828,7 +930,7 @@ Resume, merge, or discard:
|
|
|
828
930
|
|
|
829
931
|
resume: cd <worktree-path> && <command> --continue --worktree <worktree-path>
|
|
830
932
|
merge: git checkout <base-ref> && git merge <branch> # resolve conflicts manually
|
|
831
|
-
discard: git worktree remove <worktree-path> && git branch -D <branch>
|
|
933
|
+
discard: git worktree remove <worktree-path> && git branch -D <branch> && spur projects remove <worktree-path>
|
|
832
934
|
```
|
|
833
935
|
|
|
834
936
|
The report reuses the [`--next` chain contract](flag-glossary.md#--next-chain-contract) halt-report
|
|
@@ -867,8 +969,10 @@ fallback, because `<name>` was explicit and unambiguous intent.
|
|
|
867
969
|
- **`dev-next`** does not get `--worktree` — it dispatches a single step; per-step isolation is not
|
|
868
970
|
worth the worktree cost. `dev-run` is different: it drives a whole task pipeline, so it does get
|
|
869
971
|
the flag.
|
|
870
|
-
- **`--mode parallel`** is rejected when combined with `--worktree` —
|
|
871
|
-
|
|
972
|
+
- **`--mode parallel`** is rejected when combined with `--worktree` — parallel mode already
|
|
973
|
+
isolates each task in its own worktree
|
|
974
|
+
([§ Parallel isolation](#parallel-isolation---mode-parallel)), and reuse mode
|
|
975
|
+
(`--worktree <name>`) has no per-task meaning.
|
|
872
976
|
- **`--mode implement`** is rejected when combined with `--worktree` on `dev-run` — that mode *is*
|
|
873
977
|
the pipeline's implement stage (bug-742) and runs in whatever tree the driver set up; a second
|
|
874
978
|
worktree would split one task's evidence across two trees.
|
|
@@ -882,13 +986,6 @@ While the batch runs, corpus writes (`spur task update`, `spur feature update`)
|
|
|
882
986
|
the merge (WT-4) or manual integration (WT-5) propagates the writes back. Worth one line in each
|
|
883
987
|
command doc so it does not read as a bug.
|
|
884
988
|
|
|
885
|
-
## Still out of scope
|
|
886
|
-
|
|
887
|
-
- **Interactive within-step Q&A** — a headless subprocess `agent.run` agent asking the operator a
|
|
888
|
-
real question. This waits for the workspace module + inbox module + the agent fleet.
|
|
889
|
-
`sp:super-planner` surfaces blockers/HITL only at the **batch boundary** (between task runs), not
|
|
890
|
-
from inside a pipeline step.
|
|
891
|
-
|
|
892
989
|
## Gate preflight (dogfood 2026-08-21, feature A3)
|
|
893
990
|
|
|
894
991
|
The A3 batch burned multiple full `spur-check-new` runs (~2 min each) that failed only at the tail
|
|
@@ -930,21 +1027,107 @@ time. Before launching a full `spur-check-new`:
|
|
|
930
1027
|
| 0510 R2 (feature-derived strict preflight) | Step 1 — "Feature-derived strict preflight (R2, task 0510)" |
|
|
931
1028
|
| 0510 R5 (metadata-only host controller) | Step 3.4 + projected `task show` / trace snippets in Step 1, 2.3, 3.1 |
|
|
932
1029
|
|
|
933
|
-
## Parallel
|
|
1030
|
+
## Parallel isolation (`--mode parallel`)
|
|
934
1031
|
|
|
935
|
-
|
|
1032
|
+
Under `--mode parallel` (task 0931) every concurrently running task gets its **own create-mode git
|
|
1033
|
+
worktree** (`sp/run-<wbs>-<short-id>`, cut from the current base-ref tip). Two task pipelines never
|
|
1034
|
+
share a working tree, and the main tree receives no task writes while the batch runs. The per-task
|
|
1035
|
+
pipeline (`task-pipeline.yaml`) is unchanged — only where it runs and when dependents may start
|
|
1036
|
+
differ. The fan-out decision framework (independence, overlap, budget) stays owned by
|
|
1037
|
+
[sp:parallel-execution](../../parallel-execution/SKILL.md); this section owns the isolation +
|
|
1038
|
+
integration lifecycle and reuses WT-1…WT-5 by reference — it defines no new worktree mechanics.
|
|
936
1039
|
|
|
937
|
-
|
|
1040
|
+
### Driver loop
|
|
938
1041
|
|
|
939
|
-
|
|
940
|
-
|
|
941
|
-
|
|
942
|
-
|
|
943
|
-
|
|
1042
|
+
```
|
|
1043
|
+
WT-1 once on the main tree (dirty → abort) # one precheck for the whole batch, not per task
|
|
1044
|
+
ready = topo frontier; running = {}; integrated = set()
|
|
1045
|
+
while ready or running:
|
|
1046
|
+
while |running| < CONCURRENCY and ready has t with deps(t) ⊆ integrated:
|
|
1047
|
+
WT-2 create sp/run-<wbs>-<short-id> (branch sp/run-<wbs>-<short-id>) from BASE_REF tip
|
|
1048
|
+
WT-3 marker {command: dev-runall, selector: <wbs>, batchId: <batchId>}
|
|
1049
|
+
# WT-3 schema fields (path, branch, baseRef, baseSha) + batchId shared by every marker
|
|
1050
|
+
RUN[t] = (cd "$WT" && spur workflow run task-pipeline.yaml \
|
|
1051
|
+
--vars '{"wbs":"<wbs>","deferFeatureSync":"true",...}' --async --json)
|
|
1052
|
+
wait for any RUN terminal # trace poll --follow --timeout 600000; stale-evidence rule applies
|
|
1053
|
+
on done: WT-3b commit on $BRANCH → integrate(t)
|
|
1054
|
+
on failed: WT-5 retain (marker retained); failure policy — stop-the-batch default / --keep-going subtree skip
|
|
1055
|
+
on timeout/paused: WT-5 retain; run is non-terminal (HITL approve pause or poll bound) — stop-the-batch
|
|
1056
|
+
default / --keep-going subtree skip; the report marks the task `non-terminal`, never `done`
|
|
1057
|
+
post: feature sync + refresh per touched feature, one chore(corpus) commit (generated regions, R5)
|
|
1058
|
+
emit batch report (per-task outcomes + preflight skips + recovery hints + batch verdict)
|
|
1059
|
+
```
|
|
944
1060
|
|
|
945
|
-
**
|
|
1061
|
+
- **Bound (R2).** `CONCURRENCY` is the [`--concurrency <n>`](flag-glossary.md#flag-concurrency)
|
|
1062
|
+
value: default **2**, `n ≥ 1`; at most that many pipelines run at once.
|
|
1063
|
+
- **Eligibility (R2).** A task becomes eligible only when all of its in-set dependencies are
|
|
1064
|
+
**integrated** onto the base ref — pipeline-terminal is not enough, because its branch must
|
|
1065
|
+
rebase over theirs. Omitting `--mode` stays sequential; this loop is entered only via
|
|
1066
|
+
`--mode parallel` on `/sp:dev-runall`.
|
|
1067
|
+
- **Workdir.** `spur workflow run` records the launch workdir (0784 R1), so launch with `cwd` = the
|
|
1068
|
+
task worktree: resume and `.spur/run/` artifact paths resolve there. Read the verdict from
|
|
1069
|
+
`$WT/.spur/run/<wbs>-verdict.json` before integration; WT-4a persists it into the invoking tree.
|
|
946
1070
|
|
|
947
|
-
|
|
1071
|
+
### Integration — rebase, then fast-forward only (R3)
|
|
1072
|
+
|
|
1073
|
+
Pipeline-terminal is not terminal under parallel mode. The orchestrator integrates each succeeded
|
|
1074
|
+
task from the main tree; integrations are **serialized** (one at a time) in completion order:
|
|
1075
|
+
|
|
1076
|
+
```
|
|
1077
|
+
integrate(t):
|
|
1078
|
+
git -C "$WT" rebase "$BASE_REF" # replay onto the CURRENT base tip (re-read per integration)
|
|
1079
|
+
├─ rebase fails → git -C "$WT" rebase --abort → conflict path (R4) below
|
|
1080
|
+
└─ rebase clean → git checkout "$BASE_REF" → zero-commit guard → git merge --ff-only "$BRANCH"
|
|
1081
|
+
→ WT-4a verdict persist → WT-4b holders → WT-4c registry
|
|
1082
|
+
→ git worktree remove → git branch -d → marker merged
|
|
1083
|
+
```
|
|
1084
|
+
|
|
1085
|
+
The merge is always `--ff-only`: a base that will not fast-forward after a clean rebase is a
|
|
1086
|
+
conflict (R4). The pipeline never creates a merge commit, and no conflict is ever resolved
|
|
1087
|
+
automatically. `BASE_REF` is re-read at each integration so a task that finished late rebases over
|
|
1088
|
+
everything integrated before it.
|
|
1089
|
+
|
|
1090
|
+
### Conflict — retain, report, block (R4)
|
|
1091
|
+
|
|
1092
|
+
A failed rebase is aborted (`git -C "$WT" rebase --abort`), which leaves the worktree clean on its
|
|
1093
|
+
original branch tip. The worktree and branch are **retained** (WT-5, marker `retained`) and the
|
|
1094
|
+
task's outcome is `integration-conflict`. There is no auto-resolution — not for generated paths,
|
|
1095
|
+
not for anything. The batch report's row names the worktree path, the branch, and the manual
|
|
1096
|
+
commands:
|
|
1097
|
+
|
|
1098
|
+
```
|
|
1099
|
+
resume: cd <worktree-path> && git rebase <BASE_REF> # the operator resolves, never the driver
|
|
1100
|
+
merge: git checkout <BASE_REF> && git merge --ff-only <branch>
|
|
1101
|
+
discard: git worktree remove <worktree-path> && git branch -D <branch> && spur projects remove <worktree-path>
|
|
1102
|
+
```
|
|
1103
|
+
|
|
1104
|
+
The task's dependent subtree is blocked under the normal failure policy (Step 4). Resuming a
|
|
1105
|
+
retained parallel batch with `--continue` is future work; today a retained task is resumed
|
|
1106
|
+
per-task with `/sp:dev-run <wbs> --worktree <branch>`.
|
|
1107
|
+
|
|
1108
|
+
### Generated regions — defer the sync, regenerate once (R5)
|
|
1109
|
+
|
|
1110
|
+
The only per-task writer of feature files is the `record` step's post-record feature sync
|
|
1111
|
+
(`task-pipeline.yaml`, the `feature-sync-bounded` wrapper). Parallel launches set
|
|
1112
|
+
the pipeline var `deferFeatureSync: "true"` (default `"false"`): the record step appends
|
|
1113
|
+
`feature sync deferred to batch integration` to the task report and skips the sync, so task
|
|
1114
|
+
branches never touch feature files or `docs/features/INDEX.md`. After the last integration, on the
|
|
1115
|
+
base ref, the orchestrator runs the same bounded wrapper plus `spur feature refresh --feature <f>`
|
|
1116
|
+
once per touched feature and commits the result as one `chore(corpus)` commit. Sequential and
|
|
1117
|
+
inline runs keep the default `"false"` and are unchanged. Any rebase conflict — on a generated
|
|
1118
|
+
path or any other — is an R4 `integration-conflict`; there is no path-based exception.
|
|
1119
|
+
|
|
1120
|
+
**Report.** Step 5's per-task outcome vocabulary gains the parallel-only `integration-conflict`;
|
|
1121
|
+
its row carries `worktree`, `branch`, and the manual commands above. Under parallel mode `done`
|
|
1122
|
+
means **integrated onto the base ref**, not merely pipeline-terminal.
|
|
1123
|
+
|
|
1124
|
+
**`--worktree` is rejected under parallel mode.** `--worktree --mode parallel` fails with
|
|
1125
|
+
"parallel mode already isolates each task in its own worktree" — reuse mode (`--worktree <name>`)
|
|
1126
|
+
has no per-task meaning (WT-7). Run parallel batches without `--worktree`, or run them sequentially
|
|
1127
|
+
with it.
|
|
1128
|
+
|
|
1129
|
+
**See also:** `sp:parallel-execution` skill (fan-out decision framework), `sp:super-planner` agent
|
|
1130
|
+
(parallel mode), `execution-batch.md` § Worktree isolation (WT-1…WT-7 mechanics).
|
|
948
1131
|
|
|
949
1132
|
## Subagent execution disciplines
|
|
950
1133
|
|
|
@@ -956,18 +1139,63 @@ Parallel fan-out and any subagent dispatch obey the four disciplines owned by
|
|
|
956
1139
|
- **Per-role model selection** — the cheapest model that fits each role (`--agent` pins the executor; the discipline picks the model per role).
|
|
957
1140
|
- **Never pre-judge the reviewer** — verify/review subagents receive artifact + contract only; no pre-rated severity, no "do not flag X".
|
|
958
1141
|
|
|
959
|
-
##
|
|
960
|
-
|
|
961
|
-
|
|
962
|
-
|
|
963
|
-
|
|
964
|
-
|
|
965
|
-
|
|
966
|
-
|
|
967
|
-
|
|
968
|
-
|
|
969
|
-
|
|
970
|
-
|
|
971
|
-
|
|
972
|
-
|
|
973
|
-
|
|
1142
|
+
## Batch continuation (`--continue`) — identity binding + checkpoint reconciliation (task 0919)
|
|
1143
|
+
|
|
1144
|
+
`--continue` resumes an interrupted batch against the batch's **original identity** — never
|
|
1145
|
+
whatever a fresh selector resolution or a newer unrelated checkpoint would suggest.
|
|
1146
|
+
|
|
1147
|
+
### BC-1 — Re-bind the original frozen identity (R1; AC1/AC5)
|
|
1148
|
+
|
|
1149
|
+
Resume identity comes from persisted batch artifacts, in priority order:
|
|
1150
|
+
|
|
1151
|
+
1. **WT-3 marker** (worktree batches): the marker's `command` + `selector` and worktree
|
|
1152
|
+
name/branch re-derive the original launch (the WT-6 fallback already resolves this way).
|
|
1153
|
+
2. **Persisted batch report** (`.spur/run/worktree-<marker-id>-batch-report.md`, or the
|
|
1154
|
+
invoking-tree Step 5 report of a non-worktree batch): its `Plan:` row IS the frozen ordered
|
|
1155
|
+
membership — the resumed loop iterates exactly those WBS rows.
|
|
1156
|
+
|
|
1157
|
+
Re-running the selector is a **validation, not a re-definition**: tasks that newly match the
|
|
1158
|
+
selector but are absent from the frozen plan are reported `not-admitted` and never executed —
|
|
1159
|
+
freshly listed tasks do not join a resumed batch implicitly. A frozen task whose dependencies
|
|
1160
|
+
changed after freeze may invalidate its admission: report `blocked (admission invalidated —
|
|
1161
|
+
re-plan required)` and stop for operator decision. The authorized frozen set is never silently
|
|
1162
|
+
rewritten (Design).
|
|
1163
|
+
|
|
1164
|
+
### BC-2 — Checkpoints are hints (R2; AC2)
|
|
1165
|
+
|
|
1166
|
+
Do **not** pick the resume point by mtime alone: `ls -t … | head -1` over `.spur/memory/sessions/`
|
|
1167
|
+
is identity-blind, so an unrelated later session (different feature or task) must not become
|
|
1168
|
+
authoritative. Select candidate checkpoints by identity first, newest first within the match:
|
|
1169
|
+
|
|
1170
|
+
- frontmatter `workflow` is `task-pipeline` (or the batch's own workflow), and
|
|
1171
|
+
- `feature_id` equals the batch's feature selector, or `task_wbs` is a member of the frozen plan.
|
|
1172
|
+
|
|
1173
|
+
Checkpoints matching neither rule are ignored regardless of recency. A matched checkpoint only
|
|
1174
|
+
**hints** where the loop left off (`phase`, `next_action` — surface them to the operator); the
|
|
1175
|
+
driver then reconciles against authoritative state before skipping or repeating any task:
|
|
1176
|
+
|
|
1177
|
+
- **Skip** requires ALL of: task file status `done` AND a persisted PASS verdict artifact
|
|
1178
|
+
(`.spur/run/<wbs>-verdict.json`, or the worktree-persisted
|
|
1179
|
+
`.spur/run/worktree-<marker-id>-verdicts/<wbs>-verdict.json`) consistent with the task's
|
|
1180
|
+
current metadata. Anything less is not a valid skip.
|
|
1181
|
+
- **Stale or mismatched evidence** — checkpoint claims done but the verdict artifact is missing,
|
|
1182
|
+
the task file moved backwards, or the recorded `source_commit`/`digest` no longer matches —
|
|
1183
|
+
yields outcome `recheck`: re-run that task's pipeline. When reconciliation cannot proceed
|
|
1184
|
+
safely (lost worktree, unresolvable marker), report `blocked` with the reason. Never treat an
|
|
1185
|
+
unverified claim as done.
|
|
1186
|
+
|
|
1187
|
+
### BC-3 — Ordering, write ownership, terminal mix, evidence (R3/R4; AC3/AC4)
|
|
1188
|
+
|
|
1189
|
+
Sequential dependency-correct execution remains the default; opt-in `--mode parallel` keeps the
|
|
1190
|
+
existing proven-independence requirements with one writer per tree (Step 3). Mixed terminal
|
|
1191
|
+
results resume under the unchanged failure policy (Step 4). The Step 5 report keeps per-task
|
|
1192
|
+
outcomes distinct — `done | failed | blocked | skipped | not-attempted` plus the resume-only
|
|
1193
|
+
`recheck` and `not-admitted` — and the batch verdict stays `halted`/`aborted`: a resumed,
|
|
1194
|
+
partially-complete batch is never reported `clean`. Worktree evidence survives cleanup via the
|
|
1195
|
+
invoking-tree persistence in Step 5 (task 0720 R3); a persistence failure routes to **WT-5**
|
|
1196
|
+
(retain tree + branch), so partial outcomes are never silently lost and never read as a completed
|
|
1197
|
+
batch.
|
|
1198
|
+
|
|
1199
|
+
See [cross-cutting.md](cross-cutting.md) § "Session Checkpoint Convention" for the canonical
|
|
1200
|
+
frontmatter fields and the per-task owner-mismatch semantics
|
|
1201
|
+
(`packages/app/src/workflow/checkpoint-contract.ts`).
|
|
@@ -138,7 +138,7 @@ spur workflow trace "$RUN" --follow --output # streams the run; --output shows
|
|
|
138
138
|
burned ~110 min in 47 sleeps and 55 trace polls waiting on one run; ADR-047 mandates pipe-free
|
|
139
139
|
observation). `--follow` is a blocking human-streaming mode (no `--json`); run it in the session
|
|
140
140
|
background and let its exit report the terminal verdict. Use `spur workflow trace "$RUN" --follow
|
|
141
|
-
--output` to stream the non-interactive agent output to `.spur/run/<runId>.
|
|
141
|
+
--output` to stream the non-interactive agent output to `.spur/run/<runId>.md` as it lands.
|
|
142
142
|
|
|
143
143
|
Synchronous invocation (`--json` without `--async`) is acceptable **only** for short pipelines
|
|
144
144
|
(< 2 min, e.g. precheck-only or a dry-run). Do not use it for the full task pipeline.
|
|
@@ -126,6 +126,17 @@ Batch operation only (`dev-refineall`, `dev-runall`). When a task in the batch
|
|
|
126
126
|
fails, skip its in-batch dependents and continue the independent ones, instead of the default
|
|
127
127
|
halt-on-first-failure. Never silently retried; the failure is still reported.
|
|
128
128
|
|
|
129
|
+
### `--concurrency <n>` — parallel batch worker bound
|
|
130
|
+
|
|
131
|
+
**Anchor:** `#flag-concurrency`.
|
|
132
|
+
|
|
133
|
+
`dev-runall` under `--mode parallel` (task 0931): at most `<n>` task pipelines run at once
|
|
134
|
+
(default 2, `n ≥ 1`). A dependent task starts only after every in-set dependency has been
|
|
135
|
+
**integrated** onto the base ref — see
|
|
136
|
+
[execution-batch.md § Parallel isolation](execution-batch.md#parallel-isolation---mode-parallel).
|
|
137
|
+
Ignored in sequential mode. Each worker is one more concurrent quality gate; raise it only when
|
|
138
|
+
the host has the CPU for it.
|
|
139
|
+
|
|
129
140
|
### `--continue` — resume an interrupted batch from checkpoint
|
|
130
141
|
|
|
131
142
|
**Anchor:** `#flag-continue`.
|
|
@@ -248,7 +259,8 @@ the effect are per-command — this flag is a family, not one behavior:
|
|
|
248
259
|
offered target (rank-1 or the explicit id) and forwards `--auto` to the children. Optional feature
|
|
249
260
|
id names the target instead of offering rank 1.
|
|
250
261
|
- `dev-debug` `[<wbs>]` — **attaches** findings to an existing task. Optional WBS names it.
|
|
251
|
-
- `dev-dogfood` (no value) — **
|
|
262
|
+
- `dev-dogfood` (no value) — **creates** a new review-template task for the run's findings
|
|
263
|
+
(`spur task create --template review`); it does not attach to or update the task under test.
|
|
252
264
|
|
|
253
265
|
### `--since <ref>` — lower bound on a range
|
|
254
266
|
|
|
@@ -333,7 +345,9 @@ pauses, even under `--auto`.
|
|
|
333
345
|
**Anchor:** `#flag-max-retry`.
|
|
334
346
|
|
|
335
347
|
Bound the retry loop on fix-family commands (`dev-dogfood`, `dev-fixall`, `dev-gtd`). After `n` consecutive
|
|
336
|
-
failed fix attempts, stop and ask the operator rather than looping indefinitely.
|
|
348
|
+
failed fix attempts, stop and ask the operator rather than looping indefinitely. On `dev-dogfood`
|
|
349
|
+
the default is `2` (fix mode) and `--max-retry 0` selects observe-only — matching the backing
|
|
350
|
+
`sp:dogfood-testing` skill; the command table and the skill must not drift on this default.
|
|
337
351
|
|
|
338
352
|
### `--full` — rewrite a `--next` run as full pipeline
|
|
339
353
|
|
|
@@ -428,8 +442,9 @@ merges but never removes. This keeps the continue-the-work loop stable — after
|
|
|
428
442
|
`-`**, so `--worktree --auto` is the bare create form and `--agent`/`--feature`/etc. are never
|
|
429
443
|
swallowed as the name. `--worktree=<name>` is the unambiguous spelling. `/sp:dev-next` does not get
|
|
430
444
|
the flag (single _step_; not worth the worktree cost — unlike `dev-run`, which isolates a whole
|
|
431
|
-
task pipeline), `--worktree --mode parallel` is rejected (
|
|
432
|
-
|
|
445
|
+
task pipeline), `--worktree --mode parallel` is rejected (parallel mode already isolates each task
|
|
446
|
+
in its own worktree — [execution-batch.md § Parallel isolation](execution-batch.md#parallel-isolation---mode-parallel)),
|
|
447
|
+
and `--worktree --mode implement` is rejected on `dev-run` (that mode _is_ the pipeline's
|
|
433
448
|
implement stage and runs in the driver's tree). The full lifecycle — name resolution, dirty-tree
|
|
434
449
|
precheck, creation or adoption, crash-safe marker, merge-or-retain, and `--continue` re-entry — is
|
|
435
450
|
specified in [execution-batch.md § Worktree isolation](execution-batch.md#worktree-isolation---worktree-name).
|