task-pipeline-skill 1.33.0 → 1.37.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +207 -0
- package/CONTRIBUTING.md +66 -0
- package/README.md +23 -0
- package/SKILL-CARD.md +1 -1
- package/cursor/rules/task-pipeline.mdc +34 -9
- package/package.json +1 -1
- package/plugins/task-pipeline/.claude-plugin/plugin.json +1 -1
- package/plugins/task-pipeline/skills/task-pipeline/SKILL.md +1 -0
- package/plugins/task-pipeline/skills/task-pipeline/pipeline.schema.json +133 -21
- package/plugins/task-pipeline/skills/task-pipeline/references/artifacts.md +12 -5
- package/plugins/task-pipeline/skills/task-pipeline/references/companion-skills.md +20 -3
- package/plugins/task-pipeline/skills/task-pipeline/references/continuity.md +10 -0
- package/plugins/task-pipeline/skills/task-pipeline/references/exposure.md +8 -1
- package/plugins/task-pipeline/skills/task-pipeline/references/loop-guard.md +52 -2
- package/plugins/task-pipeline/skills/task-pipeline/references/portability.md +1 -0
- package/plugins/task-pipeline/skills/task-pipeline/references/progress.md +190 -0
- package/plugins/task-pipeline/skills/task-pipeline/references/retrospective.md +88 -3
- package/plugins/task-pipeline/skills/task-pipeline/references/stages.md +109 -9
- package/plugins/task-pipeline/skills/task-pipeline/templates/README.md +1 -0
- package/plugins/task-pipeline/skills/task-pipeline/templates/run.md +77 -0
|
@@ -4,18 +4,32 @@
|
|
|
4
4
|
"title": "task-pipeline config",
|
|
5
5
|
"description": "Generic contract for a pipeline config. An ordered list of stages; each stage is run by the host project's own skills/agents and guarded by a typed gate. The framework imposes no specific stages, skills, or gate assignments — those are entirely the host project's config. Copy pipeline.example.json and rewrite it to match your project.",
|
|
6
6
|
"type": "object",
|
|
7
|
-
"required": [
|
|
7
|
+
"required": [
|
|
8
|
+
"stages"
|
|
9
|
+
],
|
|
8
10
|
"additionalProperties": true,
|
|
9
11
|
"properties": {
|
|
10
|
-
"version": {
|
|
12
|
+
"version": {
|
|
13
|
+
"type": "integer",
|
|
14
|
+
"minimum": 1
|
|
15
|
+
},
|
|
11
16
|
"stages": {
|
|
12
17
|
"type": "array",
|
|
13
18
|
"minItems": 1,
|
|
14
19
|
"description": "The pipeline stages, in order. Any number, any names — your project's real stages.",
|
|
15
|
-
"items": {
|
|
20
|
+
"items": {
|
|
21
|
+
"$ref": "#/definitions/stage"
|
|
22
|
+
}
|
|
23
|
+
},
|
|
24
|
+
"release": {
|
|
25
|
+
"$ref": "#/definitions/release"
|
|
26
|
+
},
|
|
27
|
+
"run": {
|
|
28
|
+
"$ref": "#/definitions/run"
|
|
16
29
|
},
|
|
17
|
-
"
|
|
18
|
-
|
|
30
|
+
"retro": {
|
|
31
|
+
"$ref": "#/definitions/retro"
|
|
32
|
+
}
|
|
19
33
|
},
|
|
20
34
|
"definitions": {
|
|
21
35
|
"run": {
|
|
@@ -25,12 +39,17 @@
|
|
|
25
39
|
"properties": {
|
|
26
40
|
"loop": {
|
|
27
41
|
"type": "object",
|
|
28
|
-
"required": [
|
|
42
|
+
"required": [
|
|
43
|
+
"mode"
|
|
44
|
+
],
|
|
29
45
|
"additionalProperties": true,
|
|
30
46
|
"description": "Whether the run advances item by item without a discretionary check-in. It NEVER collapses a manual gate or an outward action — a generic flag is not a specific authorization.",
|
|
31
47
|
"properties": {
|
|
32
48
|
"mode": {
|
|
33
|
-
"enum": [
|
|
49
|
+
"enum": [
|
|
50
|
+
"off",
|
|
51
|
+
"interval"
|
|
52
|
+
],
|
|
34
53
|
"description": "off (the default when absent) = the run pauses between items as it always did. interval = the run is armed with the harness's own loop primitive and advances one item per fire, stopping only at a manual gate, an unresolvable block, a genuine ambiguity, or completion."
|
|
35
54
|
},
|
|
36
55
|
"interval": {
|
|
@@ -43,6 +62,18 @@
|
|
|
43
62
|
"description": "How this harness arms it, e.g. '/loop'. Harness-specific and therefore project-recorded rather than assumed: on a harness with no loop primitive, omit it — the mode then degrades to prose discipline plus the build ledger, and the run says so instead of implying it is armed."
|
|
44
63
|
}
|
|
45
64
|
}
|
|
65
|
+
},
|
|
66
|
+
"review": {
|
|
67
|
+
"type": "object",
|
|
68
|
+
"additionalProperties": true,
|
|
69
|
+
"description": "The review loop's ceiling. Absent, the default is 3 rounds per artifact. It is a DECISION POINT rather than a stop: at the cap the run prints new findings versus findings caused by its own fixes, per round, and either the pair ends the loop or the operator continues it out loud. A flat stop would be wrong — this repository's ten-round runs were still finding real defects on round nine. Doctrine: references/loop-guard.md.",
|
|
70
|
+
"properties": {
|
|
71
|
+
"maxRounds": {
|
|
72
|
+
"type": "integer",
|
|
73
|
+
"minimum": 1,
|
|
74
|
+
"description": "Rounds per artifact before the run stops reviewing and measures. Counted from the run ledger's `touch:` pass numbers, never from memory; a round that finds nothing ends the loop by definition and is not counted."
|
|
75
|
+
}
|
|
76
|
+
}
|
|
46
77
|
}
|
|
47
78
|
}
|
|
48
79
|
},
|
|
@@ -50,48 +81,129 @@
|
|
|
50
81
|
"type": "object",
|
|
51
82
|
"additionalProperties": true,
|
|
52
83
|
"description": "Optional release automation, entirely project-defined and INDIVIDUALLY TOGGLEABLE. Omit the whole object, or set enabled:false, to turn release automation off for this project. The framework ships an example workflow (.github/workflows/release.yml) whose job is gated on a repo variable so each project arms it on its own; this block is the declarative counterpart the orchestrator reads.",
|
|
53
|
-
"required": [
|
|
84
|
+
"required": [
|
|
85
|
+
"enabled"
|
|
86
|
+
],
|
|
54
87
|
"properties": {
|
|
55
|
-
"enabled": {
|
|
56
|
-
|
|
88
|
+
"enabled": {
|
|
89
|
+
"type": "boolean",
|
|
90
|
+
"description": "Master on/off toggle. false (or the object omitted) = no release automation for this project."
|
|
91
|
+
},
|
|
92
|
+
"trigger": {
|
|
93
|
+
"enum": [
|
|
94
|
+
"tag",
|
|
95
|
+
"manual",
|
|
96
|
+
"push",
|
|
97
|
+
"none"
|
|
98
|
+
],
|
|
99
|
+
"description": "What kicks off a release (e.g. a pushed vX.Y.Z tag, manual dispatch)."
|
|
100
|
+
},
|
|
57
101
|
"steps": {
|
|
58
102
|
"type": "array",
|
|
59
|
-
"items": {
|
|
103
|
+
"items": {
|
|
104
|
+
"type": "string",
|
|
105
|
+
"minLength": 1
|
|
106
|
+
},
|
|
60
107
|
"description": "Ordered release actions, in prose — project-specific (e.g. create a GitHub release, npm publish). Human-only steps (2FA publish) are named as such."
|
|
61
108
|
},
|
|
62
109
|
"verify": {
|
|
63
110
|
"type": "array",
|
|
64
|
-
"items": {
|
|
111
|
+
"items": {
|
|
112
|
+
"type": "string",
|
|
113
|
+
"minLength": 1
|
|
114
|
+
},
|
|
65
115
|
"description": "Post-release smoke checks that must pass after a release — this is the release's own post-deploy gate (stage 8 applied to shipping the package itself)."
|
|
66
116
|
}
|
|
67
117
|
}
|
|
68
118
|
},
|
|
69
119
|
"stage": {
|
|
70
120
|
"type": "object",
|
|
71
|
-
"required": [
|
|
121
|
+
"required": [
|
|
122
|
+
"state",
|
|
123
|
+
"skills",
|
|
124
|
+
"gate"
|
|
125
|
+
],
|
|
72
126
|
"additionalProperties": true,
|
|
73
127
|
"properties": {
|
|
74
|
-
"id": {
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
128
|
+
"id": {
|
|
129
|
+
"type": "integer",
|
|
130
|
+
"description": "Optional ordinal."
|
|
131
|
+
},
|
|
132
|
+
"state": {
|
|
133
|
+
"type": "string",
|
|
134
|
+
"minLength": 1,
|
|
135
|
+
"description": "Unique stable key for the stage."
|
|
136
|
+
},
|
|
137
|
+
"name": {
|
|
138
|
+
"type": "string",
|
|
139
|
+
"description": "Optional human label."
|
|
140
|
+
},
|
|
141
|
+
"model": {
|
|
142
|
+
"type": "string",
|
|
143
|
+
"description": "Optional model for the stage. Prefer a provider-agnostic token over a vendor id, which goes stale as generations ship and may not exist on the operator's provider at all: 'default' = the model confirmed for this run (recommended: the most capable reasoning model the environment offers), 'inherit' = whatever the operator is currently on. A literal id is allowed but treated as an example, not a contract."
|
|
144
|
+
},
|
|
78
145
|
"skills": {
|
|
79
146
|
"type": "array",
|
|
80
147
|
"minItems": 1,
|
|
81
148
|
"description": "The skill(s)/agent(s) that execute this stage. Any names your environment resolves — this is where you plug in your OWN skills.",
|
|
82
|
-
"items": {
|
|
149
|
+
"items": {
|
|
150
|
+
"type": "string",
|
|
151
|
+
"minLength": 1
|
|
152
|
+
}
|
|
83
153
|
},
|
|
84
154
|
"gate": {
|
|
85
155
|
"type": "object",
|
|
86
|
-
"required": [
|
|
156
|
+
"required": [
|
|
157
|
+
"type",
|
|
158
|
+
"check"
|
|
159
|
+
],
|
|
87
160
|
"additionalProperties": true,
|
|
88
161
|
"description": "Condition that must pass before advancing to the next stage.",
|
|
89
162
|
"properties": {
|
|
90
163
|
"type": {
|
|
91
|
-
"enum": [
|
|
164
|
+
"enum": [
|
|
165
|
+
"auto",
|
|
166
|
+
"manual"
|
|
167
|
+
],
|
|
92
168
|
"description": "auto = the orchestrator verifies `check` itself (pass/fail); manual = wait for the operator's explicit go."
|
|
93
169
|
},
|
|
94
|
-
"check": {
|
|
170
|
+
"check": {
|
|
171
|
+
"type": "string",
|
|
172
|
+
"minLength": 1,
|
|
173
|
+
"description": "The gate condition, in prose."
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
},
|
|
179
|
+
"retro": {
|
|
180
|
+
"type": "object",
|
|
181
|
+
"additionalProperties": true,
|
|
182
|
+
"description": "Optional. Governs what the retrospective does BEYOND writing to the project's own docs/superpowers/retro.md, which always happens. Omit it and nothing leaves the repository: silence arms nothing, exactly as it authorises no deploy.",
|
|
183
|
+
"properties": {
|
|
184
|
+
"publish": {
|
|
185
|
+
"type": "object",
|
|
186
|
+
"required": [
|
|
187
|
+
"repo"
|
|
188
|
+
],
|
|
189
|
+
"additionalProperties": true,
|
|
190
|
+
"description": "Publish skill-level insights as issues on the skill's own repository, so a defect in the pipeline is fixed once instead of rediscovered in every project. OFF unless this object is present — opening an issue in another repository is an outward act, and an outward act taken from a generic flag is one nobody authorized. The body is printed in full before it is sent, and the printed string and the sent string are the same string. Doctrine, including the five redaction rules: references/retrospective.md.",
|
|
191
|
+
"properties": {
|
|
192
|
+
"repo": {
|
|
193
|
+
"type": "string",
|
|
194
|
+
"pattern": "^[^/\\s]+/[^/\\s]+$",
|
|
195
|
+
"description": "owner/name of the SKILL's repository, never the host project's."
|
|
196
|
+
},
|
|
197
|
+
"label": {
|
|
198
|
+
"type": "string",
|
|
199
|
+
"description": "Applied to every issue opened this way, so the operator can read them as a stream rather than find them by accident."
|
|
200
|
+
},
|
|
201
|
+
"redact": {
|
|
202
|
+
"enum": [
|
|
203
|
+
"strict"
|
|
204
|
+
],
|
|
205
|
+
"description": "Only 'strict' exists, and it is not a level among others: the five rules are the contract. A field with one legal value is here so a reader asking 'what redaction applies?' finds the answer in the config rather than assuming none does."
|
|
206
|
+
}
|
|
95
207
|
}
|
|
96
208
|
}
|
|
97
209
|
}
|
|
@@ -59,9 +59,15 @@ at a glance.
|
|
|
59
59
|
> external skill**. A host project may relocate the root via its `CLAUDE.md`; keep
|
|
60
60
|
> the shape, keep the slugs.
|
|
61
61
|
|
|
62
|
-
|
|
63
|
-
stage
|
|
64
|
-
|
|
62
|
+
**Every** run keeps a **git-ignored** run ledger at `.task-pipeline/run.md`, seeded at
|
|
63
|
+
stage 0 from [`../templates/run.md`](../templates/run.md). Three line shapes: a
|
|
64
|
+
`stage:` verdict when a gate returns, an `iter:` line when an iteration closes, and a
|
|
65
|
+
`touch:` line per file per repeating pass. Two readers depend on it —
|
|
66
|
+
[`loop-guard.md`](loop-guard.md) detects churn from the `touch:` lines after a lost
|
|
67
|
+
context, and [`progress.md`](progress.md) derives the stage rail and the iteration
|
|
68
|
+
counter from the other two. It was described as loop-only until 2026-08-10 and, in
|
|
69
|
+
practice, written by no run at all: the guard that calls its own detection *mechanical*
|
|
70
|
+
had no input in any run to date.
|
|
65
71
|
|
|
66
72
|
Stage 5 also creates a **git-ignored** scratch workspace per plan at
|
|
67
73
|
`.task-pipeline/build/<plan-basename>/` — ledger, task briefs, implementer reports,
|
|
@@ -158,12 +164,13 @@ plugins/task-pipeline/
|
|
|
158
164
|
acceptance.md retrospective.md # stage 10 (close-out, then the retro)
|
|
159
165
|
audit.md # cross-cutting: the ladder + seams
|
|
160
166
|
loop-guard.md # cross-cutting: churn detection
|
|
167
|
+
progress.md # cross-cutting: what the run prints about itself
|
|
161
168
|
stages.md model-tiering.md # gates, model policy
|
|
162
169
|
conventions.md artifacts.md # host conventions, this layout
|
|
163
170
|
companion-skills.md # optional companions + preflight
|
|
164
171
|
templates/ # skeletons seeded into a host project
|
|
165
|
-
hygiene.sh
|
|
166
|
-
README.md
|
|
172
|
+
hygiene.sh docgate.sh # -> scripts/check-hygiene.sh, check-docs.sh
|
|
173
|
+
README.md # the index — and it is the list, not a copy of it
|
|
167
174
|
cursor/rules/task-pipeline.mdc # Cursor channel (self-contained rule)
|
|
168
175
|
bin/task-pipeline.js # npx installer (package task-pipeline-skill)
|
|
169
176
|
install.sh # POSIX installer
|
|
@@ -43,7 +43,8 @@ better, plus one that is required only for user-facing work.
|
|
|
43
43
|
|
|
44
44
|
| Skill / tool | Needed for | Required? | Install |
|
|
45
45
|
|---|---|---|---|
|
|
46
|
-
| **super-ux** (`ux-foundation`, `ux-flows`, `ux-scenarios`, `ux-audit`, `/ux`, `/ux-lint`) | stage 3 UX track | **Required for any user-facing task** | `/plugin marketplace add ssheleg/super-ux` → `/plugin install super-ux@super-ux` (or `npx skills add ssheleg/super-ux`) |
|
|
46
|
+
| **super-ux** (`ux-foundation`, `ux-flows`, `ux-scenarios`, `ux-audit`, `/ux`, `/ux-lint` — **and the copy half**: `copywriting`, `brand-voice`, `/brand-init`, `/copy`, `/brand-lint`, plus `/vision`) | stage 3 — the **UX track** *and* the **COPY track**. This row named six surfaces until 2026-08-10 while super-ux shipped eight skills and fifteen commands: the whole brand-and-copy half was invisible to this pipeline, so a run built scenarios and screens and then wrote the interface strings by taste | **Required for any user-facing task** | `/plugin marketplace add ssheleg/super-ux` → `/plugin install super-ux@super-ux` (or `npx skills add ssheleg/super-ux`) |
|
|
47
|
+
| **sheleg-design** (`/sheleg-design`) | stage 3 — the **VISUAL track**: tokens and themes, typography and rhythm, motion and how it degrades to rest, the visual language a brand is recognised by. It answers *how it looks*, which no other companion here answers — `super-ux` decides what the interface must do, `copywriting` how it sounds. Before 2026-08-10 this skill appeared once in the whole bundle, as a name in a list | **Recommended** on any task with a visual surface; never a gate. Absent → the run says the visual layer shipped **undesigned**, which is the honest name for picking values at the keyboard | `/plugin marketplace add ssheleg/sheleg-design` → `/plugin install sheleg-design@sheleg-design-skill` |
|
|
47
48
|
| **context7** (MCP — call tools fully qualified: `context7:resolve-library-id`, `context7:query-docs`) | stage 1 docs study | Recommended (web-search fallback) | connect the context7 MCP server |
|
|
48
49
|
| **Figma** (MCP) | stage 3 UX track, when the project designs visually — super-ux mirrors each `SCR-` screen/state into a frame | Optional, **UI + Figma-on only**. Absent → super-ux degrades to text-only *by itself and never blocks*, so shipping a UI feature with no mockups becomes a silent scope call — which is why the stage-0 sweep decides it | connect the Figma MCP server (`/mcp`, or your claude.ai connectors) |
|
|
49
50
|
| **[obsidian-wiki](https://github.com/ar9av/obsidian-wiki)** (`wiki-query`, `wiki-update`) | **stage 0 harvest** (query what's already known) **+ stage 9 sync** | **Recommended** — never a gate; absent → harvest runs on repo docs alone | `pip install obsidian-wiki` → `obsidian-wiki setup --vault /path/to/your/vault` |
|
|
@@ -77,9 +78,16 @@ exchange:
|
|
|
77
78
|
|
|
78
79
|
```
|
|
79
80
|
Pipeline companions (stage doctrine is built in — nothing to install for it):
|
|
80
|
-
✗ super-ux — this task looks user-facing; required for the UX track
|
|
81
|
+
✗ super-ux — this task looks user-facing; required for the UX track,
|
|
82
|
+
and it also owns the COPY track (copywriting, brand-voice):
|
|
81
83
|
/plugin marketplace add ssheleg/super-ux
|
|
82
84
|
/plugin install super-ux@super-ux
|
|
85
|
+
✗ sheleg-design — this task has a visual surface; it owns the VISUAL track:
|
|
86
|
+
tokens, themes, typography, rhythm, motion and its rest state:
|
|
87
|
+
/plugin marketplace add ssheleg/sheleg-design
|
|
88
|
+
/plugin install sheleg-design@sheleg-design-skill
|
|
89
|
+
(running without it — the visual layer ships undesigned,
|
|
90
|
+
and the close-out says so in those words)
|
|
83
91
|
✓ context7 — ready
|
|
84
92
|
✗ Figma MCP — this task is user-facing and the project designs in Figma
|
|
85
93
|
(docs/ux/foundation.md → Design tooling). Without it the
|
|
@@ -120,7 +128,16 @@ Install the ✗ items you want, answer the model line, then say "continue".
|
|
|
120
128
|
Rules:
|
|
121
129
|
|
|
122
130
|
- Only flag **super-ux** when the task implies a UI (the stage-0 grill decides;
|
|
123
|
-
when unsure, flag it — a false positive costs one install).
|
|
131
|
+
when unsure, flag it — a false positive costs one install). It arms **two** tracks,
|
|
132
|
+
not one: the UX chain and the copy layer.
|
|
133
|
+
- **sheleg-design**: flag it when the task has a **visual** surface — a page, a screen,
|
|
134
|
+
a themed component, a landing, a dashboard. Detect via a resolving `/sheleg-design`.
|
|
135
|
+
**A CLI, a library, a backend service or an internal script does not flag it**, and
|
|
136
|
+
neither does a purely structural change to an existing screen: choosing a palette for
|
|
137
|
+
a log parser is how a recommendation is taught to be noise, and this bundle already
|
|
138
|
+
spent a rule learning that about the browser. Absent → the run continues and says the
|
|
139
|
+
visual layer shipped **undesigned**; that is a weaker claim and the close-out records
|
|
140
|
+
it as one, exactly as it does for a surface verified by reading the diff.
|
|
124
141
|
- **obsidian-wiki**: detect via `~/.obsidian-wiki/config` or a resolving
|
|
125
142
|
`wiki-query`/`wiki-update`. Present → say `✓ ready` and use it in the harvest.
|
|
126
143
|
Absent → print the two install lines **once** and continue; never ask twice in a
|
|
@@ -140,6 +140,16 @@ whole cycle that no gate reads. It cites the measurement or it is not written
|
|
|
140
140
|
`B-NNN`, not a description**: *"next up: B-014"* can be checked against the file,
|
|
141
141
|
*"next up: the export fix"* cannot.
|
|
142
142
|
|
|
143
|
+
**And it is printed, in one fixed shape** ([`progress.md`](progress.md)):
|
|
144
|
+
|
|
145
|
+
```
|
|
146
|
+
▶ <topic> · <module> (N/M) · <id> <stage> <gate> · iter <N> · gates <N/M> · next B-NNN
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
The iteration number is a count of `iter:` lines in `.task-pipeline/run.md`, not a
|
|
150
|
+
number the agent is carrying. After a compaction the agent's count is gone and the
|
|
151
|
+
file's is not — which is the argument the build ledger already won one stage down.
|
|
152
|
+
|
|
143
153
|
## Parked at a manual gate
|
|
144
154
|
|
|
145
155
|
A fixed interval firing into a `manual` gate is a nag. Five minutes later it fires
|
|
@@ -37,9 +37,16 @@ defect. Until then it is named honestly.
|
|
|
37
37
|
## The components, each named
|
|
38
38
|
|
|
39
39
|
```
|
|
40
|
-
exposure:
|
|
40
|
+
exposure: N unverified · never checked · N releases carry one
|
|
41
41
|
```
|
|
42
42
|
|
|
43
|
+
**The example carries no numbers, deliberately.** It said `99 unverified` and
|
|
44
|
+
`31 releases since the last human confirmation` until 2026-08-10 — one figure lifted
|
|
45
|
+
from this repository's live count, which drifts, and one wording the code has never
|
|
46
|
+
printed. A worked example that disagrees with its own output teaches the wrong format
|
|
47
|
+
to every reader who trusts the doctrine over the terminal, and this one disagreed in
|
|
48
|
+
both directions at once.
|
|
49
|
+
|
|
43
50
|
- **unverified** — rows whose `Human` reads `never`.
|
|
44
51
|
- **since** — days since the newest `Human` date. When **no** row has ever been
|
|
45
52
|
confirmed, this prints the literal **`never checked`**, not `0 days`: zero would read
|
|
@@ -22,6 +22,7 @@ searches.
|
|
|
22
22
|
|
|
23
23
|
- Bookkeeping — the thing that makes detection mechanical
|
|
24
24
|
- Detection — any one of these trips the guard
|
|
25
|
+
- The review loop — a cap that measures rather than stops
|
|
25
26
|
- The break protocol
|
|
26
27
|
- When to stop and hand back
|
|
27
28
|
- Rationalizations
|
|
@@ -30,7 +31,8 @@ searches.
|
|
|
30
31
|
|
|
31
32
|
You cannot detect churn from memory, especially after compaction. Every repeating
|
|
32
33
|
pass appends one line to the run's ledger (`.task-pipeline/build/<plan>/progress.md`
|
|
33
|
-
for stage 5; `.task-pipeline/run.md` for stage-level and program-level loops
|
|
34
|
+
for stage 5; `.task-pipeline/run.md` for stage-level and program-level loops —
|
|
35
|
+
**seeded at stage 0** from [`../templates/run.md`](../templates/run.md)):
|
|
34
36
|
|
|
35
37
|
```
|
|
36
38
|
touch: <file> — pass <N> (<stage|round|module>) — reason: <finding id / gate item>
|
|
@@ -40,6 +42,12 @@ One line per file per pass. The reason must name **what forced the edit** — a
|
|
|
40
42
|
finding id, a failed gate item, an operator instruction. "Cleanup", "polish" and
|
|
41
43
|
"while I was there" are not reasons; they are churn with better manners.
|
|
42
44
|
|
|
45
|
+
**This ledger was required here from the day the file shipped and written by no run
|
|
46
|
+
until 2026-08-10.** The detection below calls itself mechanical; with no ledger it had
|
|
47
|
+
no input at all, so the guard sat on rung 1 while every reader took it for rung 3
|
|
48
|
+
([`gates.md`](gates.md) → *Axis B*). It is now seeded at stage 0 and named in that
|
|
49
|
+
stage's gate — which is the whole difference between a rule and a rule that runs.
|
|
50
|
+
|
|
43
51
|
## Detection — any one of these trips the guard
|
|
44
52
|
|
|
45
53
|
1. **Revert-oscillation.** An edit restores something an earlier pass in this run
|
|
@@ -59,7 +67,48 @@ finding id, a failed gate item, an operator instruction. "Cleanup", "polish" and
|
|
|
59
67
|
|
|
60
68
|
Caps that trip the guard by themselves: **5 fix rounds** per task
|
|
61
69
|
([`build.md`](build.md)), **2 re-entries** per stage per artifact, **3 passes** per
|
|
62
|
-
module in the program loop
|
|
70
|
+
module in the program loop, and **3 review rounds** per artifact — which is not a stop
|
|
71
|
+
but a measurement, below.
|
|
72
|
+
|
|
73
|
+
## The review loop — a cap that measures rather than stops
|
|
74
|
+
|
|
75
|
+
The caps above govern loops that **edit**. A review loop does both: the reader finds,
|
|
76
|
+
the run fixes, the reader reads again. It had no cap at all until 2026-08-10, and this
|
|
77
|
+
repository's own run stamps say what that cost — **ten rounds, ten, eight, four,
|
|
78
|
+
three** — against a stated ceiling of two re-entries per stage. Nothing tripped,
|
|
79
|
+
because a review round was named in no cap.
|
|
80
|
+
|
|
81
|
+
**A flat cap would have been the wrong fix.** Every one of those runs recorded *"none
|
|
82
|
+
from my probes"* beside its count: the reader was still finding real defects on round
|
|
83
|
+
nine. Stopping at two would have shipped them.
|
|
84
|
+
|
|
85
|
+
So the cap is a **decision point**. Default **3 rounds** per artifact, recorded in
|
|
86
|
+
`pipeline.json` → `run.review.maxRounds`. On reaching it, stop reviewing and print the
|
|
87
|
+
pair [`audit.md`](audit.md) already defines — new findings, and findings caused by this
|
|
88
|
+
run's own fixes — per round:
|
|
89
|
+
|
|
90
|
+
```
|
|
91
|
+
review cap reached — 3 rounds — artifact: test/validate.py
|
|
92
|
+
round 1: 12 new · 0 self-inflicted
|
|
93
|
+
round 2: 5 new · 1 self-inflicted
|
|
94
|
+
round 3: 1 new · 3 self-inflicted
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
- **Self-inflicted ≥ new** — the axis is exhausted ([`audit.md`](audit.md) → *Every
|
|
98
|
+
pass changes the axis*). Stop. Every remaining finding becomes a board row with its
|
|
99
|
+
evidence ([`backlog.md`](backlog.md)); none is dropped.
|
|
100
|
+
- **New > self-inflicted** — the reader is still paying. Continuing is then the
|
|
101
|
+
operator's call, made with the numbers in hand rather than out of fatigue.
|
|
102
|
+
|
|
103
|
+
**The pair is the whole point.** A round count alone says how tired everyone is; the
|
|
104
|
+
pair says whether the loop still produces anything. Measured on one file once, the
|
|
105
|
+
guard's shapes and the run's own prose disagreed — one still paying, one exhausted —
|
|
106
|
+
so a single number would have stopped the half that was working and continued the half
|
|
107
|
+
that was not.
|
|
108
|
+
|
|
109
|
+
**Rounds are counted from the ledger, never from memory**: distinct `pass N` values on
|
|
110
|
+
`touch:` lines at the review stage. A round that finds nothing ends the loop by
|
|
111
|
+
definition and needs no counting.
|
|
63
112
|
|
|
64
113
|
## The break protocol
|
|
65
114
|
|
|
@@ -109,6 +158,7 @@ far cheaper than a third round of the same argument.
|
|
|
109
158
|
| Excuse | Reality |
|
|
110
159
|
|---|---|
|
|
111
160
|
| "One more pass and it converges" | Two passes with the same reason already proved it doesn't. The disagreement is above the code. |
|
|
161
|
+
| "The reviewer is still finding things, so keep going" | Then say so with the pair: new versus self-inflicted, per round. If new still leads, that is an argument. Ten rounds with nobody counting is not. |
|
|
112
162
|
| "I'll just revert to what worked" | That is the oscillation, not the exit. Name A and B first. |
|
|
113
163
|
| "The reviewer keeps changing its mind" | Different findings on the same lines mean the requirement is ambiguous. That's a spec question. |
|
|
114
164
|
| "Tidying while I'm in the file" | Untracked edits are what make churn invisible. One reason per change, in the ledger. |
|
|
@@ -52,6 +52,7 @@ a row pointing outside the bundle is the defect this file exists to catch.
|
|
|
52
52
|
| The entry audit and what it inspects | `references/setup.md` |
|
|
53
53
|
| The ladder, seams, axis rotation, ratchets | `references/audit.md` |
|
|
54
54
|
| Loop detection and its caps | `references/loop-guard.md` |
|
|
55
|
+
| **What the run prints about itself** — the header block, the rail, the iteration line | `references/progress.md` |
|
|
55
56
|
| **The run mode** — item-by-item pacing, default off, what it never collapses | `references/continuity.md` |
|
|
56
57
|
| **The context budget** — the evidence rule and what a flush actually updates | `references/continuity.md` |
|
|
57
58
|
| **The board** — the work-list between runs, its computed priority, and the ledger seam it resolves | `references/backlog.md` |
|
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
# Progress — saying which pipeline this is, and where in it
|
|
2
|
+
|
|
3
|
+
**One job: make the run's position a printed fact instead of something the operator
|
|
4
|
+
reconstructs from the last thing that scrolled past.**
|
|
5
|
+
|
|
6
|
+
A pipeline that never says where it is has one specific failure, and it is not
|
|
7
|
+
confusion — it is that **a stage looks done because nothing printed**.
|
|
8
|
+
[`stages.md`](stages.md) opens with a checklist for exactly that reason and then marks
|
|
9
|
+
it *"copy it, tick it"*: an instruction with no gate behind it, which is rung 1
|
|
10
|
+
behaving like rung 3 ([`gates.md`](gates.md) → *Axis B*). This file is that checklist
|
|
11
|
+
promoted to something the run must emit.
|
|
12
|
+
|
|
13
|
+
**Boundary.** This file decides **what is printed and when**. It decides nothing about
|
|
14
|
+
what is true: every number on the block has a home somewhere else and is read from
|
|
15
|
+
there. A progress line that computes its own counts is the fourth copy of the truth,
|
|
16
|
+
and [`continuity.md`](continuity.md) already says what happens to those — nobody
|
|
17
|
+
maintains them and the next run reads them as current.
|
|
18
|
+
|
|
19
|
+
---
|
|
20
|
+
|
|
21
|
+
## Contents
|
|
22
|
+
|
|
23
|
+
- The two boundaries, and only those two
|
|
24
|
+
- The header block
|
|
25
|
+
- The iteration line
|
|
26
|
+
- The rail is computed, never eleven
|
|
27
|
+
- What each glyph means
|
|
28
|
+
- Every number is borrowed
|
|
29
|
+
- Absent is a word, never a zero
|
|
30
|
+
- The run ledger this reads from
|
|
31
|
+
- Rationalizations
|
|
32
|
+
|
|
33
|
+
## The two boundaries, and only those two
|
|
34
|
+
|
|
35
|
+
**Task start** and **iteration close**. Nothing else.
|
|
36
|
+
|
|
37
|
+
An iteration is already defined — *one item taken to its gate*
|
|
38
|
+
([`continuity.md`](continuity.md) → *What one iteration means*) — and that definition
|
|
39
|
+
is what makes this cheap. Printing per agent turn would put a bar above every tool
|
|
40
|
+
call, and a block that appears fifty times a run is a block nobody reads, including
|
|
41
|
+
the one time it says something.
|
|
42
|
+
|
|
43
|
+
Between the two boundaries the run prints whatever it normally prints. This file adds
|
|
44
|
+
no narration.
|
|
45
|
+
|
|
46
|
+
## The header block
|
|
47
|
+
|
|
48
|
+
Emitted **once, before stage 0's first question**, and again whenever the module
|
|
49
|
+
changes:
|
|
50
|
+
|
|
51
|
+
```
|
|
52
|
+
task-pipeline v1.34.0 · pipeline-audit · module P1 «the progress print» (1 of 4)
|
|
53
|
+
0 ✓ 1 ✓ 2 ✓ 3 ▶ 4 · 5 · 6 · 7 · 8 · 9 · 10 ·
|
|
54
|
+
███████░░░░░░░░░░░░░░░░░░░ gates 3/11 · now 3 Spec · manual
|
|
55
|
+
board B-028 · carry-over 0 rows · exposure 99 never · unlooked 0
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
Four lines, and each one answers a question an operator otherwise has to ask:
|
|
59
|
+
|
|
60
|
+
| Line | Answers |
|
|
61
|
+
|---|---|
|
|
62
|
+
| 1 | *which skill, which version, which programme, which module of how many* |
|
|
63
|
+
| 2 | *which stages are closed, which one is live* |
|
|
64
|
+
| 3 | *how far along, what is running now, will it stop for me* |
|
|
65
|
+
| 4 | *what is queued, what is deferred, what nobody has confirmed* |
|
|
66
|
+
|
|
67
|
+
**The module segment is omitted when there is no module map.** A task that stage 2
|
|
68
|
+
never decomposed has no module, and printing `(1 of 1)` turns an absence into a claim
|
|
69
|
+
— the shape [`audit.md`](audit.md) is built around. Where stage 2 recorded `single
|
|
70
|
+
module: <name>` ([`decomposition.md`](decomposition.md)), print that phrase instead.
|
|
71
|
+
|
|
72
|
+
## The iteration line
|
|
73
|
+
|
|
74
|
+
Emitted at the **close** of every iteration, one line:
|
|
75
|
+
|
|
76
|
+
```
|
|
77
|
+
▶ pipeline-audit · P1 (1/4) · 5 Dev auto · iter 3 · gates 5/11 · next B-025
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
**`next` cites a `B-NNN`, never a description.** That rule is
|
|
81
|
+
[`continuity.md`](continuity.md)'s and it is the reason this line exists at all:
|
|
82
|
+
*"next up is X"* was already the one sentence in a loop that no gate reads. A board id
|
|
83
|
+
can be checked against `docs/superpowers/backlog.md`; *"next up: the export fix"*
|
|
84
|
+
cannot.
|
|
85
|
+
|
|
86
|
+
**Nothing queued is `next —`, printed.** A loop that reaches an empty board says so;
|
|
87
|
+
omitting the field is indistinguishable from forgetting it.
|
|
88
|
+
|
|
89
|
+
## The rail is computed, never eleven
|
|
90
|
+
|
|
91
|
+
The stage ids on the rail come from the project's `pipeline.json` → `stages[]`. They
|
|
92
|
+
are **not** the eleven in [`../pipeline.example.json`](../pipeline.example.json), which
|
|
93
|
+
is this plugin's *example* flow — a host project replaces it with its own stages
|
|
94
|
+
(`SKILL.md` → *Bring your own skills*).
|
|
95
|
+
|
|
96
|
+
A bar reading `gates 5/11` in a project with six stages is a false success in the
|
|
97
|
+
purest form the pipeline has: a summary that is confidently wrong about the thing it
|
|
98
|
+
summarises, printed in the place designed to be trusted at a glance.
|
|
99
|
+
|
|
100
|
+
So the rail carries **no stage count of its own**. Read the array, print what is in it.
|
|
101
|
+
Six stages give six positions.
|
|
102
|
+
|
|
103
|
+
## What each glyph means
|
|
104
|
+
|
|
105
|
+
| Glyph | Means | Written when |
|
|
106
|
+
|---|---|---|
|
|
107
|
+
| `✓` | the stage's **gate passed** | the gate's own verdict was recorded |
|
|
108
|
+
| `▶` | in flight | the stage was entered and its gate has not returned |
|
|
109
|
+
| `·` | not entered | — |
|
|
110
|
+
| `✗` | entered, gate returned a failure | the verdict said so |
|
|
111
|
+
| `⊘` | skipped, **with the reason on the same run's record** | the short path, or a stage the brief excluded |
|
|
112
|
+
|
|
113
|
+
**`✓` means the gate passed — not that the stage was walked.** This is the whole
|
|
114
|
+
integrity of the block. A rail is a summary, and a summary is the easiest artefact in
|
|
115
|
+
a run to write from memory rather than from the record; a glyph set by recollection is
|
|
116
|
+
[`gates.md`](gates.md)'s *false success* with a nicer typeface. Derive each glyph from
|
|
117
|
+
the verdict the gate wrote, in the run ledger, and from nothing else.
|
|
118
|
+
|
|
119
|
+
**`⊘` may never be silent.** A skipped stage with no recorded reason is exactly what a
|
|
120
|
+
`·` looks like from outside, and the two mean opposite things.
|
|
121
|
+
|
|
122
|
+
## Every number is borrowed
|
|
123
|
+
|
|
124
|
+
| Field | Its home |
|
|
125
|
+
|---|---|
|
|
126
|
+
| `board B-NNN` | `docs/superpowers/backlog.md` ([`backlog.md`](backlog.md)) |
|
|
127
|
+
| `carry-over N rows` | the run's carry-over ledger, as printed beside every gate verdict |
|
|
128
|
+
| `exposure N never` | `docs/superpowers/verification.md` ([`exposure.md`](exposure.md)) |
|
|
129
|
+
| `unlooked N` | the gate's own disclosure ([`gates.md`](gates.md) → *Disclosures*) |
|
|
130
|
+
| `gates N/M` | the run ledger's verdict rows, and `pipeline.json` → `stages[]` |
|
|
131
|
+
|
|
132
|
+
**None of these is recomputed here.** If a number on the block disagrees with the
|
|
133
|
+
number beside a gate verdict, the block is wrong — that direction, always, because the
|
|
134
|
+
gate looked and the block quoted.
|
|
135
|
+
|
|
136
|
+
This also settles what the block is *not*: it is neither a ratchet nor a disclosure of
|
|
137
|
+
its own ([`gates.md`](gates.md) → *Ratchets*, *Disclosures*). It sets no floor and
|
|
138
|
+
carries no target. It is a **restatement with a citation**, and the citation is the
|
|
139
|
+
only reason a restatement is allowed here at all.
|
|
140
|
+
|
|
141
|
+
## Absent is a word, never a zero
|
|
142
|
+
|
|
143
|
+
Where a value does not exist, print the word:
|
|
144
|
+
|
|
145
|
+
```
|
|
146
|
+
board — · carry-over 0 rows · exposure — · unlooked 0
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
`exposure —` says *no verification ledger in this project*. `exposure 0` says *nothing
|
|
150
|
+
is unconfirmed*, which is the opposite claim, and it is the same inversion
|
|
151
|
+
[`exposure.md`](exposure.md) refuses when it prints `never checked` rather than
|
|
152
|
+
`0 days`. A zero standing in for an absence is how a project learns it is safe.
|
|
153
|
+
|
|
154
|
+
`carry-over 0 rows` **is** a real zero and prints as one: the ledger exists and holds
|
|
155
|
+
nothing.
|
|
156
|
+
|
|
157
|
+
## The run ledger this reads from
|
|
158
|
+
|
|
159
|
+
`.task-pipeline/run.md`, seeded at stage 0 from
|
|
160
|
+
[`../templates/run.md`](../templates/run.md), one file per run.
|
|
161
|
+
|
|
162
|
+
It already had a second owner before this file existed:
|
|
163
|
+
[`loop-guard.md`](loop-guard.md) names it as the record that makes churn detection
|
|
164
|
+
**mechanical**, and calls it the only memory that survives compaction. It was never
|
|
165
|
+
written by any run — the detector had no input, and the guard was doctrine wearing a
|
|
166
|
+
script's clothes. One file serves both readers: the guard reads the `touch:` lines,
|
|
167
|
+
this block reads the verdict rows and the iteration counter.
|
|
168
|
+
|
|
169
|
+
Three kinds of line, appended, never rewritten:
|
|
170
|
+
|
|
171
|
+
```
|
|
172
|
+
stage: 3 Spec — gate manual — verdict pass — 2026-08-10T14:02Z
|
|
173
|
+
iter: 3 — item B-025 — closed at gate 6
|
|
174
|
+
touch: test/validate.py — pass 2 (stage 7) — reason: F-014
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
**The counter is a count of `iter:` lines, not a number the agent remembers.** After a
|
|
178
|
+
compaction the agent's memory of "iteration 3" is gone and the file's is not, which is
|
|
179
|
+
the entire argument for keeping it on disk rather than in the reply.
|
|
180
|
+
|
|
181
|
+
## Rationalizations
|
|
182
|
+
|
|
183
|
+
| The excuse | What is actually true |
|
|
184
|
+
|---|---|
|
|
185
|
+
| "The operator can see the stages scroll by" | They can see that *something* printed. A stage that ended silently and a stage that never started look identical in a transcript, which is the failure `stages.md`'s checklist was written for and never enforced. |
|
|
186
|
+
| "A progress bar is decoration" | Then delete the numbers and keep the bar. The objection is really to the bar; the four borrowed counts are the payload, and they are the ones nobody prints today. |
|
|
187
|
+
| "I know which stage I'm on, I'll write the rail from memory" | Then the rail is a claim about the run rather than a reading of it, and it will be right until the run it most matters on. Derive it from the verdicts. |
|
|
188
|
+
| "There's no module map, I'll put (1 of 1)" | An undecomposed task has no module. `(1 of 1)` is an invented denominator, and a reader cannot tell it from a real one. |
|
|
189
|
+
| "Printing it every iteration is noise" | One line. The block is four, and it appears at task start. If that is noise, the run is emitting far worse elsewhere. |
|
|
190
|
+
| "The ledger is bureaucracy, I'll count iterations in my head" | Your head does not survive compaction. That is not a hypothetical here — it is why `loop-guard.md` asked for this file in the first place. |
|