task-pipeline-skill 1.88.1 → 1.90.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +109 -0
- package/README.md +2 -2
- package/SKILL-CARD.md +1 -1
- package/package.json +3 -3
- package/plugins/task-pipeline/.claude-plugin/plugin.json +1 -1
- package/plugins/task-pipeline/agents/verifier-product.md +3 -1
- package/plugins/task-pipeline/agents/verifier-visual.md +115 -0
- package/plugins/task-pipeline/agents/verifier.md +2 -1
- package/plugins/task-pipeline/skills/task-pipeline/SKILL.md +5 -5
- package/plugins/task-pipeline/skills/task-pipeline/graph.schema.json +22 -1
- package/plugins/task-pipeline/skills/task-pipeline/pipeline.example.json +4 -4
- package/plugins/task-pipeline/skills/task-pipeline/references/acceptance.md +12 -0
- package/plugins/task-pipeline/skills/task-pipeline/references/audit.md +14 -8
- package/plugins/task-pipeline/skills/task-pipeline/references/browser.md +107 -3
- package/plugins/task-pipeline/skills/task-pipeline/references/build.md +41 -2
- package/plugins/task-pipeline/skills/task-pipeline/references/certification.md +46 -5
- package/plugins/task-pipeline/skills/task-pipeline/references/companion-skills.md +29 -0
- package/plugins/task-pipeline/skills/task-pipeline/references/conventions.md +3 -3
- package/plugins/task-pipeline/skills/task-pipeline/references/doctrine-map.md +1 -1
- package/plugins/task-pipeline/skills/task-pipeline/references/grill.md +15 -9
- package/plugins/task-pipeline/skills/task-pipeline/references/loop-guard.md +23 -0
- package/plugins/task-pipeline/skills/task-pipeline/references/portability.md +1 -1
- package/plugins/task-pipeline/skills/task-pipeline/references/spec.md +27 -3
- package/plugins/task-pipeline/skills/task-pipeline/references/stages.md +83 -13
- package/plugins/task-pipeline/skills/task-pipeline/references/work-graph.md +2 -2
- package/plugins/task-pipeline/skills/task-pipeline/scripts/graph.py +45 -16
- package/plugins/task-pipeline/skills/task-pipeline/scripts/stage_checkpoint.py +21 -1
- package/plugins/task-pipeline/skills/task-pipeline/scripts/visual_gate.py +728 -0
- package/plugins/task-pipeline/skills/task-pipeline/templates/brief.md +5 -2
- package/plugins/task-pipeline/skills/task-pipeline/templates/browser-claims.json +223 -1
- package/plugins/task-pipeline/skills/task-pipeline/templates/run.md +10 -0
|
@@ -187,10 +187,28 @@ never that the work was skipped quietly.
|
|
|
187
187
|
- **UI early-detect:** one branch of the grill is always "does this touch a
|
|
188
188
|
user-facing surface (web/mobile/CLI/TUI)?". If yes → surface **super-ux**
|
|
189
189
|
now (use it if installed; otherwise give the install line — see SKILL.md
|
|
190
|
-
*Prerequisites*); this arms the stage-3 UX track.
|
|
190
|
+
*Prerequisites*); this arms the stage-3 UX track. **And the same branch records
|
|
191
|
+
the surface's class** — the next bullet.
|
|
192
|
+
- **The surface class.** Every user-facing task's brief carries
|
|
193
|
+
`surface_class: flagship | product | internal | ad`, and the class selects the gate
|
|
194
|
+
profile the visual layer is held to for the rest of the run:
|
|
195
|
+
|
|
196
|
+
| Class | What it is | Director record (stage 3) | Visual half (stages 5–6) | Stage 10 |
|
|
197
|
+
|---|---|---|---|---|
|
|
198
|
+
| `flagship` | the surface a product is judged by — landing, onboarding, paywall, a hero screen | the full record | **gate**: full matrix, pairwise across every axis, the full rubric | approved contact sheet |
|
|
199
|
+
| `product` | an ordinary screen of the product | the short record — Brief, Mode, References, Markers, Open | **gate**: every state and the mandatory pairs; the gate items of the rubric | approved contact sheet |
|
|
200
|
+
| `internal` | an admin panel, an internal tool, a CLI | none owed | recommended; the project linter is the floor | the functional look |
|
|
201
|
+
| `ad` | a creative that runs as an advertisement or a store asset | Brief, Mode, References, Markers, ADA (its rubric profile and safe zones), Open | **gate**: the ad profile and the safe zones | approved contact sheet |
|
|
202
|
+
|
|
203
|
+
Ask it as one question with a recommended answer read off the request — a landing or a
|
|
204
|
+
paywall is `flagship` unless the operator says otherwise — and never leave it to stage
|
|
205
|
+
5: a class decided at build time is decided by whoever wants the build to pass. A
|
|
206
|
+
work-graph node that builds the surface copies the class as `surface_class`, and
|
|
207
|
+
`graph.py certify` then owes the fourth, `visual` reading on `flagship` and `product`
|
|
208
|
+
([`certification.md`](certification.md) → *The fourth reading*).
|
|
191
209
|
- **Artifact:** lock the resolved decisions into a **task brief** committed at
|
|
192
|
-
`<artifacts>/specs/YYYY-MM-DD-<topic>-brief.md` (scope, users/UI verdict,
|
|
193
|
-
constraints, assumptions, explicitly-deferred items, done-criteria) **plus the
|
|
210
|
+
`<artifacts>/specs/YYYY-MM-DD-<topic>-brief.md` (scope, users/UI verdict and,
|
|
211
|
+
for a user-facing task, `surface_class`, constraints, assumptions, explicitly-deferred items, done-criteria) **plus the
|
|
194
212
|
autonomy sweep's per-stage answers and the model decision**. Seed it from
|
|
195
213
|
the skill's `templates/brief.md` skeleton — but only when absent, never
|
|
196
214
|
overwrite an existing brief. Stages 2–4 build on this brief; stages 5–10 read
|
|
@@ -214,7 +232,8 @@ never that the work was skipped quietly.
|
|
|
214
232
|
a recorded answer or an explicit deferral, **every answer that contradicted a
|
|
215
233
|
harvested source has a recorded resolution** (which governs, and whether the doc
|
|
216
234
|
is now stale), no open contradictions, **every
|
|
217
|
-
autonomy-sweep row is answered or explicitly marked "stop and ask here"**,
|
|
235
|
+
autonomy-sweep row is answered or explicitly marked "stop and ask here"**, **a
|
|
236
|
+
user-facing task's brief names its `surface_class`**, the
|
|
218
237
|
**REQ table is written and every row names its check**, the carry-over ledger is
|
|
219
238
|
seeded, **`.task-pipeline/run.md` exists and the header block has been printed**
|
|
220
239
|
([`progress.md`](progress.md)), the model decision is recorded, and the operator
|
|
@@ -314,8 +333,9 @@ never that the work was skipped quietly.
|
|
|
314
333
|
ssheleg/super-ux`). super-ux builds a traced chain — walk it top-down (see its
|
|
315
334
|
`system-map.md`):
|
|
316
335
|
0. **Destination first, when Figma is on.** The brief already names the team/org
|
|
317
|
-
and the
|
|
318
|
-
|
|
336
|
+
and the files — **one per surface** (App, Web, ASO: store screenshots, icon and
|
|
337
|
+
logo), never one file for every frame; `docs/ux/foundation.md` → *Design tooling*
|
|
338
|
+
is the canonical record. Confirm each **resolves** before drawing. **Never create a file while a
|
|
319
339
|
recorded one resolves; if it doesn't resolve, stop and ask — never create a
|
|
320
340
|
replacement** (that is the duplicate, and it hides a permissions problem).
|
|
321
341
|
A creation happens at most once, in the named team, and its URL is written to
|
|
@@ -350,7 +370,10 @@ never that the work was skipped quietly.
|
|
|
350
370
|
disagree together on one screen. The full doctrine — each track's scope and
|
|
351
371
|
out-of-scope, the refusal sentences, the four contradictions the check
|
|
352
372
|
catches — is [`spec.md`](spec.md) → *The COPY and VISUAL tracks, and their
|
|
353
|
-
convergence*, its one home.
|
|
373
|
+
convergence*, its one home. **The VISUAL track leaves a trace, not a fact**: a
|
|
374
|
+
director record at `docs/design/<surface>/director-record.md`, whose fields the gate
|
|
375
|
+
reads by the brief's `surface_class` (`python3 scripts/visual_gate.py record <file>
|
|
376
|
+
--class <c>`), and a refusal is that same file saying `Mode: declined` and why.
|
|
354
377
|
- **Spec:** write the approved design to
|
|
355
378
|
`<artifacts>/specs/YYYY-MM-DD-<topic>-design.md` and commit it. Lock all
|
|
356
379
|
shared contracts (types, schemas, signatures, file layout). For UI tasks the
|
|
@@ -367,12 +390,23 @@ never that the work was skipped quietly.
|
|
|
367
390
|
designed, validated and approved; scenarios validated in `docs/ux/scenarios.md`;
|
|
368
391
|
the linter passes; every user-facing spec requirement traces to a scenario ID
|
|
369
392
|
(or an explicit v1-mode/tiny-project waiver by the operator). **With Figma on:
|
|
370
|
-
the canonical record names one file
|
|
371
|
-
|
|
372
|
-
|
|
393
|
+
the canonical record names one file per surface (App, Web, ASO), and every
|
|
394
|
+
`screens.md` frame link's `:fileKey` is one of them** — a string match, not a
|
|
395
|
+
judgement (`python3 scripts/visual_gate.py filekeys --record docs/ux/foundation.md
|
|
396
|
+
--screens docs/ux/screens.md`); a key outside the set means the run drew in a file
|
|
397
|
+
nobody recorded and nobody will open. **Every user-facing string went
|
|
373
398
|
through the COPY track or the refusal is recorded**, and **the visual layer went
|
|
374
399
|
through the VISUAL track or the refusal is recorded** — a recorded refusal passes
|
|
375
400
|
this gate and an unmentioned one does not, which is the only difference that matters.
|
|
401
|
+
**And the VISUAL track is checked by its trace, not by the fact that it ran:** on a
|
|
402
|
+
`flagship`, `product` or `ad` surface the director record exists and carries the
|
|
403
|
+
fields its class owes — `python3 scripts/visual_gate.py record
|
|
404
|
+
docs/design/<surface>/director-record.md --class <surface_class>` exits 0. That
|
|
405
|
+
command checks the headings itself and runs the record's own validator from `sheleg-design`
|
|
406
|
+
(`--check-record`) where it is installed and new enough; where it is not, the
|
|
407
|
+
validator reads **NOT_RUN** beside the verdict — never PASS — and the gate stands on
|
|
408
|
+
the floor, said so. A record saying `Mode: declined` with its reason is the refusal,
|
|
409
|
+
and it passes.
|
|
376
410
|
**Where both tracks ran, their convergence check is recorded** — findings with the
|
|
377
411
|
ruling, or `Tracks converge: clean`; a screen where each track is right alone and they
|
|
378
412
|
disagree together is the defect neither track's own review can see.
|
|
@@ -435,7 +469,18 @@ never that the work was skipped quietly.
|
|
|
435
469
|
requires — never parked silently** — a browser finding filed without a ruling is the
|
|
436
470
|
diff-review verdict wearing a screenshot; the look was worth taking only if it can
|
|
437
471
|
still change the code or is on record as deliberately not doing so. Absent, say the surface was verified by reading
|
|
438
|
-
the diff and treat it as the weaker claim it is.
|
|
472
|
+
the diff and treat it as the weaker claim it is. **On a surface whose brief names a
|
|
473
|
+
`surface_class`, the project linter runs here too**, after each task that changes a
|
|
474
|
+
rendered surface — `python3 scripts/visual_gate.py lint <dir>`; an S1 finding is
|
|
475
|
+
fixed in the task, and NOT_RUN (exit 3) is recorded as such
|
|
476
|
+
([`browser.md`](browser.md) → *The visual half*). **With Figma on, the token drift
|
|
477
|
+
probe runs beside it** after a task that touches the token file or a tokenised
|
|
478
|
+
component: `python3 scripts/visual_gate.py tokens --figma <variables.json> --css
|
|
479
|
+
<tokens.css>`. A FAIL is fixed in the task, renaming one side to match the other.
|
|
480
|
+
No export is NOT_RUN, recorded as such. A component whose API the task changed
|
|
481
|
+
updates its Code Connect mapping in the same change
|
|
482
|
+
([`build.md`](build.md) → *Code Connect, kept*). A slop marker caught while the
|
|
483
|
+
implementer is dispatched costs a line; caught on the contact sheet it costs a round. Stage 6 repeats this over the whole tree; this one catches it while the
|
|
439
484
|
implementer that wrote it is still dispatched. The matrix pointed this companion at
|
|
440
485
|
stages 5–6 from the day it was added and **this stage had never named it** — found by
|
|
441
486
|
the guard comparing the two, not by a reader.
|
|
@@ -472,7 +517,10 @@ never that the work was skipped quietly.
|
|
|
472
517
|
printed beside their floors; the **full** suite is green (not just the new tests); new/changed code
|
|
473
518
|
is covered; **every check this run added or widened has been probed both ways —
|
|
474
519
|
seen rejecting a planted defect and passing the clean tree, asserted on its exit
|
|
475
|
-
code** ([`probing.md`](probing.md)); no `skip`/`xfail` smuggling a red suite past the gate
|
|
520
|
+
code** ([`probing.md`](probing.md)); no `skip`/`xfail` smuggling a red suite past the gate; **on a
|
|
521
|
+
`flagship`, `product` or `ad` surface where the VISUAL track ran, the visual half's
|
|
522
|
+
`visual_gate.py sheet` exits 0** — NOT_RUN stops and asks, it is not green
|
|
523
|
+
([`browser.md`](browser.md) → *The visual half*). Never advance
|
|
476
524
|
to deploy on a red or partial run. **The carry-over count is printed beside this
|
|
477
525
|
verdict** — a ratchet nobody prints is a TODO with a better name
|
|
478
526
|
([`audit.md`](audit.md)) — **and so are the disclosures**, `abstained` and
|
|
@@ -502,6 +550,21 @@ never that the work was skipped quietly.
|
|
|
502
550
|
run that answers *the surface was checked* by pointing at its spec suite has answered
|
|
503
551
|
a different question. Where the suite is the thing that changed, the look is what
|
|
504
552
|
proves it runs against a page that renders.
|
|
553
|
+
- **The visual half is a separate check, and on most user-facing classes a gate.** The
|
|
554
|
+
look above reads the accessibility tree; it cannot say whether the surface looks like
|
|
555
|
+
what was designed. Where the stage-3 VISUAL track ran on a `flagship`, `product` or
|
|
556
|
+
`ad` surface, stage 6 also owes **the contact sheet** — frames over the `SCR-NN` states
|
|
557
|
+
× viewport × theme × text × locale (pairwise, plus the mandatory pairs), each with its
|
|
558
|
+
capture record, diffed against its Figma frame or approved baseline, the project
|
|
559
|
+
linter run, the token drift probe run over the whole token file where Figma is on
|
|
560
|
+
(`visual_gate.py tokens`; NOT_RUN without an export, never PASS), and the rubric
|
|
561
|
+
read by a judge that is not the builder — and its command
|
|
562
|
+
exits 0: `python3 scripts/visual_gate.py sheet <contact-sheet.json> --class
|
|
563
|
+
<surface_class> --artifact-root <frames>` (NOT_RUN, exit 3, is not green). On
|
|
564
|
+
`internal` it is recommended and the linter is the floor. The re-render budget is
|
|
565
|
+
**one, two at most**, then `unresolved` to the person — never round three — and only
|
|
566
|
+
external, specific feedback starts a round. The procedure and the order of the checks:
|
|
567
|
+
[`browser.md`](browser.md) → *The visual half*.
|
|
505
568
|
- **What the look finds is fixed here.** A rendering defect found at stage 6 is a
|
|
506
569
|
stage-6 finding: fix it, look again, then call the stage green. Filing it to the
|
|
507
570
|
board and advancing is how a run reports *checked in a browser* for a page it has
|
|
@@ -672,7 +735,14 @@ never that the work was skipped quietly.
|
|
|
672
735
|
surface**: read it against the spec section that covers its `SCR-` id and against
|
|
673
736
|
what shipped. The super-ux linter proves a frame link exists, is named right and
|
|
674
737
|
is not stale — it cannot read the picture, so a frame promising a limit, a meter
|
|
675
|
-
or a tier nobody built passes every lint there is.
|
|
738
|
+
or a tier nobody built passes every lint there is. **Where the VISUAL track ran, the
|
|
739
|
+
walk carries one more row: visual intent ↔ final render** — the director record (the
|
|
740
|
+
brief's falsifier, the signature moment, the rubric written before any render) read
|
|
741
|
+
against the **approved contact sheet**. On a `flagship`, `product` or `ad` surface the
|
|
742
|
+
sheet is approved (`visual_gate.py sheet … --require-approval` exits 0), and the last
|
|
743
|
+
`review:` line per surface — how many human rounds it took — is copied into the
|
|
744
|
+
acceptance file, because `.task-pipeline/run.md` does not outlive the run
|
|
745
|
+
([`audit.md`](audit.md) → the `V→R` seam). An absence
|
|
676
746
|
becomes a **new REQ row with its check** and *then* the table is written;
|
|
677
747
|
appending after the table is how acceptance goes green over a gap. Findings that
|
|
678
748
|
belong to a lower layer go back to that layer (spec → stage 3, plan → stage 4).
|
|
@@ -59,7 +59,7 @@ conditional on the code, never merely sequenced after it.**
|
|
|
59
59
|
| `goal` | the release goal | `0` · `3` unstated |
|
|
60
60
|
| `add` | the id it allocated | `0` · `1` refused |
|
|
61
61
|
| `park` | the id and the reason | `0` · `1` refused |
|
|
62
|
-
| `certify` | the round, and on a failure every `breaks` finding with its fix and its check | `0`
|
|
62
|
+
| `certify` | the round, and on a failure every `breaks` finding with its fix and its check | `0` every owed tier passed — three, or four with `visual` on a flagship or product node · `1` a tier failed, a required one is missing, or a report is malformed |
|
|
63
63
|
| `close` | the goal, the new frontier count, and what was not verified | `0` · `1` refused **or the verdict stops the run** |
|
|
64
64
|
| `producer` | what produced this proof — actor, model, runtime, skill, config, commit, trace | `0` |
|
|
65
65
|
| `doctrine` | how many of the bundle's reference files this run opened | `0` |
|
|
@@ -84,7 +84,7 @@ A **`parked`** node is the single exemption: it is the one node nobody will clos
|
|
|
84
84
|
*n/a — parked* in that field is confidence without correctness. `park` never removes what
|
|
85
85
|
the node said it would run.
|
|
86
86
|
|
|
87
|
-
**A node is closed by three readings, not one.** `certify` takes one tier report from each of `unit`, `seam` and `product` — dispatched blind and in parallel — requires all three to pass, and assembles the seven-key verdict `close` consumes. `close`'s contract is unchanged; what changed is that the verdict is now built from three readings at different distances instead of written from one, because a change can be correct where it was made and wrong one level out. A failing round records itself and leaves the node open. Doctrine: [`certification.md`](certification.md).
|
|
87
|
+
**A node is closed by three readings, not one.** `certify` takes one tier report from each of `unit`, `seam` and `product` — dispatched blind and in parallel — requires all three to pass, and assembles the seven-key verdict `close` consumes. `close`'s contract is unchanged; what changed is that the verdict is now built from three readings at different distances instead of written from one, because a change can be correct where it was made and wrong one level out. A failing round records itself and leaves the node open. **A node whose `surface_class` is `flagship` or `product` owes a fourth, `visual` report** — the contact sheet read against the director record — and `certify` refuses the round without it; on any other node a `visual` report is accepted when given. Doctrine: [`certification.md`](certification.md).
|
|
88
88
|
|
|
89
89
|
**`close` stamps the commit; the verifier never supplies it.** A verdict written after the
|
|
90
90
|
tree moved is evidence about a different tree, and an agent cannot name the wrong commit if
|
|
@@ -268,6 +268,14 @@ def violations(graph):
|
|
|
268
268
|
"two commands cannot say which one closed the node, and the "
|
|
269
269
|
"verifier reports its output as one evidence row")
|
|
270
270
|
|
|
271
|
+
# The surface class (stage 0, `references/stages.md`) decides whether `certify`
|
|
272
|
+
# owes the fourth, `visual` reading. A class outside the four would silently
|
|
273
|
+
# require nothing, which is the one outcome a typo must not have.
|
|
274
|
+
if "surface_class" in n and n["surface_class"] not in SURFACE_CLASSES:
|
|
275
|
+
out.append(f"{nid}: surface_class is {n['surface_class']!r} — it must be one of "
|
|
276
|
+
f"{', '.join(SURFACE_CLASSES)}. A class nobody recognises requires no "
|
|
277
|
+
"visual reading, so a typo here would drop the tier that reads the pixels")
|
|
278
|
+
|
|
271
279
|
if n.get("status") == "done":
|
|
272
280
|
ev = n.get("evidence")
|
|
273
281
|
if not isinstance(ev, list) or not [e for e in ev
|
|
@@ -574,6 +582,9 @@ def verdict_violations(v):
|
|
|
574
582
|
# tests of the neighbours the change can reach
|
|
575
583
|
# product one level out again — the documentation, the scenarios, how this
|
|
576
584
|
# behaviour interacts with the rest of the product
|
|
585
|
+
# visual the pixels — the contact sheet, the director record, the project
|
|
586
|
+
# linter and the rubric. Owed only where the node's `surface_class` is
|
|
587
|
+
# flagship or product; accepted, and counted, wherever it is given
|
|
577
588
|
#
|
|
578
589
|
# **All three must pass, and blind is the point.** Three agents that read each
|
|
579
590
|
# other's reports are one opinion with three signatures; the disagreement is the
|
|
@@ -584,6 +595,15 @@ def verdict_violations(v):
|
|
|
584
595
|
# stamp at three levels is worse than one verifier, because it costs three times
|
|
585
596
|
# as much and reads as three times the assurance.
|
|
586
597
|
TIERS = ("unit", "seam", "product")
|
|
598
|
+
# The fourth reading, `visual`, reads the PIXELS — the contact sheet, the director
|
|
599
|
+
# record, the project linter's output and the rubric — which none of the three opens.
|
|
600
|
+
# It is owed by a node whose `surface_class` is one of VISUAL_REQUIRED, accepted when
|
|
601
|
+
# given on any other node, and blind to the other three exactly as they are to each
|
|
602
|
+
# other. `references/certification.md` → *The fourth reading*.
|
|
603
|
+
VISUAL_TIER = "visual"
|
|
604
|
+
ALL_TIERS = TIERS + (VISUAL_TIER,)
|
|
605
|
+
SURFACE_CLASSES = ("flagship", "product", "internal", "ad")
|
|
606
|
+
VISUAL_REQUIRED = ("flagship", "product")
|
|
587
607
|
TIER_KEYS = ("node", "tier", "verdict", "scope", "confirms", "findings",
|
|
588
608
|
"evidence", "not_examined")
|
|
589
609
|
TIER_VERDICTS = ("pass", "fail")
|
|
@@ -596,10 +616,10 @@ SEVERITIES = ("breaks", "risk")
|
|
|
596
616
|
# "the unit tier's verdict…"). Widened to the tense and possessive forms the
|
|
597
617
|
# reader planted; still a closed list on purpose — a looser net here starts
|
|
598
618
|
# matching a report's honest prose about its OWN tier.
|
|
599
|
-
CROSS_TIER = re.compile(r"\b(?:unit|seam|product)\s+tier(?:'s)?\s+"
|
|
619
|
+
CROSS_TIER = re.compile(r"\b(?:unit|seam|product|visual)\s+tier(?:'s)?\s+"
|
|
600
620
|
r"(?:passed|failed|says|said|confirm\w*|verdict|report)"
|
|
601
621
|
r"|\btier\s+\d\s+(?:passed|failed|says|said|confirm\w*)"
|
|
602
|
-
r"|as\s+the\s+(?:unit|seam|product)\s+tier", re.I)
|
|
622
|
+
r"|as\s+the\s+(?:unit|seam|product|visual)\s+tier", re.I)
|
|
603
623
|
|
|
604
624
|
|
|
605
625
|
def tier_violations(t):
|
|
@@ -622,9 +642,9 @@ def tier_violations(t):
|
|
|
622
642
|
|
|
623
643
|
if not isinstance(t["node"], str) or not t["node"].startswith(NODE_ID):
|
|
624
644
|
out.append("tier report `node` is %r, which is not a node id" % (t["node"],))
|
|
625
|
-
if t["tier"] not in
|
|
645
|
+
if t["tier"] not in ALL_TIERS:
|
|
626
646
|
out.append("tier report `tier` is %r — it must be one of %s"
|
|
627
|
-
% (t["tier"], ", ".join(
|
|
647
|
+
% (t["tier"], ", ".join(ALL_TIERS)))
|
|
628
648
|
if t["verdict"] not in TIER_VERDICTS:
|
|
629
649
|
out.append("tier report `verdict` is %r — it must be `pass` or `fail`, because "
|
|
630
650
|
"a certification that admits a third state admits a maybe"
|
|
@@ -1482,6 +1502,14 @@ def cmd_certify(graph, args):
|
|
|
1482
1502
|
die("certification is missing the %s report(s) — all three are required, because "
|
|
1483
1503
|
"the level nobody read is the level the defect survives at"
|
|
1484
1504
|
% ", ".join("`%s`" % m for m in missing))
|
|
1505
|
+
sclass = node.get("surface_class")
|
|
1506
|
+
if sclass in VISUAL_REQUIRED and VISUAL_TIER not in reports:
|
|
1507
|
+
die("certification is missing the `%s` report — %s is a %s surface, and on one the "
|
|
1508
|
+
"pixels are part of the requirement: none of unit, seam or product opens the "
|
|
1509
|
+
"contact sheet, so without the fourth reading nobody looked at what a user sees "
|
|
1510
|
+
"(references/certification.md → *The fourth reading*)" % (VISUAL_TIER, nid, sclass))
|
|
1511
|
+
# The tiers THIS round read, in a stable order: the three always, `visual` when given.
|
|
1512
|
+
tiers = [x for x in ALL_TIERS if x in reports]
|
|
1485
1513
|
|
|
1486
1514
|
# The stamp, read here and never accepted from a report — same law as `close`.
|
|
1487
1515
|
import subprocess
|
|
@@ -1493,7 +1521,7 @@ def cmd_certify(graph, args):
|
|
|
1493
1521
|
|
|
1494
1522
|
prior = node.get("certification") or {}
|
|
1495
1523
|
round_no = int(prior.get("round") or 0) + 1
|
|
1496
|
-
tiers_now = {x: reports[x]["verdict"] for x in
|
|
1524
|
+
tiers_now = {x: reports[x]["verdict"] for x in tiers}
|
|
1497
1525
|
history = list(prior.get("history") or []) + [tiers_now]
|
|
1498
1526
|
node["certification"] = {
|
|
1499
1527
|
"round": round_no,
|
|
@@ -1502,11 +1530,11 @@ def cmd_certify(graph, args):
|
|
|
1502
1530
|
"history": history,
|
|
1503
1531
|
}
|
|
1504
1532
|
|
|
1505
|
-
failed = [x for x in
|
|
1533
|
+
failed = [x for x in tiers if tiers_now[x] == "fail"]
|
|
1506
1534
|
|
|
1507
1535
|
# Churn, measured. A tier that has failed in every round so far is the one the
|
|
1508
1536
|
# operator needs named; counting it here is what makes the loop visible.
|
|
1509
|
-
churning = [x for x in
|
|
1537
|
+
churning = [x for x in tiers
|
|
1510
1538
|
if len(history) >= 2 and all(h.get(x) == "fail" for h in history)]
|
|
1511
1539
|
|
|
1512
1540
|
save(args.graph, graph)
|
|
@@ -1545,19 +1573,19 @@ def cmd_certify(graph, args):
|
|
|
1545
1573
|
# exactly what `can_continue_around: true` says)
|
|
1546
1574
|
verdict = {
|
|
1547
1575
|
"node": nid,
|
|
1548
|
-
"done": [c for x in
|
|
1576
|
+
"done": [c for x in tiers for c in reports[x]["confirms"]],
|
|
1549
1577
|
"not_done": [],
|
|
1550
1578
|
"not_verified": ["%s: %s" % (x, n)
|
|
1551
|
-
for x in
|
|
1579
|
+
for x in tiers for n in reports[x]["not_examined"]],
|
|
1552
1580
|
"blockers": [
|
|
1553
1581
|
{"what": "%s (%s, found by the `%s` tier)" % (f["what"], f["where"], x),
|
|
1554
1582
|
"blocks": [], "can_continue_around": True}
|
|
1555
|
-
for x in
|
|
1583
|
+
for x in tiers for f in reports[x]["findings"]
|
|
1556
1584
|
if f.get("severity") == "risk"
|
|
1557
1585
|
],
|
|
1558
1586
|
"replan": {"possible": True, "add": [], "park": [],
|
|
1559
|
-
"why": "certified at all
|
|
1560
|
-
"evidence": ["%s: %s" % (x, e) for x in
|
|
1587
|
+
"why": "certified at all %d tiers in round %d" % (len(tiers), round_no)},
|
|
1588
|
+
"evidence": ["%s: %s" % (x, e) for x in tiers for e in reports[x]["evidence"]],
|
|
1561
1589
|
}
|
|
1562
1590
|
# Proof identity (FIX-PF-02.01): the certification tested THIS tree, so it
|
|
1563
1591
|
# records the commit it tested into the verdict it hands `close`. Without it
|
|
@@ -1579,7 +1607,7 @@ def cmd_certify(graph, args):
|
|
|
1579
1607
|
# cannot hand the run a verdict its own consumer refuses.
|
|
1580
1608
|
broken = verdict_violations(verdict)
|
|
1581
1609
|
if broken:
|
|
1582
|
-
die("
|
|
1610
|
+
die("every tier passed and the assembled verdict is still malformed — this "
|
|
1583
1611
|
"is a defect in `certify`, not in the reports:\n " + "\n ".join(broken))
|
|
1584
1612
|
|
|
1585
1613
|
out = args.verdict_out or os.path.join(os.path.dirname(args.graph) or ".",
|
|
@@ -1589,7 +1617,7 @@ def cmd_certify(graph, args):
|
|
|
1589
1617
|
json.dump(verdict, fh, indent=2, ensure_ascii=False)
|
|
1590
1618
|
fh.write("\n")
|
|
1591
1619
|
os.replace(tmp, out)
|
|
1592
|
-
print("%s: certified at
|
|
1620
|
+
print("%s: certified at %s in round %d" % (nid, ", ".join(tiers), round_no))
|
|
1593
1621
|
print("verdict written to %s — close it with:" % out)
|
|
1594
1622
|
print(" graph.py close --verdict %s" % out)
|
|
1595
1623
|
return 0
|
|
@@ -1797,8 +1825,9 @@ VERBS = {
|
|
|
1797
1825
|
"coverage": (cmd_coverage, "every requirement and the nodes serving it; exits 1 on a gap"),
|
|
1798
1826
|
"add": (cmd_add, "add a node mid-run"),
|
|
1799
1827
|
"park": (cmd_park, "park a node, carrying the reason"),
|
|
1800
|
-
"certify": (cmd_certify, "require three independent tier reports
|
|
1801
|
-
"the verdict
|
|
1828
|
+
"certify": (cmd_certify, "require three independent tier reports (four on a "
|
|
1829
|
+
"flagship or product surface), then emit the verdict "
|
|
1830
|
+
"`close` consumes"),
|
|
1802
1831
|
"close": (cmd_close, "consume a verdict, close one node and re-plan"),
|
|
1803
1832
|
}
|
|
1804
1833
|
|
|
@@ -37,6 +37,9 @@ What it keeps, and where:
|
|
|
37
37
|
the episode Observatory kept (`keptAs`), the token is dropped, and the run is told to read
|
|
38
38
|
the workflow before writing again. A stale writer therefore stops after one refusal.
|
|
39
39
|
|
|
40
|
+
Ledger lines it reads besides `stage:`: `review:` (the contact sheet's human rounds, stage 6
|
|
41
|
+
and 10 on a visual surface), each carried as evidence on the stage it names.
|
|
42
|
+
|
|
40
43
|
Ledger lines it appends, all of the existing `event:` shape:
|
|
41
44
|
|
|
42
45
|
event: memory — checkpoint <stepId> rev <n> <workflowId> — <ISO-8601>
|
|
@@ -63,6 +66,10 @@ STATE_NAME = "memory.json"
|
|
|
63
66
|
OWNER = "agent:task-pipeline"
|
|
64
67
|
STAGE = re.compile(r"^stage:\s*(\d+)\s+(.+?)\s+—\s+gate\s+(\S+)\s+—\s+verdict\s+(\S+)\s+—\s+(\S+)\s*$")
|
|
65
68
|
TOPIC = re.compile(r"^Run:\s*`([^`]+)`")
|
|
69
|
+
# The contact sheet's human rounds (`references/browser.md` → *The visual half*): one line per
|
|
70
|
+
# return or approval. Carried into the checkpoint of the stage it names, so the number of
|
|
71
|
+
# passes a surface took is measured at the boundary rather than remembered at the end.
|
|
72
|
+
REVIEW = re.compile(r"^review:\s*(\d+)\s+—\s+surface\s+(.+?)\s+—\s+rounds\s+(\d+)\s+—\s+(\S+)\s+—\s+\S+\s*$")
|
|
66
73
|
STATUS = {"pass": "done", "skip": "done", "fail": "blocked"}
|
|
67
74
|
|
|
68
75
|
|
|
@@ -81,6 +88,17 @@ def _now() -> str:
|
|
|
81
88
|
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
82
89
|
|
|
83
90
|
|
|
91
|
+
def _reviews(path: pathlib.Path) -> dict[int, list[str]]:
|
|
92
|
+
"""`review:` lines by the stage id they name, as evidence strings."""
|
|
93
|
+
out: dict[int, list[str]] = {}
|
|
94
|
+
for line in path.read_text(encoding="utf-8").splitlines():
|
|
95
|
+
m = REVIEW.match(line)
|
|
96
|
+
if m:
|
|
97
|
+
out.setdefault(int(m.group(1)), []).append(
|
|
98
|
+
f"review: {m.group(2)} rounds {m.group(3)} {m.group(4)}")
|
|
99
|
+
return out
|
|
100
|
+
|
|
101
|
+
|
|
84
102
|
def _read_ledger(path: pathlib.Path) -> tuple[str, list[tuple]]:
|
|
85
103
|
try:
|
|
86
104
|
text = path.read_text(encoding="utf-8")
|
|
@@ -160,8 +178,10 @@ def emit(ledger: pathlib.Path, project: str | None, constraints: list[str],
|
|
|
160
178
|
# one thing a successor must never act without (found by the live receipt, 2026-10-05).
|
|
161
179
|
constraints = list(dict.fromkeys([*state.get("constraints", []), *constraints]))
|
|
162
180
|
credentials = list(dict.fromkeys([*state.get("credentials", []), *credentials]))
|
|
181
|
+
reviews = _reviews(ledger)
|
|
163
182
|
done = [{"step_id": f"stage-{s[0]}", "result": f"{s[1]}: gate {s[2]}, verdict {s[3]} at {s[4]}",
|
|
164
|
-
"evidence": [f"ledger: {ledger.name}"
|
|
183
|
+
"evidence": [f"ledger: {ledger.name}", *reviews.get(s[0], [])]}
|
|
184
|
+
for s in stages if s[3] in ("pass", "skip")]
|
|
165
185
|
nxt = sid + 1 if verdict in ("pass", "skip") else sid
|
|
166
186
|
last = max(names) if names else 10
|
|
167
187
|
open_steps = [] if sid >= last and verdict == "pass" else [{
|