task-pipeline-skill 1.88.1 → 1.90.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (31) hide show
  1. package/CHANGELOG.md +109 -0
  2. package/README.md +2 -2
  3. package/SKILL-CARD.md +1 -1
  4. package/package.json +3 -3
  5. package/plugins/task-pipeline/.claude-plugin/plugin.json +1 -1
  6. package/plugins/task-pipeline/agents/verifier-product.md +3 -1
  7. package/plugins/task-pipeline/agents/verifier-visual.md +115 -0
  8. package/plugins/task-pipeline/agents/verifier.md +2 -1
  9. package/plugins/task-pipeline/skills/task-pipeline/SKILL.md +5 -5
  10. package/plugins/task-pipeline/skills/task-pipeline/graph.schema.json +22 -1
  11. package/plugins/task-pipeline/skills/task-pipeline/pipeline.example.json +4 -4
  12. package/plugins/task-pipeline/skills/task-pipeline/references/acceptance.md +12 -0
  13. package/plugins/task-pipeline/skills/task-pipeline/references/audit.md +14 -8
  14. package/plugins/task-pipeline/skills/task-pipeline/references/browser.md +107 -3
  15. package/plugins/task-pipeline/skills/task-pipeline/references/build.md +41 -2
  16. package/plugins/task-pipeline/skills/task-pipeline/references/certification.md +46 -5
  17. package/plugins/task-pipeline/skills/task-pipeline/references/companion-skills.md +29 -0
  18. package/plugins/task-pipeline/skills/task-pipeline/references/conventions.md +3 -3
  19. package/plugins/task-pipeline/skills/task-pipeline/references/doctrine-map.md +1 -1
  20. package/plugins/task-pipeline/skills/task-pipeline/references/grill.md +15 -9
  21. package/plugins/task-pipeline/skills/task-pipeline/references/loop-guard.md +23 -0
  22. package/plugins/task-pipeline/skills/task-pipeline/references/portability.md +1 -1
  23. package/plugins/task-pipeline/skills/task-pipeline/references/spec.md +27 -3
  24. package/plugins/task-pipeline/skills/task-pipeline/references/stages.md +83 -13
  25. package/plugins/task-pipeline/skills/task-pipeline/references/work-graph.md +2 -2
  26. package/plugins/task-pipeline/skills/task-pipeline/scripts/graph.py +45 -16
  27. package/plugins/task-pipeline/skills/task-pipeline/scripts/stage_checkpoint.py +21 -1
  28. package/plugins/task-pipeline/skills/task-pipeline/scripts/visual_gate.py +728 -0
  29. package/plugins/task-pipeline/skills/task-pipeline/templates/brief.md +5 -2
  30. package/plugins/task-pipeline/skills/task-pipeline/templates/browser-claims.json +223 -1
  31. package/plugins/task-pipeline/skills/task-pipeline/templates/run.md +10 -0
@@ -187,10 +187,28 @@ never that the work was skipped quietly.
187
187
  - **UI early-detect:** one branch of the grill is always "does this touch a
188
188
  user-facing surface (web/mobile/CLI/TUI)?". If yes → surface **super-ux**
189
189
  now (use it if installed; otherwise give the install line — see SKILL.md
190
- *Prerequisites*); this arms the stage-3 UX track.
190
+ *Prerequisites*); this arms the stage-3 UX track. **And the same branch records
191
+ the surface's class** — the next bullet.
192
+ - **The surface class.** Every user-facing task's brief carries
193
+ `surface_class: flagship | product | internal | ad`, and the class selects the gate
194
+ profile the visual layer is held to for the rest of the run:
195
+
196
+ | Class | What it is | Director record (stage 3) | Visual half (stages 5–6) | Stage 10 |
197
+ |---|---|---|---|---|
198
+ | `flagship` | the surface a product is judged by — landing, onboarding, paywall, a hero screen | the full record | **gate**: full matrix, pairwise across every axis, the full rubric | approved contact sheet |
199
+ | `product` | an ordinary screen of the product | the short record — Brief, Mode, References, Markers, Open | **gate**: every state and the mandatory pairs; the gate items of the rubric | approved contact sheet |
200
+ | `internal` | an admin panel, an internal tool, a CLI | none owed | recommended; the project linter is the floor | the functional look |
201
+ | `ad` | a creative that runs as an advertisement or a store asset | Brief, Mode, References, Markers, ADA (its rubric profile and safe zones), Open | **gate**: the ad profile and the safe zones | approved contact sheet |
202
+
203
+ Ask it as one question with a recommended answer read off the request — a landing or a
204
+ paywall is `flagship` unless the operator says otherwise — and never leave it to stage
205
+ 5: a class decided at build time is decided by whoever wants the build to pass. A
206
+ work-graph node that builds the surface copies the class as `surface_class`, and
207
+ `graph.py certify` then owes the fourth, `visual` reading on `flagship` and `product`
208
+ ([`certification.md`](certification.md) → *The fourth reading*).
191
209
  - **Artifact:** lock the resolved decisions into a **task brief** committed at
192
- `<artifacts>/specs/YYYY-MM-DD-<topic>-brief.md` (scope, users/UI verdict,
193
- constraints, assumptions, explicitly-deferred items, done-criteria) **plus the
210
+ `<artifacts>/specs/YYYY-MM-DD-<topic>-brief.md` (scope, users/UI verdict and,
211
+ for a user-facing task, `surface_class`, constraints, assumptions, explicitly-deferred items, done-criteria) **plus the
194
212
  autonomy sweep's per-stage answers and the model decision**. Seed it from
195
213
  the skill's `templates/brief.md` skeleton — but only when absent, never
196
214
  overwrite an existing brief. Stages 2–4 build on this brief; stages 5–10 read
@@ -214,7 +232,8 @@ never that the work was skipped quietly.
214
232
  a recorded answer or an explicit deferral, **every answer that contradicted a
215
233
  harvested source has a recorded resolution** (which governs, and whether the doc
216
234
  is now stale), no open contradictions, **every
217
- autonomy-sweep row is answered or explicitly marked "stop and ask here"**, the
235
+ autonomy-sweep row is answered or explicitly marked "stop and ask here"**, **a
236
+ user-facing task's brief names its `surface_class`**, the
218
237
  **REQ table is written and every row names its check**, the carry-over ledger is
219
238
  seeded, **`.task-pipeline/run.md` exists and the header block has been printed**
220
239
  ([`progress.md`](progress.md)), the model decision is recorded, and the operator
@@ -314,8 +333,9 @@ never that the work was skipped quietly.
314
333
  ssheleg/super-ux`). super-ux builds a traced chain — walk it top-down (see its
315
334
  `system-map.md`):
316
335
  0. **Destination first, when Figma is on.** The brief already names the team/org
317
- and the file; `docs/ux/foundation.md` → *Design tooling* is the canonical
318
- record. Confirm it **resolves** before drawing. **Never create a file while a
336
+ and the files — **one per surface** (App, Web, ASO: store screenshots, icon and
337
+ logo), never one file for every frame; `docs/ux/foundation.md` → *Design tooling*
338
+ is the canonical record. Confirm each **resolves** before drawing. **Never create a file while a
319
339
  recorded one resolves; if it doesn't resolve, stop and ask — never create a
320
340
  replacement** (that is the duplicate, and it hides a permissions problem).
321
341
  A creation happens at most once, in the named team, and its URL is written to
@@ -350,7 +370,10 @@ never that the work was skipped quietly.
350
370
  disagree together on one screen. The full doctrine — each track's scope and
351
371
  out-of-scope, the refusal sentences, the four contradictions the check
352
372
  catches — is [`spec.md`](spec.md) → *The COPY and VISUAL tracks, and their
353
- convergence*, its one home.
373
+ convergence*, its one home. **The VISUAL track leaves a trace, not a fact**: a
374
+ director record at `docs/design/<surface>/director-record.md`, whose fields the gate
375
+ reads by the brief's `surface_class` (`python3 scripts/visual_gate.py record <file>
376
+ --class <c>`), and a refusal is that same file saying `Mode: declined` and why.
354
377
  - **Spec:** write the approved design to
355
378
  `<artifacts>/specs/YYYY-MM-DD-<topic>-design.md` and commit it. Lock all
356
379
  shared contracts (types, schemas, signatures, file layout). For UI tasks the
@@ -367,12 +390,23 @@ never that the work was skipped quietly.
367
390
  designed, validated and approved; scenarios validated in `docs/ux/scenarios.md`;
368
391
  the linter passes; every user-facing spec requirement traces to a scenario ID
369
392
  (or an explicit v1-mode/tiny-project waiver by the operator). **With Figma on:
370
- the canonical record names one file, and every `screens.md` frame link carries
371
- that same `:fileKey`** — a string match, not a judgement; a differing key means
372
- the run drew in a second file nobody will open. **Every user-facing string went
393
+ the canonical record names one file per surface (App, Web, ASO), and every
394
+ `screens.md` frame link's `:fileKey` is one of them** — a string match, not a
395
+ judgement (`python3 scripts/visual_gate.py filekeys --record docs/ux/foundation.md
396
+ --screens docs/ux/screens.md`); a key outside the set means the run drew in a file
397
+ nobody recorded and nobody will open. **Every user-facing string went
373
398
  through the COPY track or the refusal is recorded**, and **the visual layer went
374
399
  through the VISUAL track or the refusal is recorded** — a recorded refusal passes
375
400
  this gate and an unmentioned one does not, which is the only difference that matters.
401
+ **And the VISUAL track is checked by its trace, not by the fact that it ran:** on a
402
+ `flagship`, `product` or `ad` surface the director record exists and carries the
403
+ fields its class owes — `python3 scripts/visual_gate.py record
404
+ docs/design/<surface>/director-record.md --class <surface_class>` exits 0. That
405
+ command checks the headings itself and runs the record's own validator from `sheleg-design`
406
+ (`--check-record`) where it is installed and new enough; where it is not, the
407
+ validator reads **NOT_RUN** beside the verdict — never PASS — and the gate stands on
408
+ the floor, said so. A record saying `Mode: declined` with its reason is the refusal,
409
+ and it passes.
376
410
  **Where both tracks ran, their convergence check is recorded** — findings with the
377
411
  ruling, or `Tracks converge: clean`; a screen where each track is right alone and they
378
412
  disagree together is the defect neither track's own review can see.
@@ -435,7 +469,18 @@ never that the work was skipped quietly.
435
469
  requires — never parked silently** — a browser finding filed without a ruling is the
436
470
  diff-review verdict wearing a screenshot; the look was worth taking only if it can
437
471
  still change the code or is on record as deliberately not doing so. Absent, say the surface was verified by reading
438
- the diff and treat it as the weaker claim it is. Stage 6 repeats this over the whole tree; this one catches it while the
472
+ the diff and treat it as the weaker claim it is. **On a surface whose brief names a
473
+ `surface_class`, the project linter runs here too**, after each task that changes a
474
+ rendered surface — `python3 scripts/visual_gate.py lint <dir>`; an S1 finding is
475
+ fixed in the task, and NOT_RUN (exit 3) is recorded as such
476
+ ([`browser.md`](browser.md) → *The visual half*). **With Figma on, the token drift
477
+ probe runs beside it** after a task that touches the token file or a tokenised
478
+ component: `python3 scripts/visual_gate.py tokens --figma <variables.json> --css
479
+ <tokens.css>`. A FAIL is fixed in the task, renaming one side to match the other.
480
+ No export is NOT_RUN, recorded as such. A component whose API the task changed
481
+ updates its Code Connect mapping in the same change
482
+ ([`build.md`](build.md) → *Code Connect, kept*). A slop marker caught while the
483
+ implementer is dispatched costs a line; caught on the contact sheet it costs a round. Stage 6 repeats this over the whole tree; this one catches it while the
439
484
  implementer that wrote it is still dispatched. The matrix pointed this companion at
440
485
  stages 5–6 from the day it was added and **this stage had never named it** — found by
441
486
  the guard comparing the two, not by a reader.
@@ -472,7 +517,10 @@ never that the work was skipped quietly.
472
517
  printed beside their floors; the **full** suite is green (not just the new tests); new/changed code
473
518
  is covered; **every check this run added or widened has been probed both ways —
474
519
  seen rejecting a planted defect and passing the clean tree, asserted on its exit
475
- code** ([`probing.md`](probing.md)); no `skip`/`xfail` smuggling a red suite past the gate. Never advance
520
+ code** ([`probing.md`](probing.md)); no `skip`/`xfail` smuggling a red suite past the gate; **on a
521
+ `flagship`, `product` or `ad` surface where the VISUAL track ran, the visual half's
522
+ `visual_gate.py sheet` exits 0** — NOT_RUN stops and asks, it is not green
523
+ ([`browser.md`](browser.md) → *The visual half*). Never advance
476
524
  to deploy on a red or partial run. **The carry-over count is printed beside this
477
525
  verdict** — a ratchet nobody prints is a TODO with a better name
478
526
  ([`audit.md`](audit.md)) — **and so are the disclosures**, `abstained` and
@@ -502,6 +550,21 @@ never that the work was skipped quietly.
502
550
  run that answers *the surface was checked* by pointing at its spec suite has answered
503
551
  a different question. Where the suite is the thing that changed, the look is what
504
552
  proves it runs against a page that renders.
553
+ - **The visual half is a separate check, and on most user-facing classes a gate.** The
554
+ look above reads the accessibility tree; it cannot say whether the surface looks like
555
+ what was designed. Where the stage-3 VISUAL track ran on a `flagship`, `product` or
556
+ `ad` surface, stage 6 also owes **the contact sheet** — frames over the `SCR-NN` states
557
+ × viewport × theme × text × locale (pairwise, plus the mandatory pairs), each with its
558
+ capture record, diffed against its Figma frame or approved baseline, the project
559
+ linter run, the token drift probe run over the whole token file where Figma is on
560
+ (`visual_gate.py tokens`; NOT_RUN without an export, never PASS), and the rubric
561
+ read by a judge that is not the builder — and its command
562
+ exits 0: `python3 scripts/visual_gate.py sheet <contact-sheet.json> --class
563
+ <surface_class> --artifact-root <frames>` (NOT_RUN, exit 3, is not green). On
564
+ `internal` it is recommended and the linter is the floor. The re-render budget is
565
+ **one, two at most**, then `unresolved` to the person — never round three — and only
566
+ external, specific feedback starts a round. The procedure and the order of the checks:
567
+ [`browser.md`](browser.md) → *The visual half*.
505
568
  - **What the look finds is fixed here.** A rendering defect found at stage 6 is a
506
569
  stage-6 finding: fix it, look again, then call the stage green. Filing it to the
507
570
  board and advancing is how a run reports *checked in a browser* for a page it has
@@ -672,7 +735,14 @@ never that the work was skipped quietly.
672
735
  surface**: read it against the spec section that covers its `SCR-` id and against
673
736
  what shipped. The super-ux linter proves a frame link exists, is named right and
674
737
  is not stale — it cannot read the picture, so a frame promising a limit, a meter
675
- or a tier nobody built passes every lint there is. An absence
738
+ or a tier nobody built passes every lint there is. **Where the VISUAL track ran, the
739
+ walk carries one more row: visual intent ↔ final render** — the director record (the
740
+ brief's falsifier, the signature moment, the rubric written before any render) read
741
+ against the **approved contact sheet**. On a `flagship`, `product` or `ad` surface the
742
+ sheet is approved (`visual_gate.py sheet … --require-approval` exits 0), and the last
743
+ `review:` line per surface — how many human rounds it took — is copied into the
744
+ acceptance file, because `.task-pipeline/run.md` does not outlive the run
745
+ ([`audit.md`](audit.md) → the `V→R` seam). An absence
676
746
  becomes a **new REQ row with its check** and *then* the table is written;
677
747
  appending after the table is how acceptance goes green over a gap. Findings that
678
748
  belong to a lower layer go back to that layer (spec → stage 3, plan → stage 4).
@@ -59,7 +59,7 @@ conditional on the code, never merely sequenced after it.**
59
59
  | `goal` | the release goal | `0` · `3` unstated |
60
60
  | `add` | the id it allocated | `0` · `1` refused |
61
61
  | `park` | the id and the reason | `0` · `1` refused |
62
- | `certify` | the round, and on a failure every `breaks` finding with its fix and its check | `0` all three tiers passed · `1` a tier failed, or a report is malformed |
62
+ | `certify` | the round, and on a failure every `breaks` finding with its fix and its check | `0` every owed tier passed — three, or four with `visual` on a flagship or product node · `1` a tier failed, a required one is missing, or a report is malformed |
63
63
  | `close` | the goal, the new frontier count, and what was not verified | `0` · `1` refused **or the verdict stops the run** |
64
64
  | `producer` | what produced this proof — actor, model, runtime, skill, config, commit, trace | `0` |
65
65
  | `doctrine` | how many of the bundle's reference files this run opened | `0` |
@@ -84,7 +84,7 @@ A **`parked`** node is the single exemption: it is the one node nobody will clos
84
84
  *n/a — parked* in that field is confidence without correctness. `park` never removes what
85
85
  the node said it would run.
86
86
 
87
- **A node is closed by three readings, not one.** `certify` takes one tier report from each of `unit`, `seam` and `product` — dispatched blind and in parallel — requires all three to pass, and assembles the seven-key verdict `close` consumes. `close`'s contract is unchanged; what changed is that the verdict is now built from three readings at different distances instead of written from one, because a change can be correct where it was made and wrong one level out. A failing round records itself and leaves the node open. Doctrine: [`certification.md`](certification.md).
87
+ **A node is closed by three readings, not one.** `certify` takes one tier report from each of `unit`, `seam` and `product` — dispatched blind and in parallel — requires all three to pass, and assembles the seven-key verdict `close` consumes. `close`'s contract is unchanged; what changed is that the verdict is now built from three readings at different distances instead of written from one, because a change can be correct where it was made and wrong one level out. A failing round records itself and leaves the node open. **A node whose `surface_class` is `flagship` or `product` owes a fourth, `visual` report** — the contact sheet read against the director record — and `certify` refuses the round without it; on any other node a `visual` report is accepted when given. Doctrine: [`certification.md`](certification.md).
88
88
 
89
89
  **`close` stamps the commit; the verifier never supplies it.** A verdict written after the
90
90
  tree moved is evidence about a different tree, and an agent cannot name the wrong commit if
@@ -268,6 +268,14 @@ def violations(graph):
268
268
  "two commands cannot say which one closed the node, and the "
269
269
  "verifier reports its output as one evidence row")
270
270
 
271
+ # The surface class (stage 0, `references/stages.md`) decides whether `certify`
272
+ # owes the fourth, `visual` reading. A class outside the four would silently
273
+ # require nothing, which is the one outcome a typo must not have.
274
+ if "surface_class" in n and n["surface_class"] not in SURFACE_CLASSES:
275
+ out.append(f"{nid}: surface_class is {n['surface_class']!r} — it must be one of "
276
+ f"{', '.join(SURFACE_CLASSES)}. A class nobody recognises requires no "
277
+ "visual reading, so a typo here would drop the tier that reads the pixels")
278
+
271
279
  if n.get("status") == "done":
272
280
  ev = n.get("evidence")
273
281
  if not isinstance(ev, list) or not [e for e in ev
@@ -574,6 +582,9 @@ def verdict_violations(v):
574
582
  # tests of the neighbours the change can reach
575
583
  # product one level out again — the documentation, the scenarios, how this
576
584
  # behaviour interacts with the rest of the product
585
+ # visual the pixels — the contact sheet, the director record, the project
586
+ # linter and the rubric. Owed only where the node's `surface_class` is
587
+ # flagship or product; accepted, and counted, wherever it is given
577
588
  #
578
589
  # **All three must pass, and blind is the point.** Three agents that read each
579
590
  # other's reports are one opinion with three signatures; the disagreement is the
@@ -584,6 +595,15 @@ def verdict_violations(v):
584
595
  # stamp at three levels is worse than one verifier, because it costs three times
585
596
  # as much and reads as three times the assurance.
586
597
  TIERS = ("unit", "seam", "product")
598
+ # The fourth reading, `visual`, reads the PIXELS — the contact sheet, the director
599
+ # record, the project linter's output and the rubric — which none of the three opens.
600
+ # It is owed by a node whose `surface_class` is one of VISUAL_REQUIRED, accepted when
601
+ # given on any other node, and blind to the other three exactly as they are to each
602
+ # other. `references/certification.md` → *The fourth reading*.
603
+ VISUAL_TIER = "visual"
604
+ ALL_TIERS = TIERS + (VISUAL_TIER,)
605
+ SURFACE_CLASSES = ("flagship", "product", "internal", "ad")
606
+ VISUAL_REQUIRED = ("flagship", "product")
587
607
  TIER_KEYS = ("node", "tier", "verdict", "scope", "confirms", "findings",
588
608
  "evidence", "not_examined")
589
609
  TIER_VERDICTS = ("pass", "fail")
@@ -596,10 +616,10 @@ SEVERITIES = ("breaks", "risk")
596
616
  # "the unit tier's verdict…"). Widened to the tense and possessive forms the
597
617
  # reader planted; still a closed list on purpose — a looser net here starts
598
618
  # matching a report's honest prose about its OWN tier.
599
- CROSS_TIER = re.compile(r"\b(?:unit|seam|product)\s+tier(?:'s)?\s+"
619
+ CROSS_TIER = re.compile(r"\b(?:unit|seam|product|visual)\s+tier(?:'s)?\s+"
600
620
  r"(?:passed|failed|says|said|confirm\w*|verdict|report)"
601
621
  r"|\btier\s+\d\s+(?:passed|failed|says|said|confirm\w*)"
602
- r"|as\s+the\s+(?:unit|seam|product)\s+tier", re.I)
622
+ r"|as\s+the\s+(?:unit|seam|product|visual)\s+tier", re.I)
603
623
 
604
624
 
605
625
  def tier_violations(t):
@@ -622,9 +642,9 @@ def tier_violations(t):
622
642
 
623
643
  if not isinstance(t["node"], str) or not t["node"].startswith(NODE_ID):
624
644
  out.append("tier report `node` is %r, which is not a node id" % (t["node"],))
625
- if t["tier"] not in TIERS:
645
+ if t["tier"] not in ALL_TIERS:
626
646
  out.append("tier report `tier` is %r — it must be one of %s"
627
- % (t["tier"], ", ".join(TIERS)))
647
+ % (t["tier"], ", ".join(ALL_TIERS)))
628
648
  if t["verdict"] not in TIER_VERDICTS:
629
649
  out.append("tier report `verdict` is %r — it must be `pass` or `fail`, because "
630
650
  "a certification that admits a third state admits a maybe"
@@ -1482,6 +1502,14 @@ def cmd_certify(graph, args):
1482
1502
  die("certification is missing the %s report(s) — all three are required, because "
1483
1503
  "the level nobody read is the level the defect survives at"
1484
1504
  % ", ".join("`%s`" % m for m in missing))
1505
+ sclass = node.get("surface_class")
1506
+ if sclass in VISUAL_REQUIRED and VISUAL_TIER not in reports:
1507
+ die("certification is missing the `%s` report — %s is a %s surface, and on one the "
1508
+ "pixels are part of the requirement: none of unit, seam or product opens the "
1509
+ "contact sheet, so without the fourth reading nobody looked at what a user sees "
1510
+ "(references/certification.md → *The fourth reading*)" % (VISUAL_TIER, nid, sclass))
1511
+ # The tiers THIS round read, in a stable order: the three always, `visual` when given.
1512
+ tiers = [x for x in ALL_TIERS if x in reports]
1485
1513
 
1486
1514
  # The stamp, read here and never accepted from a report — same law as `close`.
1487
1515
  import subprocess
@@ -1493,7 +1521,7 @@ def cmd_certify(graph, args):
1493
1521
 
1494
1522
  prior = node.get("certification") or {}
1495
1523
  round_no = int(prior.get("round") or 0) + 1
1496
- tiers_now = {x: reports[x]["verdict"] for x in TIERS}
1524
+ tiers_now = {x: reports[x]["verdict"] for x in tiers}
1497
1525
  history = list(prior.get("history") or []) + [tiers_now]
1498
1526
  node["certification"] = {
1499
1527
  "round": round_no,
@@ -1502,11 +1530,11 @@ def cmd_certify(graph, args):
1502
1530
  "history": history,
1503
1531
  }
1504
1532
 
1505
- failed = [x for x in TIERS if tiers_now[x] == "fail"]
1533
+ failed = [x for x in tiers if tiers_now[x] == "fail"]
1506
1534
 
1507
1535
  # Churn, measured. A tier that has failed in every round so far is the one the
1508
1536
  # operator needs named; counting it here is what makes the loop visible.
1509
- churning = [x for x in TIERS
1537
+ churning = [x for x in tiers
1510
1538
  if len(history) >= 2 and all(h.get(x) == "fail" for h in history)]
1511
1539
 
1512
1540
  save(args.graph, graph)
@@ -1545,19 +1573,19 @@ def cmd_certify(graph, args):
1545
1573
  # exactly what `can_continue_around: true` says)
1546
1574
  verdict = {
1547
1575
  "node": nid,
1548
- "done": [c for x in TIERS for c in reports[x]["confirms"]],
1576
+ "done": [c for x in tiers for c in reports[x]["confirms"]],
1549
1577
  "not_done": [],
1550
1578
  "not_verified": ["%s: %s" % (x, n)
1551
- for x in TIERS for n in reports[x]["not_examined"]],
1579
+ for x in tiers for n in reports[x]["not_examined"]],
1552
1580
  "blockers": [
1553
1581
  {"what": "%s (%s, found by the `%s` tier)" % (f["what"], f["where"], x),
1554
1582
  "blocks": [], "can_continue_around": True}
1555
- for x in TIERS for f in reports[x]["findings"]
1583
+ for x in tiers for f in reports[x]["findings"]
1556
1584
  if f.get("severity") == "risk"
1557
1585
  ],
1558
1586
  "replan": {"possible": True, "add": [], "park": [],
1559
- "why": "certified at all three tiers in round %d" % round_no},
1560
- "evidence": ["%s: %s" % (x, e) for x in TIERS for e in reports[x]["evidence"]],
1587
+ "why": "certified at all %d tiers in round %d" % (len(tiers), round_no)},
1588
+ "evidence": ["%s: %s" % (x, e) for x in tiers for e in reports[x]["evidence"]],
1561
1589
  }
1562
1590
  # Proof identity (FIX-PF-02.01): the certification tested THIS tree, so it
1563
1591
  # records the commit it tested into the verdict it hands `close`. Without it
@@ -1579,7 +1607,7 @@ def cmd_certify(graph, args):
1579
1607
  # cannot hand the run a verdict its own consumer refuses.
1580
1608
  broken = verdict_violations(verdict)
1581
1609
  if broken:
1582
- die("all three tiers passed and the assembled verdict is still malformed — this "
1610
+ die("every tier passed and the assembled verdict is still malformed — this "
1583
1611
  "is a defect in `certify`, not in the reports:\n " + "\n ".join(broken))
1584
1612
 
1585
1613
  out = args.verdict_out or os.path.join(os.path.dirname(args.graph) or ".",
@@ -1589,7 +1617,7 @@ def cmd_certify(graph, args):
1589
1617
  json.dump(verdict, fh, indent=2, ensure_ascii=False)
1590
1618
  fh.write("\n")
1591
1619
  os.replace(tmp, out)
1592
- print("%s: certified at unit, seam and product in round %d" % (nid, round_no))
1620
+ print("%s: certified at %s in round %d" % (nid, ", ".join(tiers), round_no))
1593
1621
  print("verdict written to %s — close it with:" % out)
1594
1622
  print(" graph.py close --verdict %s" % out)
1595
1623
  return 0
@@ -1797,8 +1825,9 @@ VERBS = {
1797
1825
  "coverage": (cmd_coverage, "every requirement and the nodes serving it; exits 1 on a gap"),
1798
1826
  "add": (cmd_add, "add a node mid-run"),
1799
1827
  "park": (cmd_park, "park a node, carrying the reason"),
1800
- "certify": (cmd_certify, "require three independent tier reports, then emit "
1801
- "the verdict `close` consumes"),
1828
+ "certify": (cmd_certify, "require three independent tier reports (four on a "
1829
+ "flagship or product surface), then emit the verdict "
1830
+ "`close` consumes"),
1802
1831
  "close": (cmd_close, "consume a verdict, close one node and re-plan"),
1803
1832
  }
1804
1833
 
@@ -37,6 +37,9 @@ What it keeps, and where:
37
37
  the episode Observatory kept (`keptAs`), the token is dropped, and the run is told to read
38
38
  the workflow before writing again. A stale writer therefore stops after one refusal.
39
39
 
40
+ Ledger lines it reads besides `stage:`: `review:` (the contact sheet's human rounds, stage 6
41
+ and 10 on a visual surface), each carried as evidence on the stage it names.
42
+
40
43
  Ledger lines it appends, all of the existing `event:` shape:
41
44
 
42
45
  event: memory — checkpoint <stepId> rev <n> <workflowId> — <ISO-8601>
@@ -63,6 +66,10 @@ STATE_NAME = "memory.json"
63
66
  OWNER = "agent:task-pipeline"
64
67
  STAGE = re.compile(r"^stage:\s*(\d+)\s+(.+?)\s+—\s+gate\s+(\S+)\s+—\s+verdict\s+(\S+)\s+—\s+(\S+)\s*$")
65
68
  TOPIC = re.compile(r"^Run:\s*`([^`]+)`")
69
+ # The contact sheet's human rounds (`references/browser.md` → *The visual half*): one line per
70
+ # return or approval. Carried into the checkpoint of the stage it names, so the number of
71
+ # passes a surface took is measured at the boundary rather than remembered at the end.
72
+ REVIEW = re.compile(r"^review:\s*(\d+)\s+—\s+surface\s+(.+?)\s+—\s+rounds\s+(\d+)\s+—\s+(\S+)\s+—\s+\S+\s*$")
66
73
  STATUS = {"pass": "done", "skip": "done", "fail": "blocked"}
67
74
 
68
75
 
@@ -81,6 +88,17 @@ def _now() -> str:
81
88
  return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
82
89
 
83
90
 
91
+ def _reviews(path: pathlib.Path) -> dict[int, list[str]]:
92
+ """`review:` lines by the stage id they name, as evidence strings."""
93
+ out: dict[int, list[str]] = {}
94
+ for line in path.read_text(encoding="utf-8").splitlines():
95
+ m = REVIEW.match(line)
96
+ if m:
97
+ out.setdefault(int(m.group(1)), []).append(
98
+ f"review: {m.group(2)} rounds {m.group(3)} {m.group(4)}")
99
+ return out
100
+
101
+
84
102
  def _read_ledger(path: pathlib.Path) -> tuple[str, list[tuple]]:
85
103
  try:
86
104
  text = path.read_text(encoding="utf-8")
@@ -160,8 +178,10 @@ def emit(ledger: pathlib.Path, project: str | None, constraints: list[str],
160
178
  # one thing a successor must never act without (found by the live receipt, 2026-10-05).
161
179
  constraints = list(dict.fromkeys([*state.get("constraints", []), *constraints]))
162
180
  credentials = list(dict.fromkeys([*state.get("credentials", []), *credentials]))
181
+ reviews = _reviews(ledger)
163
182
  done = [{"step_id": f"stage-{s[0]}", "result": f"{s[1]}: gate {s[2]}, verdict {s[3]} at {s[4]}",
164
- "evidence": [f"ledger: {ledger.name}"]} for s in stages if s[3] in ("pass", "skip")]
183
+ "evidence": [f"ledger: {ledger.name}", *reviews.get(s[0], [])]}
184
+ for s in stages if s[3] in ("pass", "skip")]
165
185
  nxt = sid + 1 if verdict in ("pass", "skip") else sid
166
186
  last = max(names) if names else 10
167
187
  open_steps = [] if sid >= last and verdict == "pass" else [{