task-pipeline-skill 1.88.1 → 1.89.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. package/CHANGELOG.md +64 -0
  2. package/README.md +1 -1
  3. package/SKILL-CARD.md +1 -1
  4. package/package.json +3 -3
  5. package/plugins/task-pipeline/.claude-plugin/plugin.json +1 -1
  6. package/plugins/task-pipeline/agents/verifier-product.md +3 -1
  7. package/plugins/task-pipeline/agents/verifier-visual.md +115 -0
  8. package/plugins/task-pipeline/agents/verifier.md +2 -1
  9. package/plugins/task-pipeline/skills/task-pipeline/SKILL.md +4 -4
  10. package/plugins/task-pipeline/skills/task-pipeline/graph.schema.json +22 -1
  11. package/plugins/task-pipeline/skills/task-pipeline/pipeline.example.json +4 -4
  12. package/plugins/task-pipeline/skills/task-pipeline/references/acceptance.md +12 -0
  13. package/plugins/task-pipeline/skills/task-pipeline/references/audit.md +14 -8
  14. package/plugins/task-pipeline/skills/task-pipeline/references/browser.md +97 -3
  15. package/plugins/task-pipeline/skills/task-pipeline/references/certification.md +46 -5
  16. package/plugins/task-pipeline/skills/task-pipeline/references/companion-skills.md +28 -0
  17. package/plugins/task-pipeline/skills/task-pipeline/references/conventions.md +3 -3
  18. package/plugins/task-pipeline/skills/task-pipeline/references/doctrine-map.md +1 -1
  19. package/plugins/task-pipeline/skills/task-pipeline/references/grill.md +15 -9
  20. package/plugins/task-pipeline/skills/task-pipeline/references/loop-guard.md +23 -0
  21. package/plugins/task-pipeline/skills/task-pipeline/references/portability.md +1 -1
  22. package/plugins/task-pipeline/skills/task-pipeline/references/spec.md +27 -3
  23. package/plugins/task-pipeline/skills/task-pipeline/references/stages.md +75 -13
  24. package/plugins/task-pipeline/skills/task-pipeline/references/work-graph.md +2 -2
  25. package/plugins/task-pipeline/skills/task-pipeline/scripts/graph.py +45 -16
  26. package/plugins/task-pipeline/skills/task-pipeline/scripts/stage_checkpoint.py +21 -1
  27. package/plugins/task-pipeline/skills/task-pipeline/scripts/visual_gate.py +589 -0
  28. package/plugins/task-pipeline/skills/task-pipeline/templates/brief.md +5 -2
  29. package/plugins/task-pipeline/skills/task-pipeline/templates/browser-claims.json +223 -1
  30. package/plugins/task-pipeline/skills/task-pipeline/templates/run.md +10 -0
@@ -268,6 +268,14 @@ def violations(graph):
268
268
  "two commands cannot say which one closed the node, and the "
269
269
  "verifier reports its output as one evidence row")
270
270
 
271
+ # The surface class (stage 0, `references/stages.md`) decides whether `certify`
272
+ # owes the fourth, `visual` reading. A class outside the four would silently
273
+ # require nothing, which is the one outcome a typo must not have.
274
+ if "surface_class" in n and n["surface_class"] not in SURFACE_CLASSES:
275
+ out.append(f"{nid}: surface_class is {n['surface_class']!r} — it must be one of "
276
+ f"{', '.join(SURFACE_CLASSES)}. A class nobody recognises requires no "
277
+ "visual reading, so a typo here would drop the tier that reads the pixels")
278
+
271
279
  if n.get("status") == "done":
272
280
  ev = n.get("evidence")
273
281
  if not isinstance(ev, list) or not [e for e in ev
@@ -574,6 +582,9 @@ def verdict_violations(v):
574
582
  # tests of the neighbours the change can reach
575
583
  # product one level out again — the documentation, the scenarios, how this
576
584
  # behaviour interacts with the rest of the product
585
+ # visual the pixels — the contact sheet, the director record, the project
586
+ # linter and the rubric. Owed only where the node's `surface_class` is
587
+ # flagship or product; accepted, and counted, wherever it is given
577
588
  #
578
589
  # **All three must pass, and blind is the point.** Three agents that read each
579
590
  # other's reports are one opinion with three signatures; the disagreement is the
@@ -584,6 +595,15 @@ def verdict_violations(v):
584
595
  # stamp at three levels is worse than one verifier, because it costs three times
585
596
  # as much and reads as three times the assurance.
586
597
  TIERS = ("unit", "seam", "product")
598
+ # The fourth reading, `visual`, reads the PIXELS — the contact sheet, the director
599
+ # record, the project linter's output and the rubric — which none of the three opens.
600
+ # It is owed by a node whose `surface_class` is one of VISUAL_REQUIRED, accepted when
601
+ # given on any other node, and blind to the other three exactly as they are to each
602
+ # other. `references/certification.md` → *The fourth reading*.
603
+ VISUAL_TIER = "visual"
604
+ ALL_TIERS = TIERS + (VISUAL_TIER,)
605
+ SURFACE_CLASSES = ("flagship", "product", "internal", "ad")
606
+ VISUAL_REQUIRED = ("flagship", "product")
587
607
  TIER_KEYS = ("node", "tier", "verdict", "scope", "confirms", "findings",
588
608
  "evidence", "not_examined")
589
609
  TIER_VERDICTS = ("pass", "fail")
@@ -596,10 +616,10 @@ SEVERITIES = ("breaks", "risk")
596
616
  # "the unit tier's verdict…"). Widened to the tense and possessive forms the
597
617
  # reader planted; still a closed list on purpose — a looser net here starts
598
618
  # matching a report's honest prose about its OWN tier.
599
- CROSS_TIER = re.compile(r"\b(?:unit|seam|product)\s+tier(?:'s)?\s+"
619
+ CROSS_TIER = re.compile(r"\b(?:unit|seam|product|visual)\s+tier(?:'s)?\s+"
600
620
  r"(?:passed|failed|says|said|confirm\w*|verdict|report)"
601
621
  r"|\btier\s+\d\s+(?:passed|failed|says|said|confirm\w*)"
602
- r"|as\s+the\s+(?:unit|seam|product)\s+tier", re.I)
622
+ r"|as\s+the\s+(?:unit|seam|product|visual)\s+tier", re.I)
603
623
 
604
624
 
605
625
  def tier_violations(t):
@@ -622,9 +642,9 @@ def tier_violations(t):
622
642
 
623
643
  if not isinstance(t["node"], str) or not t["node"].startswith(NODE_ID):
624
644
  out.append("tier report `node` is %r, which is not a node id" % (t["node"],))
625
- if t["tier"] not in TIERS:
645
+ if t["tier"] not in ALL_TIERS:
626
646
  out.append("tier report `tier` is %r — it must be one of %s"
627
- % (t["tier"], ", ".join(TIERS)))
647
+ % (t["tier"], ", ".join(ALL_TIERS)))
628
648
  if t["verdict"] not in TIER_VERDICTS:
629
649
  out.append("tier report `verdict` is %r — it must be `pass` or `fail`, because "
630
650
  "a certification that admits a third state admits a maybe"
@@ -1482,6 +1502,14 @@ def cmd_certify(graph, args):
1482
1502
  die("certification is missing the %s report(s) — all three are required, because "
1483
1503
  "the level nobody read is the level the defect survives at"
1484
1504
  % ", ".join("`%s`" % m for m in missing))
1505
+ sclass = node.get("surface_class")
1506
+ if sclass in VISUAL_REQUIRED and VISUAL_TIER not in reports:
1507
+ die("certification is missing the `%s` report — %s is a %s surface, and on one the "
1508
+ "pixels are part of the requirement: none of unit, seam or product opens the "
1509
+ "contact sheet, so without the fourth reading nobody looked at what a user sees "
1510
+ "(references/certification.md → *The fourth reading*)" % (VISUAL_TIER, nid, sclass))
1511
+ # The tiers THIS round read, in a stable order: the three always, `visual` when given.
1512
+ tiers = [x for x in ALL_TIERS if x in reports]
1485
1513
 
1486
1514
  # The stamp, read here and never accepted from a report — same law as `close`.
1487
1515
  import subprocess
@@ -1493,7 +1521,7 @@ def cmd_certify(graph, args):
1493
1521
 
1494
1522
  prior = node.get("certification") or {}
1495
1523
  round_no = int(prior.get("round") or 0) + 1
1496
- tiers_now = {x: reports[x]["verdict"] for x in TIERS}
1524
+ tiers_now = {x: reports[x]["verdict"] for x in tiers}
1497
1525
  history = list(prior.get("history") or []) + [tiers_now]
1498
1526
  node["certification"] = {
1499
1527
  "round": round_no,
@@ -1502,11 +1530,11 @@ def cmd_certify(graph, args):
1502
1530
  "history": history,
1503
1531
  }
1504
1532
 
1505
- failed = [x for x in TIERS if tiers_now[x] == "fail"]
1533
+ failed = [x for x in tiers if tiers_now[x] == "fail"]
1506
1534
 
1507
1535
  # Churn, measured. A tier that has failed in every round so far is the one the
1508
1536
  # operator needs named; counting it here is what makes the loop visible.
1509
- churning = [x for x in TIERS
1537
+ churning = [x for x in tiers
1510
1538
  if len(history) >= 2 and all(h.get(x) == "fail" for h in history)]
1511
1539
 
1512
1540
  save(args.graph, graph)
@@ -1545,19 +1573,19 @@ def cmd_certify(graph, args):
1545
1573
  # exactly what `can_continue_around: true` says)
1546
1574
  verdict = {
1547
1575
  "node": nid,
1548
- "done": [c for x in TIERS for c in reports[x]["confirms"]],
1576
+ "done": [c for x in tiers for c in reports[x]["confirms"]],
1549
1577
  "not_done": [],
1550
1578
  "not_verified": ["%s: %s" % (x, n)
1551
- for x in TIERS for n in reports[x]["not_examined"]],
1579
+ for x in tiers for n in reports[x]["not_examined"]],
1552
1580
  "blockers": [
1553
1581
  {"what": "%s (%s, found by the `%s` tier)" % (f["what"], f["where"], x),
1554
1582
  "blocks": [], "can_continue_around": True}
1555
- for x in TIERS for f in reports[x]["findings"]
1583
+ for x in tiers for f in reports[x]["findings"]
1556
1584
  if f.get("severity") == "risk"
1557
1585
  ],
1558
1586
  "replan": {"possible": True, "add": [], "park": [],
1559
- "why": "certified at all three tiers in round %d" % round_no},
1560
- "evidence": ["%s: %s" % (x, e) for x in TIERS for e in reports[x]["evidence"]],
1587
+ "why": "certified at all %d tiers in round %d" % (len(tiers), round_no)},
1588
+ "evidence": ["%s: %s" % (x, e) for x in tiers for e in reports[x]["evidence"]],
1561
1589
  }
1562
1590
  # Proof identity (FIX-PF-02.01): the certification tested THIS tree, so it
1563
1591
  # records the commit it tested into the verdict it hands `close`. Without it
@@ -1579,7 +1607,7 @@ def cmd_certify(graph, args):
1579
1607
  # cannot hand the run a verdict its own consumer refuses.
1580
1608
  broken = verdict_violations(verdict)
1581
1609
  if broken:
1582
- die("all three tiers passed and the assembled verdict is still malformed — this "
1610
+ die("every tier passed and the assembled verdict is still malformed — this "
1583
1611
  "is a defect in `certify`, not in the reports:\n " + "\n ".join(broken))
1584
1612
 
1585
1613
  out = args.verdict_out or os.path.join(os.path.dirname(args.graph) or ".",
@@ -1589,7 +1617,7 @@ def cmd_certify(graph, args):
1589
1617
  json.dump(verdict, fh, indent=2, ensure_ascii=False)
1590
1618
  fh.write("\n")
1591
1619
  os.replace(tmp, out)
1592
- print("%s: certified at unit, seam and product in round %d" % (nid, round_no))
1620
+ print("%s: certified at %s in round %d" % (nid, ", ".join(tiers), round_no))
1593
1621
  print("verdict written to %s — close it with:" % out)
1594
1622
  print(" graph.py close --verdict %s" % out)
1595
1623
  return 0
@@ -1797,8 +1825,9 @@ VERBS = {
1797
1825
  "coverage": (cmd_coverage, "every requirement and the nodes serving it; exits 1 on a gap"),
1798
1826
  "add": (cmd_add, "add a node mid-run"),
1799
1827
  "park": (cmd_park, "park a node, carrying the reason"),
1800
- "certify": (cmd_certify, "require three independent tier reports, then emit "
1801
- "the verdict `close` consumes"),
1828
+ "certify": (cmd_certify, "require three independent tier reports (four on a "
1829
+ "flagship or product surface), then emit the verdict "
1830
+ "`close` consumes"),
1802
1831
  "close": (cmd_close, "consume a verdict, close one node and re-plan"),
1803
1832
  }
1804
1833
 
@@ -37,6 +37,9 @@ What it keeps, and where:
37
37
  the episode Observatory kept (`keptAs`), the token is dropped, and the run is told to read
38
38
  the workflow before writing again. A stale writer therefore stops after one refusal.
39
39
 
40
+ Ledger lines it reads besides `stage:`: `review:` (the contact sheet's human rounds, stage 6
41
+ and 10 on a visual surface), each carried as evidence on the stage it names.
42
+
40
43
  Ledger lines it appends, all of the existing `event:` shape:
41
44
 
42
45
  event: memory — checkpoint <stepId> rev <n> <workflowId> — <ISO-8601>
@@ -63,6 +66,10 @@ STATE_NAME = "memory.json"
63
66
  OWNER = "agent:task-pipeline"
64
67
  STAGE = re.compile(r"^stage:\s*(\d+)\s+(.+?)\s+—\s+gate\s+(\S+)\s+—\s+verdict\s+(\S+)\s+—\s+(\S+)\s*$")
65
68
  TOPIC = re.compile(r"^Run:\s*`([^`]+)`")
69
+ # The contact sheet's human rounds (`references/browser.md` → *The visual half*): one line per
70
+ # return or approval. Carried into the checkpoint of the stage it names, so the number of
71
+ # passes a surface took is measured at the boundary rather than remembered at the end.
72
+ REVIEW = re.compile(r"^review:\s*(\d+)\s+—\s+surface\s+(.+?)\s+—\s+rounds\s+(\d+)\s+—\s+(\S+)\s+—\s+\S+\s*$")
66
73
  STATUS = {"pass": "done", "skip": "done", "fail": "blocked"}
67
74
 
68
75
 
@@ -81,6 +88,17 @@ def _now() -> str:
81
88
  return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
82
89
 
83
90
 
91
+ def _reviews(path: pathlib.Path) -> dict[int, list[str]]:
92
+ """`review:` lines by the stage id they name, as evidence strings."""
93
+ out: dict[int, list[str]] = {}
94
+ for line in path.read_text(encoding="utf-8").splitlines():
95
+ m = REVIEW.match(line)
96
+ if m:
97
+ out.setdefault(int(m.group(1)), []).append(
98
+ f"review: {m.group(2)} rounds {m.group(3)} {m.group(4)}")
99
+ return out
100
+
101
+
84
102
  def _read_ledger(path: pathlib.Path) -> tuple[str, list[tuple]]:
85
103
  try:
86
104
  text = path.read_text(encoding="utf-8")
@@ -160,8 +178,10 @@ def emit(ledger: pathlib.Path, project: str | None, constraints: list[str],
160
178
  # one thing a successor must never act without (found by the live receipt, 2026-10-05).
161
179
  constraints = list(dict.fromkeys([*state.get("constraints", []), *constraints]))
162
180
  credentials = list(dict.fromkeys([*state.get("credentials", []), *credentials]))
181
+ reviews = _reviews(ledger)
163
182
  done = [{"step_id": f"stage-{s[0]}", "result": f"{s[1]}: gate {s[2]}, verdict {s[3]} at {s[4]}",
164
- "evidence": [f"ledger: {ledger.name}"]} for s in stages if s[3] in ("pass", "skip")]
183
+ "evidence": [f"ledger: {ledger.name}", *reviews.get(s[0], [])]}
184
+ for s in stages if s[3] in ("pass", "skip")]
165
185
  nxt = sid + 1 if verdict in ("pass", "skip") else sid
166
186
  last = max(names) if names else 10
167
187
  open_steps = [] if sid >= last and verdict == "pass" else [{