task-pipeline-skill 1.88.1 → 1.89.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +64 -0
- package/README.md +1 -1
- package/SKILL-CARD.md +1 -1
- package/package.json +3 -3
- package/plugins/task-pipeline/.claude-plugin/plugin.json +1 -1
- package/plugins/task-pipeline/agents/verifier-product.md +3 -1
- package/plugins/task-pipeline/agents/verifier-visual.md +115 -0
- package/plugins/task-pipeline/agents/verifier.md +2 -1
- package/plugins/task-pipeline/skills/task-pipeline/SKILL.md +4 -4
- package/plugins/task-pipeline/skills/task-pipeline/graph.schema.json +22 -1
- package/plugins/task-pipeline/skills/task-pipeline/pipeline.example.json +4 -4
- package/plugins/task-pipeline/skills/task-pipeline/references/acceptance.md +12 -0
- package/plugins/task-pipeline/skills/task-pipeline/references/audit.md +14 -8
- package/plugins/task-pipeline/skills/task-pipeline/references/browser.md +97 -3
- package/plugins/task-pipeline/skills/task-pipeline/references/certification.md +46 -5
- package/plugins/task-pipeline/skills/task-pipeline/references/companion-skills.md +28 -0
- package/plugins/task-pipeline/skills/task-pipeline/references/conventions.md +3 -3
- package/plugins/task-pipeline/skills/task-pipeline/references/doctrine-map.md +1 -1
- package/plugins/task-pipeline/skills/task-pipeline/references/grill.md +15 -9
- package/plugins/task-pipeline/skills/task-pipeline/references/loop-guard.md +23 -0
- package/plugins/task-pipeline/skills/task-pipeline/references/portability.md +1 -1
- package/plugins/task-pipeline/skills/task-pipeline/references/spec.md +27 -3
- package/plugins/task-pipeline/skills/task-pipeline/references/stages.md +75 -13
- package/plugins/task-pipeline/skills/task-pipeline/references/work-graph.md +2 -2
- package/plugins/task-pipeline/skills/task-pipeline/scripts/graph.py +45 -16
- package/plugins/task-pipeline/skills/task-pipeline/scripts/stage_checkpoint.py +21 -1
- package/plugins/task-pipeline/skills/task-pipeline/scripts/visual_gate.py +589 -0
- package/plugins/task-pipeline/skills/task-pipeline/templates/brief.md +5 -2
- package/plugins/task-pipeline/skills/task-pipeline/templates/browser-claims.json +223 -1
- package/plugins/task-pipeline/skills/task-pipeline/templates/run.md +10 -0
|
@@ -268,6 +268,14 @@ def violations(graph):
|
|
|
268
268
|
"two commands cannot say which one closed the node, and the "
|
|
269
269
|
"verifier reports its output as one evidence row")
|
|
270
270
|
|
|
271
|
+
# The surface class (stage 0, `references/stages.md`) decides whether `certify`
|
|
272
|
+
# owes the fourth, `visual` reading. A class outside the four would silently
|
|
273
|
+
# require nothing, which is the one outcome a typo must not have.
|
|
274
|
+
if "surface_class" in n and n["surface_class"] not in SURFACE_CLASSES:
|
|
275
|
+
out.append(f"{nid}: surface_class is {n['surface_class']!r} — it must be one of "
|
|
276
|
+
f"{', '.join(SURFACE_CLASSES)}. A class nobody recognises requires no "
|
|
277
|
+
"visual reading, so a typo here would drop the tier that reads the pixels")
|
|
278
|
+
|
|
271
279
|
if n.get("status") == "done":
|
|
272
280
|
ev = n.get("evidence")
|
|
273
281
|
if not isinstance(ev, list) or not [e for e in ev
|
|
@@ -574,6 +582,9 @@ def verdict_violations(v):
|
|
|
574
582
|
# tests of the neighbours the change can reach
|
|
575
583
|
# product one level out again — the documentation, the scenarios, how this
|
|
576
584
|
# behaviour interacts with the rest of the product
|
|
585
|
+
# visual the pixels — the contact sheet, the director record, the project
|
|
586
|
+
# linter and the rubric. Owed only where the node's `surface_class` is
|
|
587
|
+
# flagship or product; accepted, and counted, wherever it is given
|
|
577
588
|
#
|
|
578
589
|
# **All three must pass, and blind is the point.** Three agents that read each
|
|
579
590
|
# other's reports are one opinion with three signatures; the disagreement is the
|
|
@@ -584,6 +595,15 @@ def verdict_violations(v):
|
|
|
584
595
|
# stamp at three levels is worse than one verifier, because it costs three times
|
|
585
596
|
# as much and reads as three times the assurance.
|
|
586
597
|
TIERS = ("unit", "seam", "product")
|
|
598
|
+
# The fourth reading, `visual`, reads the PIXELS — the contact sheet, the director
|
|
599
|
+
# record, the project linter's output and the rubric — which none of the three opens.
|
|
600
|
+
# It is owed by a node whose `surface_class` is one of VISUAL_REQUIRED, accepted when
|
|
601
|
+
# given on any other node, and blind to the other three exactly as they are to each
|
|
602
|
+
# other. `references/certification.md` → *The fourth reading*.
|
|
603
|
+
VISUAL_TIER = "visual"
|
|
604
|
+
ALL_TIERS = TIERS + (VISUAL_TIER,)
|
|
605
|
+
SURFACE_CLASSES = ("flagship", "product", "internal", "ad")
|
|
606
|
+
VISUAL_REQUIRED = ("flagship", "product")
|
|
587
607
|
TIER_KEYS = ("node", "tier", "verdict", "scope", "confirms", "findings",
|
|
588
608
|
"evidence", "not_examined")
|
|
589
609
|
TIER_VERDICTS = ("pass", "fail")
|
|
@@ -596,10 +616,10 @@ SEVERITIES = ("breaks", "risk")
|
|
|
596
616
|
# "the unit tier's verdict…"). Widened to the tense and possessive forms the
|
|
597
617
|
# reader planted; still a closed list on purpose — a looser net here starts
|
|
598
618
|
# matching a report's honest prose about its OWN tier.
|
|
599
|
-
CROSS_TIER = re.compile(r"\b(?:unit|seam|product)\s+tier(?:'s)?\s+"
|
|
619
|
+
CROSS_TIER = re.compile(r"\b(?:unit|seam|product|visual)\s+tier(?:'s)?\s+"
|
|
600
620
|
r"(?:passed|failed|says|said|confirm\w*|verdict|report)"
|
|
601
621
|
r"|\btier\s+\d\s+(?:passed|failed|says|said|confirm\w*)"
|
|
602
|
-
r"|as\s+the\s+(?:unit|seam|product)\s+tier", re.I)
|
|
622
|
+
r"|as\s+the\s+(?:unit|seam|product|visual)\s+tier", re.I)
|
|
603
623
|
|
|
604
624
|
|
|
605
625
|
def tier_violations(t):
|
|
@@ -622,9 +642,9 @@ def tier_violations(t):
|
|
|
622
642
|
|
|
623
643
|
if not isinstance(t["node"], str) or not t["node"].startswith(NODE_ID):
|
|
624
644
|
out.append("tier report `node` is %r, which is not a node id" % (t["node"],))
|
|
625
|
-
if t["tier"] not in
|
|
645
|
+
if t["tier"] not in ALL_TIERS:
|
|
626
646
|
out.append("tier report `tier` is %r — it must be one of %s"
|
|
627
|
-
% (t["tier"], ", ".join(
|
|
647
|
+
% (t["tier"], ", ".join(ALL_TIERS)))
|
|
628
648
|
if t["verdict"] not in TIER_VERDICTS:
|
|
629
649
|
out.append("tier report `verdict` is %r — it must be `pass` or `fail`, because "
|
|
630
650
|
"a certification that admits a third state admits a maybe"
|
|
@@ -1482,6 +1502,14 @@ def cmd_certify(graph, args):
|
|
|
1482
1502
|
die("certification is missing the %s report(s) — all three are required, because "
|
|
1483
1503
|
"the level nobody read is the level the defect survives at"
|
|
1484
1504
|
% ", ".join("`%s`" % m for m in missing))
|
|
1505
|
+
sclass = node.get("surface_class")
|
|
1506
|
+
if sclass in VISUAL_REQUIRED and VISUAL_TIER not in reports:
|
|
1507
|
+
die("certification is missing the `%s` report — %s is a %s surface, and on one the "
|
|
1508
|
+
"pixels are part of the requirement: none of unit, seam or product opens the "
|
|
1509
|
+
"contact sheet, so without the fourth reading nobody looked at what a user sees "
|
|
1510
|
+
"(references/certification.md → *The fourth reading*)" % (VISUAL_TIER, nid, sclass))
|
|
1511
|
+
# The tiers THIS round read, in a stable order: the three always, `visual` when given.
|
|
1512
|
+
tiers = [x for x in ALL_TIERS if x in reports]
|
|
1485
1513
|
|
|
1486
1514
|
# The stamp, read here and never accepted from a report — same law as `close`.
|
|
1487
1515
|
import subprocess
|
|
@@ -1493,7 +1521,7 @@ def cmd_certify(graph, args):
|
|
|
1493
1521
|
|
|
1494
1522
|
prior = node.get("certification") or {}
|
|
1495
1523
|
round_no = int(prior.get("round") or 0) + 1
|
|
1496
|
-
tiers_now = {x: reports[x]["verdict"] for x in
|
|
1524
|
+
tiers_now = {x: reports[x]["verdict"] for x in tiers}
|
|
1497
1525
|
history = list(prior.get("history") or []) + [tiers_now]
|
|
1498
1526
|
node["certification"] = {
|
|
1499
1527
|
"round": round_no,
|
|
@@ -1502,11 +1530,11 @@ def cmd_certify(graph, args):
|
|
|
1502
1530
|
"history": history,
|
|
1503
1531
|
}
|
|
1504
1532
|
|
|
1505
|
-
failed = [x for x in
|
|
1533
|
+
failed = [x for x in tiers if tiers_now[x] == "fail"]
|
|
1506
1534
|
|
|
1507
1535
|
# Churn, measured. A tier that has failed in every round so far is the one the
|
|
1508
1536
|
# operator needs named; counting it here is what makes the loop visible.
|
|
1509
|
-
churning = [x for x in
|
|
1537
|
+
churning = [x for x in tiers
|
|
1510
1538
|
if len(history) >= 2 and all(h.get(x) == "fail" for h in history)]
|
|
1511
1539
|
|
|
1512
1540
|
save(args.graph, graph)
|
|
@@ -1545,19 +1573,19 @@ def cmd_certify(graph, args):
|
|
|
1545
1573
|
# exactly what `can_continue_around: true` says)
|
|
1546
1574
|
verdict = {
|
|
1547
1575
|
"node": nid,
|
|
1548
|
-
"done": [c for x in
|
|
1576
|
+
"done": [c for x in tiers for c in reports[x]["confirms"]],
|
|
1549
1577
|
"not_done": [],
|
|
1550
1578
|
"not_verified": ["%s: %s" % (x, n)
|
|
1551
|
-
for x in
|
|
1579
|
+
for x in tiers for n in reports[x]["not_examined"]],
|
|
1552
1580
|
"blockers": [
|
|
1553
1581
|
{"what": "%s (%s, found by the `%s` tier)" % (f["what"], f["where"], x),
|
|
1554
1582
|
"blocks": [], "can_continue_around": True}
|
|
1555
|
-
for x in
|
|
1583
|
+
for x in tiers for f in reports[x]["findings"]
|
|
1556
1584
|
if f.get("severity") == "risk"
|
|
1557
1585
|
],
|
|
1558
1586
|
"replan": {"possible": True, "add": [], "park": [],
|
|
1559
|
-
"why": "certified at all
|
|
1560
|
-
"evidence": ["%s: %s" % (x, e) for x in
|
|
1587
|
+
"why": "certified at all %d tiers in round %d" % (len(tiers), round_no)},
|
|
1588
|
+
"evidence": ["%s: %s" % (x, e) for x in tiers for e in reports[x]["evidence"]],
|
|
1561
1589
|
}
|
|
1562
1590
|
# Proof identity (FIX-PF-02.01): the certification tested THIS tree, so it
|
|
1563
1591
|
# records the commit it tested into the verdict it hands `close`. Without it
|
|
@@ -1579,7 +1607,7 @@ def cmd_certify(graph, args):
|
|
|
1579
1607
|
# cannot hand the run a verdict its own consumer refuses.
|
|
1580
1608
|
broken = verdict_violations(verdict)
|
|
1581
1609
|
if broken:
|
|
1582
|
-
die("
|
|
1610
|
+
die("every tier passed and the assembled verdict is still malformed — this "
|
|
1583
1611
|
"is a defect in `certify`, not in the reports:\n " + "\n ".join(broken))
|
|
1584
1612
|
|
|
1585
1613
|
out = args.verdict_out or os.path.join(os.path.dirname(args.graph) or ".",
|
|
@@ -1589,7 +1617,7 @@ def cmd_certify(graph, args):
|
|
|
1589
1617
|
json.dump(verdict, fh, indent=2, ensure_ascii=False)
|
|
1590
1618
|
fh.write("\n")
|
|
1591
1619
|
os.replace(tmp, out)
|
|
1592
|
-
print("%s: certified at
|
|
1620
|
+
print("%s: certified at %s in round %d" % (nid, ", ".join(tiers), round_no))
|
|
1593
1621
|
print("verdict written to %s — close it with:" % out)
|
|
1594
1622
|
print(" graph.py close --verdict %s" % out)
|
|
1595
1623
|
return 0
|
|
@@ -1797,8 +1825,9 @@ VERBS = {
|
|
|
1797
1825
|
"coverage": (cmd_coverage, "every requirement and the nodes serving it; exits 1 on a gap"),
|
|
1798
1826
|
"add": (cmd_add, "add a node mid-run"),
|
|
1799
1827
|
"park": (cmd_park, "park a node, carrying the reason"),
|
|
1800
|
-
"certify": (cmd_certify, "require three independent tier reports
|
|
1801
|
-
"the verdict
|
|
1828
|
+
"certify": (cmd_certify, "require three independent tier reports (four on a "
|
|
1829
|
+
"flagship or product surface), then emit the verdict "
|
|
1830
|
+
"`close` consumes"),
|
|
1802
1831
|
"close": (cmd_close, "consume a verdict, close one node and re-plan"),
|
|
1803
1832
|
}
|
|
1804
1833
|
|
|
@@ -37,6 +37,9 @@ What it keeps, and where:
|
|
|
37
37
|
the episode Observatory kept (`keptAs`), the token is dropped, and the run is told to read
|
|
38
38
|
the workflow before writing again. A stale writer therefore stops after one refusal.
|
|
39
39
|
|
|
40
|
+
Ledger lines it reads besides `stage:`: `review:` (the contact sheet's human rounds, stage 6
|
|
41
|
+
and 10 on a visual surface), each carried as evidence on the stage it names.
|
|
42
|
+
|
|
40
43
|
Ledger lines it appends, all of the existing `event:` shape:
|
|
41
44
|
|
|
42
45
|
event: memory — checkpoint <stepId> rev <n> <workflowId> — <ISO-8601>
|
|
@@ -63,6 +66,10 @@ STATE_NAME = "memory.json"
|
|
|
63
66
|
OWNER = "agent:task-pipeline"
|
|
64
67
|
STAGE = re.compile(r"^stage:\s*(\d+)\s+(.+?)\s+—\s+gate\s+(\S+)\s+—\s+verdict\s+(\S+)\s+—\s+(\S+)\s*$")
|
|
65
68
|
TOPIC = re.compile(r"^Run:\s*`([^`]+)`")
|
|
69
|
+
# The contact sheet's human rounds (`references/browser.md` → *The visual half*): one line per
|
|
70
|
+
# return or approval. Carried into the checkpoint of the stage it names, so the number of
|
|
71
|
+
# passes a surface took is measured at the boundary rather than remembered at the end.
|
|
72
|
+
REVIEW = re.compile(r"^review:\s*(\d+)\s+—\s+surface\s+(.+?)\s+—\s+rounds\s+(\d+)\s+—\s+(\S+)\s+—\s+\S+\s*$")
|
|
66
73
|
STATUS = {"pass": "done", "skip": "done", "fail": "blocked"}
|
|
67
74
|
|
|
68
75
|
|
|
@@ -81,6 +88,17 @@ def _now() -> str:
|
|
|
81
88
|
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
82
89
|
|
|
83
90
|
|
|
91
|
+
def _reviews(path: pathlib.Path) -> dict[int, list[str]]:
|
|
92
|
+
"""`review:` lines by the stage id they name, as evidence strings."""
|
|
93
|
+
out: dict[int, list[str]] = {}
|
|
94
|
+
for line in path.read_text(encoding="utf-8").splitlines():
|
|
95
|
+
m = REVIEW.match(line)
|
|
96
|
+
if m:
|
|
97
|
+
out.setdefault(int(m.group(1)), []).append(
|
|
98
|
+
f"review: {m.group(2)} rounds {m.group(3)} {m.group(4)}")
|
|
99
|
+
return out
|
|
100
|
+
|
|
101
|
+
|
|
84
102
|
def _read_ledger(path: pathlib.Path) -> tuple[str, list[tuple]]:
|
|
85
103
|
try:
|
|
86
104
|
text = path.read_text(encoding="utf-8")
|
|
@@ -160,8 +178,10 @@ def emit(ledger: pathlib.Path, project: str | None, constraints: list[str],
|
|
|
160
178
|
# one thing a successor must never act without (found by the live receipt, 2026-10-05).
|
|
161
179
|
constraints = list(dict.fromkeys([*state.get("constraints", []), *constraints]))
|
|
162
180
|
credentials = list(dict.fromkeys([*state.get("credentials", []), *credentials]))
|
|
181
|
+
reviews = _reviews(ledger)
|
|
163
182
|
done = [{"step_id": f"stage-{s[0]}", "result": f"{s[1]}: gate {s[2]}, verdict {s[3]} at {s[4]}",
|
|
164
|
-
"evidence": [f"ledger: {ledger.name}"
|
|
183
|
+
"evidence": [f"ledger: {ledger.name}", *reviews.get(s[0], [])]}
|
|
184
|
+
for s in stages if s[3] in ("pass", "skip")]
|
|
165
185
|
nxt = sid + 1 if verdict in ("pass", "skip") else sid
|
|
166
186
|
last = max(names) if names else 10
|
|
167
187
|
open_steps = [] if sid >= last and verdict == "pass" else [{
|