@outerlayer/cli 0.2.0 → 0.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +132 -0
- package/README.md +229 -60
- package/dist/agent-setup-IFH3FB5X.js +197 -0
- package/dist/build-TZUUTRGH.js +9 -0
- package/dist/build-info.json +1 -1
- package/dist/check-7MFH6HNC.js +94 -0
- package/dist/chunk-3TFPDVTI.js +94 -0
- package/dist/{chunk-4TNZMO7V.js → chunk-4C4THABW.js} +5566 -754
- package/dist/{chunk-R5KBGVII.js → chunk-5QRP3MVS.js} +30 -14
- package/dist/chunk-5Y3QQRUH.js +300 -0
- package/dist/chunk-774HEQL6.js +178 -0
- package/dist/{work-pr-cmd-AZQPWH4H.js → chunk-7WERLFVR.js} +9 -26
- package/dist/chunk-ABIWDUMS.js +104 -0
- package/dist/chunk-AUFG23AA.js +534 -0
- package/dist/chunk-B2G7JHHB.js +42 -0
- package/dist/chunk-BCJNHMZT.js +34 -0
- package/dist/chunk-BCLJSQCV.js +56 -0
- package/dist/chunk-BFESLKPP.js +61 -0
- package/dist/chunk-BJ3KTMDH.js +63 -0
- package/dist/chunk-BKO6JEZI.js +47 -0
- package/dist/chunk-BOLTI6LR.js +37 -0
- package/dist/chunk-CBPPJSAR.js +89 -0
- package/dist/chunk-DM3VFDS3.js +1461 -0
- package/dist/{chunk-65WXAANR.js → chunk-DVDEBNQJ.js} +17 -2
- package/dist/{chunk-TFUIDMOB.js → chunk-EABW6AJQ.js} +12 -1
- package/dist/{chunk-KFYJV2ZG.js → chunk-EMY4I27X.js} +1 -1
- package/dist/{chunk-NTTPJV35.js → chunk-F47JBIAW.js} +3 -1
- package/dist/chunk-F6GFZXZ3.js +60 -0
- package/dist/{chunk-YOOSBOKS.js → chunk-F7CU5ABH.js} +1 -1
- package/dist/{emit-cmd-TWSYYEZI.js → chunk-H5NCNVDA.js} +68 -17
- package/dist/chunk-HZRNMGLF.js +2233 -0
- package/dist/chunk-LF6MLJCY.js +38 -0
- package/dist/{chunk-U32VSRLO.js → chunk-LT5TZHYH.js} +95 -634
- package/dist/chunk-LYCDNMOH.js +83 -0
- package/dist/chunk-M5POMKKW.js +1051 -0
- package/dist/{mcp-install-cmd-FDQH6SEN.js → chunk-MQ3IPHIZ.js} +35 -18
- package/dist/{chunk-D77LS3UI.js → chunk-N5FOE5PS.js} +33 -4
- package/dist/chunk-N6LURUEF.js +64 -0
- package/dist/chunk-OGFGZQA3.js +23 -0
- package/dist/chunk-QDQEUVUF.js +9 -0
- package/dist/chunk-QOZKPLVJ.js +226 -0
- package/dist/chunk-QY22NZUW.js +1764 -0
- package/dist/chunk-RA3O54FC.js +30 -0
- package/dist/chunk-RFUPN6KX.js +71 -0
- package/dist/chunk-SY4GK4SP.js +55 -0
- package/dist/chunk-TBT347UY.js +41 -0
- package/dist/chunk-TWNWNS2Q.js +1084 -0
- package/dist/{chunk-NXLLURA4.js → chunk-UFFXNXLQ.js} +115 -71
- package/dist/chunk-USO2DKBD.js +421 -0
- package/dist/chunk-W4SQCOZU.js +206 -0
- package/dist/{chunk-Q6CL5THG.js → chunk-W5FVUZXV.js} +208 -61
- package/dist/chunk-WFJ65NUD.js +383 -0
- package/dist/{chunk-A3WLZX2F.js → chunk-WIOZAJ2W.js} +15 -2
- package/dist/chunk-WO2BXCTQ.js +83 -0
- package/dist/{chunk-I3ETLSNE.js → chunk-XCCLFVXM.js} +120 -11
- package/dist/{context-materialize-6WKT3RBQ.js → chunk-XDDW4FRS.js} +134 -26
- package/dist/chunk-Y7KJLXYN.js +232 -0
- package/dist/chunk-YFHIIH4B.js +102 -0
- package/dist/chunk-YYYXRJUW.js +101 -0
- package/dist/chunk-ZM2IMMYN.js +76 -0
- package/dist/chunk-ZMYPLFG3.js +71 -0
- package/dist/chunk-ZNA27WEV.js +470 -0
- package/dist/{cli-G7DILYJY.js → cli-W62AFRSW.js} +616 -878
- package/dist/{paths-D2VGWWFI.js → cli-build-K56DK4DS.js} +1 -1
- package/dist/config-XVJYZ4IQ.js +9 -0
- package/dist/connect-cmd-ZTJONP37.js +16 -0
- package/dist/context-adopt-UMA4O5NY.js +105 -0
- package/dist/context-materialize-G2FUFC72.js +19 -0
- package/dist/dist-44CQVWGT.js +6 -0
- package/dist/docker-NEGDLU6D.js +7 -0
- package/dist/doctor-53F2PYY7.js +43 -0
- package/dist/doctor-ZLGZYMDK.js +426 -0
- package/dist/{emit-artifact-cmd-UQVT3OKW.js → emit-artifact-cmd-AKZP7TMF.js} +87 -23
- package/dist/emit-cmd-DCERAU27.js +9 -0
- package/dist/emit-criteria-cmd-7MXQVCKD.js +166 -0
- package/dist/{emit-finding-cmd-FDOU2OPD.js → emit-finding-cmd-N2FTGCVC.js} +34 -31
- package/dist/{emit-result-cmd-H2T4C2K3.js → emit-result-cmd-4CPDBMN2.js} +34 -62
- package/dist/exec-client-4XGXV3XN.js +8 -0
- package/dist/guest-init-GENRHNP7.js +8 -0
- package/dist/hook-fast-EYENPK6F.js +9 -0
- package/dist/{hook-wrap-fast-KXYNX3AD.js → hook-wrap-fast-NPLHJXFK.js} +2 -2
- package/dist/host-key-KDYZ5ECC.js +7 -0
- package/dist/import-capture-cmd-AVZ4VIVJ.js +68 -0
- package/dist/{import-ruler-cmd-7LH2QOMG.js → import-ruler-cmd-GUHL6IY5.js} +1 -1
- package/dist/index.js +4 -4
- package/dist/init-55Y4LHWF.js +36 -0
- package/dist/init-XXEKWZQP.js +221 -0
- package/dist/init-cmd-WLST2F7G.js +135 -0
- package/dist/install-cmd-7WXOIFDO.js +55 -0
- package/dist/lima-XN65D7GN.js +55 -0
- package/dist/login-browser-7XGSV7MA.js +10 -0
- package/dist/{config-POF7DEQW.js → logout-cmd-QAUFDJ6J.js} +3 -1
- package/dist/{logs-TRNPQM42.js → logs-7RYUH5A6.js} +1 -1
- package/dist/loop-JFIBP4B7.js +43 -0
- package/dist/machine-ZBPT2J3R.js +35 -0
- package/dist/mcp-install-cmd-RLK4SO3N.js +9 -0
- package/dist/{mcp-serve-cmd-57EZZOTL.js → mcp-serve-cmd-N6UD2KBX.js} +20 -9
- package/dist/paths-OYKMVYJP.js +6 -0
- package/dist/{pidfile-PTW76F56.js → pidfile-UZRH774M.js} +2 -3
- package/dist/real-deps-SU24ZA2K.js +21 -0
- package/dist/relay-L76HDX72.js +46 -0
- package/dist/settings-J2652U5N.js +7 -0
- package/dist/starter-pack-LIZYMKYQ.js +10 -0
- package/dist/{status-I27IST4L.js → status-UJXWZRN5.js} +30 -13
- package/dist/{statusline-fast-3C5OXHDD.js → statusline-fast-SVTMBMD7.js} +3 -2
- package/dist/sync-cmd-4G4A7EVN.js +28 -0
- package/dist/version-BWM6VLDI.js +6 -0
- package/dist/{watch-3DTPJETH.js → watch-7SKJKGAA.js} +19 -8
- package/dist/{work-claim-cmd-5AUNCLS2.js → work-claim-cmd-5F7VZ42N.js} +34 -21
- package/dist/work-cmd-QAUZ7MTD.js +17 -0
- package/dist/work-comment-cmd-BNCSBTLH.js +106 -0
- package/dist/{work-launch-LT663PB3.js → work-launch-YI4CJDAP.js} +1 -1
- package/dist/work-open-pr-cmd-VFYZHIIW.js +156 -0
- package/dist/work-pr-cmd-PV32ZLSI.js +16 -0
- package/package.json +13 -3
- package/skill-pack/maintained/amend/SKILL.md +104 -0
- package/skill-pack/maintained/emitting-evidence/SKILL.md +108 -0
- package/skill-pack/maintained/emitting-evidence/references/agents-snippet.md +20 -0
- package/skill-pack/maintained/outerlayer/SKILL.md +55 -0
- package/skill-pack/maintained/reporting-findings/SKILL.md +148 -0
- package/skill-pack/template/build/SKILL.md +193 -0
- package/skill-pack/template/build/references/agent-briefs.md +243 -0
- package/skill-pack/template/build/references/criteria-judge.md +91 -0
- package/skill-pack/template/build/references/evidence.md +42 -0
- package/skill-pack/template/build/references/release.md +74 -0
- package/skill-pack/template/build/references/review-briefs.md +275 -0
- package/skill-pack/template/build/references/review-loop.md +158 -0
- package/skill-pack/template/build/scripts/record-criteria.mjs +235 -0
- package/skill-pack/template/spec/SKILL.md +84 -0
- package/skill-pack/template/writing-specs/SKILL.md +134 -0
- package/dist/chunk-DCNOXRMV.js +0 -589
- package/dist/chunk-JJP7YLMN.js +0 -25
- package/dist/chunk-OZ7C3XUE.js +0 -34
- package/dist/chunk-WQ6VGRGZ.js +0 -150
- package/dist/hook-fast-J5LCDHUJ.js +0 -8
- package/dist/import-capture-cmd-EUMBGIV3.js +0 -176
- package/dist/init-PTBITAUO.js +0 -103
- package/dist/login-cmd-IRX6LZT7.js +0 -62
- package/dist/loop-ASZRX3CZ.js +0 -648
- package/dist/sync-cmd-BBAUZ5JD.js +0 -17
- package/dist/work-cmd-2P4BVX47.js +0 -18
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: build
|
|
3
|
+
description: >
|
|
4
|
+
Build one approved issue end to end: implement it, review it with a sized panel
|
|
5
|
+
of agents, run the repository's own checks, open the pull request, and attach
|
|
6
|
+
the review and criteria reports as evidence. Use when a work item exists for an
|
|
7
|
+
issue and you want it built. Invoke with OUTERLAYER_WORK=<number> claude "/build".
|
|
8
|
+
disable-model-invocation: true
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
Build one issue into an evidenced pull request. `/spec` writes and gets
|
|
12
|
+
approval for the issue. This skill builds it. It never writes the spec.
|
|
13
|
+
|
|
14
|
+
This file holds the phase order and the rules that bind every phase. Each
|
|
15
|
+
phase reads its own reference on entry, never before:
|
|
16
|
+
|
|
17
|
+
- [references/review-loop.md](references/review-loop.md): the review phase.
|
|
18
|
+
- [references/review-briefs.md](references/review-briefs.md): the reviewer, refuter and mutation briefs.
|
|
19
|
+
- [references/agent-briefs.md](references/agent-briefs.md): the briefs for every other agent.
|
|
20
|
+
- [references/release.md](references/release.md): sync, pull request, CI.
|
|
21
|
+
- [references/criteria-judge.md](references/criteria-judge.md): the brief that judges each criterion.
|
|
22
|
+
- [references/evidence.md](references/evidence.md): manual test and evidence.
|
|
23
|
+
|
|
24
|
+
Paste a brief verbatim into the agent that runs it. Add only the run's own
|
|
25
|
+
values, such as paths, the branch and the item number.
|
|
26
|
+
|
|
27
|
+
Every text a person will read follows the writing rule in the repository's
|
|
28
|
+
instructions file, if it has one. Otherwise write for a reader who was not in
|
|
29
|
+
the session: short sentences, the outcome first, and plain words. Paste that
|
|
30
|
+
rule into every agent that writes prose.
|
|
31
|
+
|
|
32
|
+
## Arguments
|
|
33
|
+
|
|
34
|
+
`/build` takes no issue number. It builds the work item named by
|
|
35
|
+
`OUTERLAYER_WORK`, the item's number. The launch is:
|
|
36
|
+
|
|
37
|
+
OUTERLAYER_WORK=<number> claude "/build"
|
|
38
|
+
|
|
39
|
+
The item exists because someone ran `outerlayer work build --issue <reference>`.
|
|
40
|
+
Given a request instead of a work item, stop and say:
|
|
41
|
+
|
|
42
|
+
Spec it first: /spec <the request as given>
|
|
43
|
+
|
|
44
|
+
Optional flag: `--unattended` means no person is present and nothing wakes
|
|
45
|
+
the session after a turn ends. Spawn every agent in the foreground and wait
|
|
46
|
+
for it in the same turn. Run the gate in the foreground with the maximum
|
|
47
|
+
timeout. End the turn only at the final report.
|
|
48
|
+
|
|
49
|
+
## State rules
|
|
50
|
+
|
|
51
|
+
These bind the orchestrator in every phase. Rule 5 also binds every spawned agent.
|
|
52
|
+
|
|
53
|
+
1. **Never end a turn waiting on nothing.** End a turn while work is in
|
|
54
|
+
flight only when something live will wake the session: a spawned agent or
|
|
55
|
+
a background command. Before saying "waiting on X", confirm X is live.
|
|
56
|
+
Work that is not live did not happen: redo it once.
|
|
57
|
+
2. **One redo per piece of work.** A second failure of the same work means
|
|
58
|
+
the orchestrator takes it over. It reads what the agent left, finishes the
|
|
59
|
+
brief itself in the foreground, and names the takeover in its status line.
|
|
60
|
+
If the gate still cannot go green, escalate.
|
|
61
|
+
3. **The gate runs in the foreground, one heavy command at a time.** The gate
|
|
62
|
+
is the repository's own pre-push checks and tests, as its instructions
|
|
63
|
+
file names them. Never run it in the background. Never start a suite while
|
|
64
|
+
an agent runs one. Never bypass a hook: no `--no-verify`, no skipped hook,
|
|
65
|
+
even when a hook fails. Fix what the hook reports.
|
|
66
|
+
4. **Checkpoint and re-orient.** Before phase 1, write a state file at
|
|
67
|
+
`"$(git rev-parse --absolute-git-dir)/build-state.json"`, replacing any
|
|
68
|
+
earlier one. Record the phase, the review round, the spawned agents, the
|
|
69
|
+
agent count, the issue reference, the pull request number, the branch, and
|
|
70
|
+
from phase 4 the tier verdict. Overwrite it at every phase change. On
|
|
71
|
+
entering a phase, after any wake, and after any compaction, read the state
|
|
72
|
+
file, `git status`, the branch, the live agent list, and the issue or pull
|
|
73
|
+
request when the phase uses them. Live sources win over the file. Treat the
|
|
74
|
+
file as absent when its branch does not match the current branch.
|
|
75
|
+
When you overwrite the file, keep its `criteria` entry: phase 2 writes it
|
|
76
|
+
and the judge reads it.
|
|
77
|
+
Beside it, create `"$(git rev-parse --absolute-git-dir)/build-reports/"`,
|
|
78
|
+
emptied of any earlier run's files. Every spawned agent writes its full
|
|
79
|
+
report there and returns a summary of at most ten lines. Give each agent
|
|
80
|
+
the absolute path. Reports never land in the working tree, so review never
|
|
81
|
+
reads them as product code. Read summaries and pass file paths, never
|
|
82
|
+
report bodies. Delete both at the end of phase 6.
|
|
83
|
+
Once a pull request exists, read `outerlayer work threads --item
|
|
84
|
+
"$OUTERLAYER_WORK" --json`. Each thread with `waitingOn` of `agent` holds a
|
|
85
|
+
person's comment. Answer each before the run ends, as the `amend` skill does.
|
|
86
|
+
5. **A session never records a pass or fail on an artifact.** It emits
|
|
87
|
+
artifacts. A person decides pass or fail.
|
|
88
|
+
6. **Observed status only.** Every status claim in the final report (a gate
|
|
89
|
+
result, a CI conclusion, mergeability) quotes command output observed
|
|
90
|
+
after the last change it describes. No quoted output means unknown, never
|
|
91
|
+
green. Quoted output goes in the report to the user. Pull request bodies
|
|
92
|
+
carry curated links, never raw output.
|
|
93
|
+
7. **Files no issue.** The build files no issue anywhere, and no agent does.
|
|
94
|
+
A defect outside the change, or a follow-up the fix pass deferred, becomes a
|
|
95
|
+
candidate issue in a report: a title, a body and a target repository. The
|
|
96
|
+
final report lists every candidate for the user to file with `/spec`.
|
|
97
|
+
8. **Reader pass.** No text reaches the git host as the session wrote it.
|
|
98
|
+
The pull request title and body, and every comment, go through one cold
|
|
99
|
+
rewrite agent. Its only inputs are the draft and the writing rule. It
|
|
100
|
+
returns the rewrite and the terms it cut or explained. Check that every
|
|
101
|
+
path, identifier, number, criterion id and link survived, and restore any
|
|
102
|
+
that did not.
|
|
103
|
+
9. **Run budget.** Spawn at most sixty agents in a run. Keep the count in the
|
|
104
|
+
state file. Past three quarters, say where the count stands in the status
|
|
105
|
+
line. A run that would exceed the budget stops and lists what is left
|
|
106
|
+
unverified. The review phase also holds a limit on agents in flight.
|
|
107
|
+
|
|
108
|
+
## Phases
|
|
109
|
+
|
|
110
|
+
Run them in order.
|
|
111
|
+
|
|
112
|
+
1. **Intake.** Read `OUTERLAYER_WORK`. If it is missing or does not parse,
|
|
113
|
+
stop and hand back:
|
|
114
|
+
|
|
115
|
+
outerlayer work build --issue <reference> # once; prints the item number
|
|
116
|
+
OUTERLAYER_WORK=<number> claude "/build"
|
|
117
|
+
|
|
118
|
+
Run `outerlayer work status --item "$OUTERLAYER_WORK" --json` and read the
|
|
119
|
+
item's tracker key. It is a GitHub `#n`, or a Linear or Jira key. Read the
|
|
120
|
+
issue through whatever tracker tool the team has. The issue is the source
|
|
121
|
+
of truth for the whole run, and every later agent is pointed at it.
|
|
122
|
+
Restate the task in one or two sentences and confirm the issue is still
|
|
123
|
+
open and unclaimed. If the item shows a live claim by another runner that
|
|
124
|
+
has not expired, stop and name it.
|
|
125
|
+
Phase 2 finds the issue's criteria. If it finds some, they are the
|
|
126
|
+
definition of done, and each declares a proof form or a test. If it finds
|
|
127
|
+
none, build from the issue's description. An issue without criteria is
|
|
128
|
+
built, never sent back. Then no judge runs.
|
|
129
|
+
Ask nothing from here until the final report. An agent that needs a ruling
|
|
130
|
+
states its assumption in its report and continues. The final report lists
|
|
131
|
+
every assumption.
|
|
132
|
+
2. **Read the criteria.** Record the issue's criteria on the item, so the
|
|
133
|
+
Criteria tab, the item's status and the judge hold the pull request to one
|
|
134
|
+
list. Save the issue body to a file and run:
|
|
135
|
+
|
|
136
|
+
node .outerlayer/skills/build/scripts/record-criteria.mjs --item "$OUTERLAYER_WORK" --issue-file <body file>
|
|
137
|
+
|
|
138
|
+
The script reads two shapes, in this order:
|
|
139
|
+
|
|
140
|
+
1. An `## Acceptance criteria` section whose items each start with an id in
|
|
141
|
+
backticks. The id and the text after it are kept as written. A
|
|
142
|
+
`(proof: <kind>)` on the line sets `proof`. Otherwise `proof` is `null`.
|
|
143
|
+
2. A heading or line that reads `Acceptance`, `Acceptance criteria` or
|
|
144
|
+
`Done when`, in any case, with or without a colon, followed by a bullet
|
|
145
|
+
list. Each bullet is one criterion. Its text is the bullet's text. Its id
|
|
146
|
+
is `AC-<item number>-NN`, numbered from `01` in list order. `proof` is
|
|
147
|
+
`null`. Never infer a proof kind from the words.
|
|
148
|
+
|
|
149
|
+
It writes the list to
|
|
150
|
+
`"$(git rev-parse --absolute-git-dir)/build-reports/criteria.json"` and runs
|
|
151
|
+
`outerlayer emit criteria` on it. The gateway decides whether a list may be
|
|
152
|
+
recorded. A 409 means the item already has a recorded list. That list
|
|
153
|
+
stands, and nothing is recorded. No command returns the item's list, so the
|
|
154
|
+
build cannot read it: it holds the work to the issue's reading of the criteria and says
|
|
155
|
+
so in the final report. Any other
|
|
156
|
+
refusal is written to the state file with its message, and the build
|
|
157
|
+
continues. The script writes the outcome under `criteria` in the state
|
|
158
|
+
file: `source` (`issue-ids`, `issue-bullets`, `item` or `none`), `recorded`,
|
|
159
|
+
`count` and `list`. On a 409, `list` is empty, and the issue's reading of the
|
|
160
|
+
criteria goes under `issueReading`. If neither shape is found, nothing is recorded and the
|
|
161
|
+
build continues at the small tier, as for any issue without criteria. Change
|
|
162
|
+
nothing in the issue. A criterion you think is wrong becomes a note in the
|
|
163
|
+
final report, never an edit. The intake summary names how many criteria were
|
|
164
|
+
recorded and from which shape, or says that none were found.
|
|
165
|
+
3. **Implement.** Branch `feat/<slug>` from the default branch. Spawn the
|
|
166
|
+
implementer with the Implementer brief from
|
|
167
|
+
[references/agent-briefs.md](references/agent-briefs.md), the issue, the
|
|
168
|
+
branch and the instructions-file pointer. The phase ends when the
|
|
169
|
+
repository's own pre-push checks and tests pass on the branch. Review
|
|
170
|
+
never opens on a red branch. A branch that cannot go green escalates.
|
|
171
|
+
4. **Review.** Read [references/review-loop.md](references/review-loop.md) on
|
|
172
|
+
entry and run it. It sizes the panel to the finished diff: one generalist
|
|
173
|
+
for a small diff, the generalist and one focused reviewer for a medium
|
|
174
|
+
one, the generalist and up to three for a large or risky one. Findings are
|
|
175
|
+
merged into classes, checked by a refuter, fixed, and rechecked once. Then
|
|
176
|
+
write the review report as an HTML file and emit it:
|
|
177
|
+
|
|
178
|
+
outerlayer emit artifact <report>.html --for code-review-ran --caption "Review of the change and what it found"
|
|
179
|
+
|
|
180
|
+
5. **Sync and pull request.** Read [references/release.md](references/release.md)
|
|
181
|
+
on entry and run it. A release agent rebases, gates and pushes. Open the
|
|
182
|
+
pull request. Then run `outerlayer work pr <number>` so the item links the
|
|
183
|
+
pull request. When the issue has criteria, a fresh agent judges them with
|
|
184
|
+
[references/criteria-judge.md](references/criteria-judge.md), and the
|
|
185
|
+
judge's HTML report is emitted:
|
|
186
|
+
|
|
187
|
+
outerlayer emit artifact <judge-report>.html --for acceptance-criteria --caption "Each criterion judged against its test"
|
|
188
|
+
|
|
189
|
+
Watch CI until every check has concluded.
|
|
190
|
+
6. **Manual test and evidence.** Read [references/evidence.md](references/evidence.md)
|
|
191
|
+
when the pull request opens, and run it. Capture the proof each criterion
|
|
192
|
+
declares, emit it, then end the run with the final report. The report
|
|
193
|
+
lists every candidate issue and every assumption.
|
|
@@ -0,0 +1,243 @@
|
|
|
1
|
+
# Briefs for every agent outside the review panel
|
|
2
|
+
|
|
3
|
+
## Contents
|
|
4
|
+
|
|
5
|
+
- How to use this file
|
|
6
|
+
- Shared rules
|
|
7
|
+
- Implementer brief
|
|
8
|
+
- Merger brief
|
|
9
|
+
- Fix brief
|
|
10
|
+
- Release brief: sync and push
|
|
11
|
+
- Release brief: CI and mergeability
|
|
12
|
+
- Evidence brief
|
|
13
|
+
|
|
14
|
+
## How to use this file
|
|
15
|
+
|
|
16
|
+
SKILL.md says which agent runs in which phase. This file says what each one
|
|
17
|
+
is told. The panel's reviewer, refuter and mutation briefs are in
|
|
18
|
+
[review-briefs.md](review-briefs.md). Paste every brief verbatim. The
|
|
19
|
+
orchestrator adds only the run's own values, such as file paths, the branch,
|
|
20
|
+
the issue reference and the pull request number, and may append a short focus
|
|
21
|
+
addendum that adds emphasis. It never edits or trims a brief.
|
|
22
|
+
|
|
23
|
+
Each brief names the report file the agent writes. The agent writes the full
|
|
24
|
+
report there and returns a summary of at most ten lines.
|
|
25
|
+
|
|
26
|
+
## Shared rules
|
|
27
|
+
|
|
28
|
+
Paste this block at the top of every brief in this file.
|
|
29
|
+
|
|
30
|
+
Read the repository's instructions file first and hold your work to it.
|
|
31
|
+
Write every text a person will read, such as reports, commit messages and
|
|
32
|
+
candidate issue bodies, in short sentences with plain words, leading with the
|
|
33
|
+
outcome.
|
|
34
|
+
|
|
35
|
+
Never create an issue in any tracker. Work that belongs in its own issue
|
|
36
|
+
becomes a candidate issue in your report: a title, a body and the target
|
|
37
|
+
repository. The user decides whether to file it.
|
|
38
|
+
|
|
39
|
+
Never ask the user a question. If you need a ruling, state your assumption in
|
|
40
|
+
your report and continue.
|
|
41
|
+
|
|
42
|
+
Run every long command, such as the full gate, the mutation tool or the
|
|
43
|
+
integration suite, in the foreground with the maximum timeout. Never
|
|
44
|
+
background one, and never end a turn waiting on a notification.
|
|
45
|
+
|
|
46
|
+
Every agent in a run shares one memory cap. Run only the test files for the
|
|
47
|
+
code you change or check, one at a time. Run a whole package's suite, a
|
|
48
|
+
coverage run, a repository-wide typecheck or the full gate only where this
|
|
49
|
+
brief names it. Two of those at once overrun the cap, and the kernel kills
|
|
50
|
+
test workers that then look like flaky tests.
|
|
51
|
+
|
|
52
|
+
Put scratch copies and extra worktrees under the system temp directory.
|
|
53
|
+
|
|
54
|
+
Never bypass a hook. Do not use `--no-verify`, skip a hook, or disable one,
|
|
55
|
+
even when it fails. Fix what the hook reports.
|
|
56
|
+
|
|
57
|
+
Never change the identity a command runs under. Never unset or override an
|
|
58
|
+
`OUTERLAYER_*` variable or the key a command would use, and never rerun a
|
|
59
|
+
refused command as someone else. A refusal is a result: write it in your
|
|
60
|
+
report with the command, and leave that step undone.
|
|
61
|
+
|
|
62
|
+
A session never records a pass or fail on an artifact.
|
|
63
|
+
|
|
64
|
+
Write your full report to the path this brief names. Then return at most ten
|
|
65
|
+
lines. Anything you leave out of the file is lost to the next stage.
|
|
66
|
+
|
|
67
|
+
## Implementer brief
|
|
68
|
+
|
|
69
|
+
The shared rules apply. Your inputs are the issue, the branch and the
|
|
70
|
+
instructions-file pointer.
|
|
71
|
+
|
|
72
|
+
The issue is the definition of done. When it has criteria, each one is a
|
|
73
|
+
requirement. Make every test cite the criterion id it proves, in the form the
|
|
74
|
+
team uses.
|
|
75
|
+
|
|
76
|
+
Write the first commit red, with no implementation. It holds one test that
|
|
77
|
+
drives the real chain, from input through storage to read-back, with nothing
|
|
78
|
+
mocked at the store boundary, plus one test per criterion written from the
|
|
79
|
+
criterion's own words. Read every failure. Each test must fail because the
|
|
80
|
+
behavior is absent, not because of a typo or a missing import. Then implement
|
|
81
|
+
until green. When the issue has no criteria, write tests from its description
|
|
82
|
+
in the same order.
|
|
83
|
+
|
|
84
|
+
Version any new database migration with the current UTC time, never a round or
|
|
85
|
+
guessed one. Two open branches that guess the same time both pass their own
|
|
86
|
+
checks, and the second to merge breaks the build for everyone.
|
|
87
|
+
|
|
88
|
+
While iterating, run only the tests of the packages the diff touches. Never
|
|
89
|
+
run the full gate first to see where things stand. When the last change is in,
|
|
90
|
+
sync with the base branch and run the repository's own pre-push checks and
|
|
91
|
+
tests, as its instructions file names them, in the foreground, once. You are
|
|
92
|
+
not done until they pass. If they cannot be made to pass, stop and escalate.
|
|
93
|
+
Never hand a red branch to review.
|
|
94
|
+
|
|
95
|
+
Do not run a mutation tool here. A mutation checker runs beside the review.
|
|
96
|
+
|
|
97
|
+
A field that crosses package boundaries needs an assertion at every seam it
|
|
98
|
+
crosses: writer, relay and reader.
|
|
99
|
+
|
|
100
|
+
A defect that also reproduces on the base branch is out of scope. It is a
|
|
101
|
+
candidate issue, never a fix here.
|
|
102
|
+
|
|
103
|
+
Close your report with a "Rules that misled you" section. List every sentence
|
|
104
|
+
in the instructions file or a skill you had to work around: the file path, the
|
|
105
|
+
line, the sentence quoted verbatim, and `broken` when its instruction fails or
|
|
106
|
+
`wrong` when it says the wrong thing. Write "none" when there were none.
|
|
107
|
+
|
|
108
|
+
## Merger brief
|
|
109
|
+
|
|
110
|
+
The shared rules apply. Your inputs are the round's report files.
|
|
111
|
+
|
|
112
|
+
Merge the findings into classes. A class is a claim about a pattern, not a
|
|
113
|
+
single site. For each class, build an inventory of every instance in the
|
|
114
|
+
repository, checked with a search. The reports name the instances they
|
|
115
|
+
happened to see. You name the rest.
|
|
116
|
+
|
|
117
|
+
Take the union, never a vote. A finding one reviewer raised and four missed
|
|
118
|
+
survives to verification exactly as one four raised does. Agreement is
|
|
119
|
+
telemetry, not evidence.
|
|
120
|
+
|
|
121
|
+
A finding without its reproduction is a note, not a class. Collect notes from
|
|
122
|
+
every report into a closing "Notes" section and never give one a class id.
|
|
123
|
+
|
|
124
|
+
Write `build-reports/round-<n>-merged.md` with one section per class: the
|
|
125
|
+
claim, its severity and kind, the full inventory, which reviewers found it,
|
|
126
|
+
and the rule citation when a report carried one. Two reports citing the same
|
|
127
|
+
sentence for one class are one citation. Ids are `r<round>-c<k>` for a class,
|
|
128
|
+
`r<round>-n<k>` for a note, `r<round>-i<k>` for a candidate issue and
|
|
129
|
+
`r<round>-x<k>` for a rule a report listed, each `k` counting from 1 within its
|
|
130
|
+
kind. A listed rule is its own entry with its citation and no severity.
|
|
131
|
+
|
|
132
|
+
Close the file with an area table: the classes grouped by the files in their
|
|
133
|
+
inventories, two classes in one area when any file appears in both. Return the
|
|
134
|
+
class table only: id, severity, kind, instance count, reviewers, rule cited or
|
|
135
|
+
none, and fix group.
|
|
136
|
+
|
|
137
|
+
## Fix brief
|
|
138
|
+
|
|
139
|
+
The shared rules apply. Your input is the list of refuter verdict files, and
|
|
140
|
+
the mutation report when the round had one. Read them. Nobody pastes the
|
|
141
|
+
findings to you.
|
|
142
|
+
|
|
143
|
+
Fix classes, not instances. For each class, repair every instance in its
|
|
144
|
+
inventory and prove nothing is left with a search whose empty result you
|
|
145
|
+
quote. Commit each refuter's failing test together with the fix that turns it
|
|
146
|
+
green.
|
|
147
|
+
|
|
148
|
+
Your brief may name a fix group: a worktree path, a group id and only that
|
|
149
|
+
group's verdict files. Then work in that worktree and touch only the files in
|
|
150
|
+
your classes' inventories and the tests beside them. A fix that needs another
|
|
151
|
+
file is reported as not fixed with the reason `crosses group`. In group mode,
|
|
152
|
+
finish with every verification test in your group green, run one file at a
|
|
153
|
+
time. Run no typecheck, package suite or gate, because other groups run beside
|
|
154
|
+
you.
|
|
155
|
+
|
|
156
|
+
A confirmed finding that also reproduces on the base branch is a candidate
|
|
157
|
+
issue, unless its repair stays inside lines this branch already changes.
|
|
158
|
+
|
|
159
|
+
With no group named, end the pass with the repository's own checks green
|
|
160
|
+
again, the verification tests passing, and the comments and docs next to
|
|
161
|
+
changed behavior brought back in line.
|
|
162
|
+
|
|
163
|
+
Write `build-reports/round-<n>-fix.md`, or `round-<n>-fix-<group>.md` in
|
|
164
|
+
group mode. Name every briefed item and its outcome: fixed, with the green
|
|
165
|
+
test and the hunks, or not fixed, with the reason. An item the report does not
|
|
166
|
+
name counts as not fixed. A green suite is not a report.
|
|
167
|
+
|
|
168
|
+
## Release brief: sync and push
|
|
169
|
+
|
|
170
|
+
The shared rules apply. Your inputs are the state file path, the branch and
|
|
171
|
+
the instructions-file pointer. You run no write command against the tracker
|
|
172
|
+
or the git host beyond the push.
|
|
173
|
+
|
|
174
|
+
Fetch the base branch, then rebase the branch onto it. Merge instead once the
|
|
175
|
+
branch is pushed and someone else may hold it. Resolve every conflict on the
|
|
176
|
+
branch, never in a web editor.
|
|
177
|
+
|
|
178
|
+
A migration this branch adds whose version is at or below the newest one on
|
|
179
|
+
the base branch is renamed to the current UTC time, with every reference to
|
|
180
|
+
its file updated.
|
|
181
|
+
|
|
182
|
+
Run the repository's own pre-push checks and tests, as its instructions file
|
|
183
|
+
names them, in the foreground with the maximum timeout. Leave any mutation
|
|
184
|
+
tool off: the review ran it over these lines already. Then run any check the
|
|
185
|
+
repository runs only in CI, when the diff touches its domain, each in the
|
|
186
|
+
foreground, and quote every result in your report. A check this machine cannot
|
|
187
|
+
run is named in the report with the reason. A check that fails is a red gate,
|
|
188
|
+
and you never push over it. Then push. The hooks run, and you never bypass
|
|
189
|
+
them.
|
|
190
|
+
|
|
191
|
+
When your brief says this is a resync of a branch already pushed and reviewed,
|
|
192
|
+
the surface is the merge's diff, not the branch's. Run the extra checks only
|
|
193
|
+
when that diff reaches their domain, and say which ones the resync did not
|
|
194
|
+
need and why.
|
|
195
|
+
|
|
196
|
+
Write `build-reports/release.md` naming every conflict you resolved and every
|
|
197
|
+
file the sync moved that the review had already cleared. The orchestrator acts
|
|
198
|
+
on that second list, so do not omit a file because the move looked harmless.
|
|
199
|
+
|
|
200
|
+
## Release brief: CI and mergeability
|
|
201
|
+
|
|
202
|
+
The shared rules apply. Your inputs are the branch and the pull request
|
|
203
|
+
number.
|
|
204
|
+
|
|
205
|
+
Watch the pull request's checks until they reach a conclusive state, by
|
|
206
|
+
watching them or by a bounded poll. Any check that waits on results recorded
|
|
207
|
+
after the session ends stays pending for the whole run. Leave it out of the
|
|
208
|
+
wait. CI is conclusive when every other check has finished. Right after a pull
|
|
209
|
+
request opens, "no checks reported" is ambiguous, not green. Query again over
|
|
210
|
+
a short window before believing it.
|
|
211
|
+
|
|
212
|
+
Confirm mergeability, and query again while it reads unknown.
|
|
213
|
+
|
|
214
|
+
Write the quoted command output to `build-reports/release.md`. Return at most
|
|
215
|
+
ten lines naming the CI conclusion, the mergeability and every failed check.
|
|
216
|
+
|
|
217
|
+
## Evidence brief
|
|
218
|
+
|
|
219
|
+
The shared rules apply, and the emitting-evidence skill governs the emit
|
|
220
|
+
mechanics. Your inputs are the issue, the pull request number, the criteria
|
|
221
|
+
that declare a proof form, and the instructions-file pointer.
|
|
222
|
+
|
|
223
|
+
Run the app and drive the feature by hand. A screenshot or video is taken in a
|
|
224
|
+
real browser against the running application, reached through its own routes
|
|
225
|
+
by a signed-in user, inside its real shell and theme. Never capture a test
|
|
226
|
+
render, a jsdom dump or a standalone file of a component. It shows neither what
|
|
227
|
+
a user sees nor that the page works. Seed what the page needs through the
|
|
228
|
+
application's own paths. When a state cannot be reached that way, the criterion
|
|
229
|
+
is not evidenced, and `evidence.md` says why. Before emitting, look at each
|
|
230
|
+
capture: it shows the application's navigation and real rows.
|
|
231
|
+
|
|
232
|
+
For each criterion, put the capture in a script under
|
|
233
|
+
`build-reports/evidence/` and run the script. A later fix can invalidate what
|
|
234
|
+
you captured, and regenerating it must be one command. The scripts go when the
|
|
235
|
+
run's report directory goes. The emitted artifacts outlive the run.
|
|
236
|
+
|
|
237
|
+
Emit each artifact with `outerlayer emit artifact ... --for <criterion id>
|
|
238
|
+
--pr <number>`. Always pass `--pr`. Regenerate with `--replaces` any evidence
|
|
239
|
+
the tree has since invalidated. Then run `outerlayer sync`.
|
|
240
|
+
|
|
241
|
+
Write `build-reports/evidence.md` naming every criterion with its artifact, or
|
|
242
|
+
with the reason none could be produced. A criterion you could not evidence is
|
|
243
|
+
a result, not a gap to leave silent.
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
# Criteria judge
|
|
2
|
+
|
|
3
|
+
Read this when the release phase spawns you. You are a fresh-context agent,
|
|
4
|
+
never the implementer. You judge whether each criterion in the issue is
|
|
5
|
+
actually proven by the test that cites it, not merely joined to one by id. For
|
|
6
|
+
a criterion that declares an artifact proof, you note that an artifact is
|
|
7
|
+
needed.
|
|
8
|
+
|
|
9
|
+
## Contents
|
|
10
|
+
|
|
11
|
+
- Inputs
|
|
12
|
+
- Steps
|
|
13
|
+
- Rubric
|
|
14
|
+
- Output
|
|
15
|
+
- Guardrails
|
|
16
|
+
|
|
17
|
+
## Inputs
|
|
18
|
+
|
|
19
|
+
The orchestrator gives you the worktree path (run every command there), the
|
|
20
|
+
base ref, the pull request number, the path of the build's state file, and the
|
|
21
|
+
path to write your report to. On a large issue it gives you one group of criteria instead.
|
|
22
|
+
Judge only that group. Other judges handle the rest.
|
|
23
|
+
|
|
24
|
+
## Steps
|
|
25
|
+
|
|
26
|
+
1. Read the criteria from the build's state file, under `criteria.list`. They
|
|
27
|
+
are the item's recorded list. If the state file lacks `criteria`, read the
|
|
28
|
+
`criteria` array in `criteria.json` in the build-reports directory. When
|
|
29
|
+
`criteria.source` is `item`, the item already had a recorded list that no
|
|
30
|
+
command returns, so `list` is empty. Judge `criteria.issueReading` instead,
|
|
31
|
+
and say in the report that it is the issue's reading, whose ids may differ
|
|
32
|
+
from the item's recorded list. Each has an id
|
|
33
|
+
and a text, and the text is usually a Given, When and Then. Judge every criterion you were given,
|
|
34
|
+
including one no line of the diff touched. A criterion with `proof: null`
|
|
35
|
+
is proven by a citing test or any artifact bound to its id.
|
|
36
|
+
2. For each criterion, find its citing tests. Search the test files for its
|
|
37
|
+
id. A raw text match is not yet a citation. The convention is a comment such
|
|
38
|
+
as `// proves <id>` directly above the test, or a test that names the id.
|
|
39
|
+
An id that appears only as sample data a test uses as input, never as a proof
|
|
40
|
+
claim, is not a citation. Skip that hit.
|
|
41
|
+
3. A criterion that declares `(proof: screenshot)`, `(proof: video)`,
|
|
42
|
+
`(proof: report)`, `(proof: log)` or `(proof: file)` is `not-judged`, with
|
|
43
|
+
the reason "artifact proof; an artifact bound to this id must be attached to
|
|
44
|
+
the pull request, which this judge cannot see from the worktree".
|
|
45
|
+
4. A criterion with no proof annotation and no citing test is `not-satisfied`,
|
|
46
|
+
with the reason "no test cites this criterion".
|
|
47
|
+
5. Otherwise, read every citing test's block, the test itself plus its
|
|
48
|
+
enclosing group names, and judge it against the rubric.
|
|
49
|
+
|
|
50
|
+
## Rubric
|
|
51
|
+
|
|
52
|
+
A criterion is satisfied only when a test asserts the observable outcome its
|
|
53
|
+
Then clause states, under the Given and When conditions it states, with an
|
|
54
|
+
assertion that would fail if that outcome did not hold.
|
|
55
|
+
|
|
56
|
+
It is not satisfied when the test only cites the id, asserts something weaker
|
|
57
|
+
than the stated outcome, mocks away the very behavior the criterion is about,
|
|
58
|
+
or has no assertion bearing on the outcome, even when it cites the id.
|
|
59
|
+
|
|
60
|
+
When unsure, record `not-satisfied` and state the doubt in the reason. A silent
|
|
61
|
+
pass is worse than a wrong-looking fail a person can overrule.
|
|
62
|
+
|
|
63
|
+
## Output
|
|
64
|
+
|
|
65
|
+
Write an HTML report to the path you were given. It has one row per criterion:
|
|
66
|
+
the id, the verdict (`satisfied`, `not-satisfied` or `not-judged`), and a
|
|
67
|
+
reason of one to three sentences, with no preamble, that quotes the decisive
|
|
68
|
+
assertion or names what is missing. Put a summary line at the top with the
|
|
69
|
+
count of each verdict. The report must open in a browser as a single file with
|
|
70
|
+
no external assets.
|
|
71
|
+
|
|
72
|
+
Then print a table of id, verdict and the first line of the reason, and end
|
|
73
|
+
with exactly one line: `result: pass` or `result: fail`. Fail when any
|
|
74
|
+
criterion is `not-satisfied`.
|
|
75
|
+
|
|
76
|
+
## Guardrails
|
|
77
|
+
|
|
78
|
+
Judge every criterion at the same depth. Read the cited test's assertions and
|
|
79
|
+
compare them to the criterion's own Then clause, every time. A well-worded
|
|
80
|
+
citing comment that restates the criterion is a hint about what to look for,
|
|
81
|
+
never a substitute for reading the assertion beneath it. A citation can be
|
|
82
|
+
wrong or stale. Never shortcut to "many others looked fine, so the rest
|
|
83
|
+
probably are."
|
|
84
|
+
|
|
85
|
+
A long run is expected, and is no reason to sample. Work through the list in
|
|
86
|
+
order, and rewrite the report after each batch so a run that stops partway
|
|
87
|
+
still leaves a readable report.
|
|
88
|
+
|
|
89
|
+
Edit no file but the report. Do not fix a test, do not touch the issue, and do
|
|
90
|
+
not run `outerlayer emit`. That step belongs to the orchestrator once it holds
|
|
91
|
+
the report.
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
# Phase 6: manual test and evidence
|
|
2
|
+
|
|
3
|
+
Read this when the pull request opens. The evidence agent starts then, in the
|
|
4
|
+
same wave as the CI watch. Everything after it in this file runs once CI is
|
|
5
|
+
conclusive and the evidence agent has returned. This is the last phase, and it
|
|
6
|
+
ends the run: it deletes the state file and the report directory. The evidence
|
|
7
|
+
brief is in [agent-briefs.md](agent-briefs.md). Read every value from an
|
|
8
|
+
earlier phase from the state file.
|
|
9
|
+
|
|
10
|
+
## The phase
|
|
11
|
+
|
|
12
|
+
The orchestrator does not run the app, drive a browser or emit an artifact
|
|
13
|
+
itself. Spawn one evidence agent in a fresh context on the Evidence brief, with
|
|
14
|
+
the issue, the pull request number, the criteria that declare a proof form, the
|
|
15
|
+
instructions-file pointer and the emitting-evidence skill.
|
|
16
|
+
|
|
17
|
+
A criterion that declares `proof: screenshot` needs a screenshot bound to its
|
|
18
|
+
id. One that declares `proof: video` needs a video. A screenshot never
|
|
19
|
+
satisfies `proof: video`. A criterion with no proof form is proven by its
|
|
20
|
+
test, and needs no artifact.
|
|
21
|
+
|
|
22
|
+
When a commit lands on the branch after the evidence agent captured, decide for
|
|
23
|
+
each capture from that commit's diff. A capture is stale when the diff touches
|
|
24
|
+
a file the capture's subject runs through: the feature's own files, the shared
|
|
25
|
+
code they import, a migration or schema behind them, or a test the capture is a
|
|
26
|
+
log of. A sync that brings in only files the feature never reaches leaves the
|
|
27
|
+
capture valid. Spawn the evidence agent again with the stale list. It reruns
|
|
28
|
+
their scripts and emits each with `--replaces`. The final report names the
|
|
29
|
+
commit and every capture with its call, kept or regenerated. A commit that
|
|
30
|
+
lands after this call reopens it. Never drive the browser yourself.
|
|
31
|
+
|
|
32
|
+
Re-read the pull request's evidence comment. Resolve or escalate every row that
|
|
33
|
+
needs attention. A row is resolved when a re-read shows it, not when the fixing
|
|
34
|
+
command exits. If no evidence comment exists, report that observation with the
|
|
35
|
+
time, never a belief.
|
|
36
|
+
|
|
37
|
+
End the report to the user with every candidate issue from every stage: title,
|
|
38
|
+
one line and target repository. Or write "no candidate issues". Hand them over.
|
|
39
|
+
Do not ask which to file, and do not file any. The user takes the ones worth
|
|
40
|
+
filing to `/spec`. List every assumption an agent stated.
|
|
41
|
+
|
|
42
|
+
Then delete the state file and the `build-reports/` directory.
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
# Phase 5: sync and pull request
|
|
2
|
+
|
|
3
|
+
Run this once the review loop reaches zero confirmed blocking findings. The
|
|
4
|
+
release briefs are in [agent-briefs.md](agent-briefs.md). Read every value
|
|
5
|
+
from an earlier phase, such as the tier verdict or the branch, from the state
|
|
6
|
+
file, not from the transcript.
|
|
7
|
+
|
|
8
|
+
## Sync and push
|
|
9
|
+
|
|
10
|
+
The orchestrator does not sync, run the gate, push or watch CI itself. Spawn
|
|
11
|
+
one release agent in a fresh context on the "sync and push" brief, with the
|
|
12
|
+
state file path, the branch and the instructions-file pointer. A check that
|
|
13
|
+
fails there has the standing of a red gate and returns to the review phase's
|
|
14
|
+
remaining budget. If its report names files the sync moved that the review had
|
|
15
|
+
already cleared, run one check round on the rebased diff before going on.
|
|
16
|
+
|
|
17
|
+
## Open the pull request
|
|
18
|
+
|
|
19
|
+
Open a scoped pull request that closes the issue. The body carries the
|
|
20
|
+
problem, the change, how to verify it, and, when the change alters what a user
|
|
21
|
+
sees or does, a link to each documentation file added or changed. It carries
|
|
22
|
+
no telemetry, round tables or reviewer roles. Those go in the final report.
|
|
23
|
+
|
|
24
|
+
Run the reader pass on the title and body before opening, and after any later
|
|
25
|
+
edit. Add an "Unverified, non-blocking" section to the review report when
|
|
26
|
+
classes were left unrefuted.
|
|
27
|
+
|
|
28
|
+
The moment the pull request exists, declare it to the work item from this
|
|
29
|
+
session:
|
|
30
|
+
|
|
31
|
+
outerlayer work pr <number>
|
|
32
|
+
|
|
33
|
+
A pull request belongs to a work item only through a declaration from a
|
|
34
|
+
session on that item. Nothing else links it, not the issue it closes and not
|
|
35
|
+
its branch. Without the declaration the item shows no pull request, no
|
|
36
|
+
evaluation and no evidence. Check that the command reports the link as created
|
|
37
|
+
or already present before going on. A CLI that does not know the command needs
|
|
38
|
+
refreshing, and the final report names that as a blocker.
|
|
39
|
+
|
|
40
|
+
## Judge the criteria
|
|
41
|
+
|
|
42
|
+
When the issue has criteria, spawn one fresh agent on
|
|
43
|
+
[criteria-judge.md](criteria-judge.md), with the worktree path, the base ref,
|
|
44
|
+
the pull request number, the path of the build's state file, and a report
|
|
45
|
+
path under `build-reports/`. Above forty
|
|
46
|
+
criteria, spawn one judge per group of about forty in the same wave, each with
|
|
47
|
+
its own report path, and merge their results by concatenation.
|
|
48
|
+
|
|
49
|
+
The judge writes an HTML report with one verdict per criterion. Emit it:
|
|
50
|
+
|
|
51
|
+
outerlayer emit artifact <judge-report>.html --for acceptance-criteria --caption "Each criterion judged against the test that cites it"
|
|
52
|
+
|
|
53
|
+
Then run `outerlayer sync`. List every unsatisfied criterion in the final
|
|
54
|
+
report. A criterion that declares a proof form such as a screenshot or video is
|
|
55
|
+
proven only by an artifact bound to it on the pull request. The judge cannot
|
|
56
|
+
see that, so confirm each has an artifact of the declared kind, and emit any
|
|
57
|
+
that is missing in the evidence phase. When the issue has no criteria, no judge
|
|
58
|
+
runs and no judge report is emitted.
|
|
59
|
+
|
|
60
|
+
## Watch CI
|
|
61
|
+
|
|
62
|
+
With the pull request open, spawn two agents in the same wave. The release
|
|
63
|
+
agent runs again on the "CI and mergeability" brief with the pull request
|
|
64
|
+
number. The evidence agent starts as [evidence.md](evidence.md) describes. The
|
|
65
|
+
manual test needs only the pull request number, and CI never reads the
|
|
66
|
+
captures, so the two waits overlap.
|
|
67
|
+
|
|
68
|
+
Act on the release agent's return. A failed check has the standing of a red
|
|
69
|
+
gate: fix it within the review phase's remaining budget or escalate. Never
|
|
70
|
+
carry it into phase 6. A persistent unknown mergeability escalates. A
|
|
71
|
+
conflict means one more sync through the release agent, told it is a resync so
|
|
72
|
+
it gates the merge's diff and not the branch's. A second conflict escalates.
|
|
73
|
+
Any fix or sync that lands after the evidence agent captured goes through
|
|
74
|
+
phase 6's late-commit rule.
|