@outerlayer/cli 0.2.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (141) hide show
  1. package/CHANGELOG.md +124 -0
  2. package/README.md +226 -59
  3. package/dist/agent-setup-IFH3FB5X.js +197 -0
  4. package/dist/build-TZUUTRGH.js +9 -0
  5. package/dist/build-info.json +1 -1
  6. package/dist/check-WQP2CZDW.js +94 -0
  7. package/dist/chunk-2E54Z3P5.js +232 -0
  8. package/dist/chunk-3TT7RQYB.js +55 -0
  9. package/dist/{chunk-U32VSRLO.js → chunk-3YPKXE5L.js} +90 -632
  10. package/dist/{chunk-Q6CL5THG.js → chunk-5DLBAXY5.js} +208 -61
  11. package/dist/{chunk-R5KBGVII.js → chunk-5QRP3MVS.js} +30 -14
  12. package/dist/chunk-5Y3QQRUH.js +300 -0
  13. package/dist/chunk-6Y63SXFO.js +383 -0
  14. package/dist/chunk-774HEQL6.js +178 -0
  15. package/dist/chunk-7A6RVXPZ.js +2233 -0
  16. package/dist/chunk-7YIOOFVJ.js +71 -0
  17. package/dist/chunk-ABIWDUMS.js +104 -0
  18. package/dist/chunk-AUFG23AA.js +534 -0
  19. package/dist/chunk-B2G7JHHB.js +42 -0
  20. package/dist/chunk-BCJNHMZT.js +34 -0
  21. package/dist/chunk-BCLJSQCV.js +56 -0
  22. package/dist/chunk-BFESLKPP.js +61 -0
  23. package/dist/chunk-BJ3KTMDH.js +63 -0
  24. package/dist/chunk-BKO6JEZI.js +47 -0
  25. package/dist/chunk-BOLTI6LR.js +37 -0
  26. package/dist/chunk-CBPPJSAR.js +89 -0
  27. package/dist/chunk-CBZB6XY6.js +102 -0
  28. package/dist/{chunk-65WXAANR.js → chunk-DVDEBNQJ.js} +17 -2
  29. package/dist/{work-pr-cmd-AZQPWH4H.js → chunk-DVYHCMB2.js} +9 -26
  30. package/dist/{chunk-TFUIDMOB.js → chunk-EABW6AJQ.js} +12 -1
  31. package/dist/{chunk-KFYJV2ZG.js → chunk-EMY4I27X.js} +1 -1
  32. package/dist/{chunk-NTTPJV35.js → chunk-F47JBIAW.js} +3 -1
  33. package/dist/chunk-F6GFZXZ3.js +60 -0
  34. package/dist/{chunk-YOOSBOKS.js → chunk-F7CU5ABH.js} +1 -1
  35. package/dist/chunk-FAJSS2WN.js +1357 -0
  36. package/dist/chunk-FS2VYGZT.js +1764 -0
  37. package/dist/chunk-FV2CCGXK.js +9 -0
  38. package/dist/{chunk-4TNZMO7V.js → chunk-G7FSV7JX.js} +3452 -751
  39. package/dist/{emit-cmd-TWSYYEZI.js → chunk-H5NCNVDA.js} +68 -17
  40. package/dist/chunk-HE2D4EY5.js +83 -0
  41. package/dist/chunk-L2DBRKRA.js +94 -0
  42. package/dist/chunk-LF6MLJCY.js +38 -0
  43. package/dist/chunk-M5POMKKW.js +1051 -0
  44. package/dist/{mcp-install-cmd-FDQH6SEN.js → chunk-MQ3IPHIZ.js} +35 -18
  45. package/dist/{chunk-D77LS3UI.js → chunk-N5FOE5PS.js} +33 -4
  46. package/dist/chunk-N6LURUEF.js +64 -0
  47. package/dist/chunk-NYC54VBD.js +1084 -0
  48. package/dist/chunk-OGFGZQA3.js +23 -0
  49. package/dist/chunk-QOZKPLVJ.js +226 -0
  50. package/dist/chunk-RA3O54FC.js +30 -0
  51. package/dist/chunk-TBT347UY.js +41 -0
  52. package/dist/{chunk-NXLLURA4.js → chunk-UFFXNXLQ.js} +115 -71
  53. package/dist/chunk-USO2DKBD.js +421 -0
  54. package/dist/chunk-W4SQCOZU.js +206 -0
  55. package/dist/{chunk-A3WLZX2F.js → chunk-WIOZAJ2W.js} +15 -2
  56. package/dist/chunk-WO2BXCTQ.js +83 -0
  57. package/dist/{chunk-I3ETLSNE.js → chunk-XCCLFVXM.js} +120 -11
  58. package/dist/{context-materialize-6WKT3RBQ.js → chunk-XDDW4FRS.js} +134 -26
  59. package/dist/chunk-YYYXRJUW.js +101 -0
  60. package/dist/chunk-ZM2IMMYN.js +76 -0
  61. package/dist/chunk-ZMYPLFG3.js +71 -0
  62. package/dist/chunk-ZNA27WEV.js +470 -0
  63. package/dist/{cli-G7DILYJY.js → cli-NHUXABYH.js} +590 -876
  64. package/dist/{paths-D2VGWWFI.js → cli-build-K56DK4DS.js} +1 -1
  65. package/dist/config-XVJYZ4IQ.js +9 -0
  66. package/dist/connect-cmd-OOG4TPU7.js +16 -0
  67. package/dist/context-adopt-UMA4O5NY.js +105 -0
  68. package/dist/context-materialize-G2FUFC72.js +19 -0
  69. package/dist/dist-RW4UPPC2.js +6 -0
  70. package/dist/docker-NEGDLU6D.js +7 -0
  71. package/dist/doctor-53F2PYY7.js +43 -0
  72. package/dist/doctor-UFVL4PZY.js +426 -0
  73. package/dist/{emit-artifact-cmd-UQVT3OKW.js → emit-artifact-cmd-KTYW2TN2.js} +24 -20
  74. package/dist/emit-cmd-DCERAU27.js +9 -0
  75. package/dist/emit-criteria-cmd-VAUUSAYD.js +166 -0
  76. package/dist/{emit-finding-cmd-FDOU2OPD.js → emit-finding-cmd-UR4ID45Z.js} +34 -31
  77. package/dist/{emit-result-cmd-H2T4C2K3.js → emit-result-cmd-FDTUKGKE.js} +34 -62
  78. package/dist/exec-client-4XGXV3XN.js +8 -0
  79. package/dist/guest-init-GENRHNP7.js +8 -0
  80. package/dist/hook-fast-EYENPK6F.js +9 -0
  81. package/dist/{hook-wrap-fast-KXYNX3AD.js → hook-wrap-fast-NPLHJXFK.js} +2 -2
  82. package/dist/host-key-KDYZ5ECC.js +7 -0
  83. package/dist/import-capture-cmd-AVZ4VIVJ.js +68 -0
  84. package/dist/{import-ruler-cmd-7LH2QOMG.js → import-ruler-cmd-GUHL6IY5.js} +1 -1
  85. package/dist/index.js +4 -4
  86. package/dist/init-DXBCOAJK.js +36 -0
  87. package/dist/init-YEXK35CF.js +221 -0
  88. package/dist/init-cmd-NHR7PRSQ.js +135 -0
  89. package/dist/install-cmd-7WXOIFDO.js +55 -0
  90. package/dist/lima-XN65D7GN.js +55 -0
  91. package/dist/login-browser-7XGSV7MA.js +10 -0
  92. package/dist/{config-POF7DEQW.js → logout-cmd-QAUFDJ6J.js} +3 -1
  93. package/dist/{logs-TRNPQM42.js → logs-7RYUH5A6.js} +1 -1
  94. package/dist/loop-VZOLL4WQ.js +43 -0
  95. package/dist/machine-WSG52J75.js +35 -0
  96. package/dist/mcp-install-cmd-RLK4SO3N.js +9 -0
  97. package/dist/{mcp-serve-cmd-57EZZOTL.js → mcp-serve-cmd-JPTP5FMS.js} +20 -9
  98. package/dist/paths-OYKMVYJP.js +6 -0
  99. package/dist/{pidfile-PTW76F56.js → pidfile-UZRH774M.js} +2 -3
  100. package/dist/real-deps-SU24ZA2K.js +21 -0
  101. package/dist/relay-L76HDX72.js +46 -0
  102. package/dist/settings-J2652U5N.js +7 -0
  103. package/dist/starter-pack-LIZYMKYQ.js +10 -0
  104. package/dist/{status-I27IST4L.js → status-2UZKKNB7.js} +30 -13
  105. package/dist/{statusline-fast-3C5OXHDD.js → statusline-fast-SVTMBMD7.js} +3 -2
  106. package/dist/sync-cmd-VPPYCNNM.js +28 -0
  107. package/dist/version-BWM6VLDI.js +6 -0
  108. package/dist/{watch-3DTPJETH.js → watch-5C4ZBBGO.js} +19 -8
  109. package/dist/{work-claim-cmd-5AUNCLS2.js → work-claim-cmd-LSO2B2EX.js} +33 -20
  110. package/dist/work-cmd-UOQO2AD4.js +17 -0
  111. package/dist/work-comment-cmd-WF3EOLAE.js +106 -0
  112. package/dist/{work-launch-LT663PB3.js → work-launch-YI4CJDAP.js} +1 -1
  113. package/dist/work-open-pr-cmd-4SSQZMYG.js +156 -0
  114. package/dist/work-pr-cmd-6AN6RRQJ.js +16 -0
  115. package/package.json +13 -3
  116. package/skill-pack/maintained/amend/SKILL.md +104 -0
  117. package/skill-pack/maintained/emitting-evidence/SKILL.md +99 -0
  118. package/skill-pack/maintained/emitting-evidence/references/agents-snippet.md +20 -0
  119. package/skill-pack/maintained/outerlayer/SKILL.md +55 -0
  120. package/skill-pack/maintained/reporting-findings/SKILL.md +148 -0
  121. package/skill-pack/template/build/SKILL.md +193 -0
  122. package/skill-pack/template/build/references/agent-briefs.md +243 -0
  123. package/skill-pack/template/build/references/criteria-judge.md +91 -0
  124. package/skill-pack/template/build/references/evidence.md +42 -0
  125. package/skill-pack/template/build/references/release.md +74 -0
  126. package/skill-pack/template/build/references/review-briefs.md +275 -0
  127. package/skill-pack/template/build/references/review-loop.md +158 -0
  128. package/skill-pack/template/build/scripts/record-criteria.mjs +235 -0
  129. package/skill-pack/template/spec/SKILL.md +84 -0
  130. package/skill-pack/template/writing-specs/SKILL.md +134 -0
  131. package/dist/chunk-DCNOXRMV.js +0 -589
  132. package/dist/chunk-JJP7YLMN.js +0 -25
  133. package/dist/chunk-OZ7C3XUE.js +0 -34
  134. package/dist/chunk-WQ6VGRGZ.js +0 -150
  135. package/dist/hook-fast-J5LCDHUJ.js +0 -8
  136. package/dist/import-capture-cmd-EUMBGIV3.js +0 -176
  137. package/dist/init-PTBITAUO.js +0 -103
  138. package/dist/login-cmd-IRX6LZT7.js +0 -62
  139. package/dist/loop-ASZRX3CZ.js +0 -648
  140. package/dist/sync-cmd-BBAUZ5JD.js +0 -17
  141. package/dist/work-cmd-2P4BVX47.js +0 -18
@@ -0,0 +1,193 @@
1
+ ---
2
+ name: build
3
+ description: >
4
+ Build one approved issue end to end: implement it, review it with a sized panel
5
+ of agents, run the repository's own checks, open the pull request, and attach
6
+ the review and criteria reports as evidence. Use when a work item exists for an
7
+ issue and you want it built. Invoke with OUTERLAYER_WORK=<number> claude "/build".
8
+ disable-model-invocation: true
9
+ ---
10
+
11
+ Build one issue into an evidenced pull request. `/spec` writes and gets
12
+ approval for the issue. This skill builds it. It never writes the spec.
13
+
14
+ This file holds the phase order and the rules that bind every phase. Each
15
+ phase reads its own reference on entry, never before:
16
+
17
+ - [references/review-loop.md](references/review-loop.md): the review phase.
18
+ - [references/review-briefs.md](references/review-briefs.md): the reviewer, refuter and mutation briefs.
19
+ - [references/agent-briefs.md](references/agent-briefs.md): the briefs for every other agent.
20
+ - [references/release.md](references/release.md): sync, pull request, CI.
21
+ - [references/criteria-judge.md](references/criteria-judge.md): the brief that judges each criterion.
22
+ - [references/evidence.md](references/evidence.md): manual test and evidence.
23
+
24
+ Paste a brief verbatim into the agent that runs it. Add only the run's own
25
+ values, such as paths, the branch and the item number.
26
+
27
+ Every text a person will read follows the writing rule in the repository's
28
+ instructions file, if it has one. Otherwise write for a reader who was not in
29
+ the session: short sentences, the outcome first, and plain words. Paste that
30
+ rule into every agent that writes prose.
31
+
32
+ ## Arguments
33
+
34
+ `/build` takes no issue number. It builds the work item named by
35
+ `OUTERLAYER_WORK`, the item's number. The launch is:
36
+
37
+ OUTERLAYER_WORK=<number> claude "/build"
38
+
39
+ The item exists because someone ran `outerlayer work build --issue <reference>`.
40
+ Given a request instead of a work item, stop and say:
41
+
42
+ Spec it first: /spec <the request as given>
43
+
44
+ Optional flag: `--unattended` means no person is present and nothing wakes
45
+ the session after a turn ends. Spawn every agent in the foreground and wait
46
+ for it in the same turn. Run the gate in the foreground with the maximum
47
+ timeout. End the turn only at the final report.
48
+
49
+ ## State rules
50
+
51
+ These bind the orchestrator in every phase. Rule 5 also binds every spawned agent.
52
+
53
+ 1. **Never end a turn waiting on nothing.** End a turn while work is in
54
+ flight only when something live will wake the session: a spawned agent or
55
+ a background command. Before saying "waiting on X", confirm X is live.
56
+ Work that is not live did not happen: redo it once.
57
+ 2. **One redo per piece of work.** A second failure of the same work means
58
+ the orchestrator takes it over. It reads what the agent left, finishes the
59
+ brief itself in the foreground, and names the takeover in its status line.
60
+ If the gate still cannot go green, escalate.
61
+ 3. **The gate runs in the foreground, one heavy command at a time.** The gate
62
+ is the repository's own pre-push checks and tests, as its instructions
63
+ file names them. Never run it in the background. Never start a suite while
64
+ an agent runs one. Never bypass a hook: no `--no-verify`, no skipped hook,
65
+ even when a hook fails. Fix what the hook reports.
66
+ 4. **Checkpoint and re-orient.** Before phase 1, write a state file at
67
+ `"$(git rev-parse --absolute-git-dir)/build-state.json"`, replacing any
68
+ earlier one. Record the phase, the review round, the spawned agents, the
69
+ agent count, the issue reference, the pull request number, the branch, and
70
+ from phase 4 the tier verdict. Overwrite it at every phase change. On
71
+ entering a phase, after any wake, and after any compaction, read the state
72
+ file, `git status`, the branch, the live agent list, and the issue or pull
73
+ request when the phase uses them. Live sources win over the file. Treat the
74
+ file as absent when its branch does not match the current branch.
75
+ When you overwrite the file, keep its `criteria` entry: phase 2 writes it
76
+ and the judge reads it.
77
+ Beside it, create `"$(git rev-parse --absolute-git-dir)/build-reports/"`,
78
+ emptied of any earlier run's files. Every spawned agent writes its full
79
+ report there and returns a summary of at most ten lines. Give each agent
80
+ the absolute path. Reports never land in the working tree, so review never
81
+ reads them as product code. Read summaries and pass file paths, never
82
+ report bodies. Delete both at the end of phase 6.
83
+ Once a pull request exists, read `outerlayer work threads --item
84
+ "$OUTERLAYER_WORK" --json`. Each thread with `waitingOn` of `agent` holds a
85
+ person's comment. Answer each before the run ends, as the `amend` skill does.
86
+ 5. **A session never records a pass or fail on an artifact.** It emits
87
+ artifacts. A person decides pass or fail.
88
+ 6. **Observed status only.** Every status claim in the final report (a gate
89
+ result, a CI conclusion, mergeability) quotes command output observed
90
+ after the last change it describes. No quoted output means unknown, never
91
+ green. Quoted output goes in the report to the user. Pull request bodies
92
+ carry curated links, never raw output.
93
+ 7. **Files no issue.** The build files no issue anywhere, and no agent does.
94
+ A defect outside the change, or a follow-up the fix pass deferred, becomes a
95
+ candidate issue in a report: a title, a body and a target repository. The
96
+ final report lists every candidate for the user to file with `/spec`.
97
+ 8. **Reader pass.** No text reaches the git host as the session wrote it.
98
+ The pull request title and body, and every comment, go through one cold
99
+ rewrite agent. Its only inputs are the draft and the writing rule. It
100
+ returns the rewrite and the terms it cut or explained. Check that every
101
+ path, identifier, number, criterion id and link survived, and restore any
102
+ that did not.
103
+ 9. **Run budget.** Spawn at most sixty agents in a run. Keep the count in the
104
+ state file. Past three quarters, say where the count stands in the status
105
+ line. A run that would exceed the budget stops and lists what is left
106
+ unverified. The review phase also holds a limit on agents in flight.
107
+
108
+ ## Phases
109
+
110
+ Run them in order.
111
+
112
+ 1. **Intake.** Read `OUTERLAYER_WORK`. If it is missing or does not parse,
113
+ stop and hand back:
114
+
115
+ outerlayer work build --issue <reference> # once; prints the item number
116
+ OUTERLAYER_WORK=<number> claude "/build"
117
+
118
+ Run `outerlayer work status --item "$OUTERLAYER_WORK" --json` and read the
119
+ item's tracker key. It is a GitHub `#n`, or a Linear or Jira key. Read the
120
+ issue through whatever tracker tool the team has. The issue is the source
121
+ of truth for the whole run, and every later agent is pointed at it.
122
+ Restate the task in one or two sentences and confirm the issue is still
123
+ open and unclaimed. If the item shows a live claim by another runner that
124
+ has not expired, stop and name it.
125
+ Phase 2 finds the issue's criteria. If it finds some, they are the
126
+ definition of done, and each declares a proof form or a test. If it finds
127
+ none, build from the issue's description. An issue without criteria is
128
+ built, never sent back. Then no judge runs.
129
+ Ask nothing from here until the final report. An agent that needs a ruling
130
+ states its assumption in its report and continues. The final report lists
131
+ every assumption.
132
+ 2. **Read the criteria.** Record the issue's criteria on the item, so the
133
+ Criteria tab, the item's status and the judge hold the pull request to one
134
+ list. Save the issue body to a file and run:
135
+
136
+ node .outerlayer/skills/build/scripts/record-criteria.mjs --item "$OUTERLAYER_WORK" --issue-file <body file>
137
+
138
+ The script reads two shapes, in this order:
139
+
140
+ 1. An `## Acceptance criteria` section whose items each start with an id in
141
+ backticks. The id and the text after it are kept as written. A
142
+ `(proof: <kind>)` on the line sets `proof`. Otherwise `proof` is `null`.
143
+ 2. A heading or line that reads `Acceptance`, `Acceptance criteria` or
144
+ `Done when`, in any case, with or without a colon, followed by a bullet
145
+ list. Each bullet is one criterion. Its text is the bullet's text. Its id
146
+ is `AC-<item number>-NN`, numbered from `01` in list order. `proof` is
147
+ `null`. Never infer a proof kind from the words.
148
+
149
+ It writes the list to
150
+ `"$(git rev-parse --absolute-git-dir)/build-reports/criteria.json"` and runs
151
+ `outerlayer emit criteria` on it. The gateway decides whether a list may be
152
+ recorded. A 409 means the item already has a recorded list. That list
153
+ stands, and nothing is recorded. No command returns the item's list, so the
154
+ build cannot read it: it holds the work to the issue's reading of the criteria and says
155
+ so in the final report. Any other
156
+ refusal is written to the state file with its message, and the build
157
+ continues. The script writes the outcome under `criteria` in the state
158
+ file: `source` (`issue-ids`, `issue-bullets`, `item` or `none`), `recorded`,
159
+ `count` and `list`. On a 409, `list` is empty, and the issue's reading of the
160
+ criteria goes under `issueReading`. If neither shape is found, nothing is recorded and the
161
+ build continues at the small tier, as for any issue without criteria. Change
162
+ nothing in the issue. A criterion you think is wrong becomes a note in the
163
+ final report, never an edit. The intake summary names how many criteria were
164
+ recorded and from which shape, or says that none were found.
165
+ 3. **Implement.** Branch `feat/<slug>` from the default branch. Spawn the
166
+ implementer with the Implementer brief from
167
+ [references/agent-briefs.md](references/agent-briefs.md), the issue, the
168
+ branch and the instructions-file pointer. The phase ends when the
169
+ repository's own pre-push checks and tests pass on the branch. Review
170
+ never opens on a red branch. A branch that cannot go green escalates.
171
+ 4. **Review.** Read [references/review-loop.md](references/review-loop.md) on
172
+ entry and run it. It sizes the panel to the finished diff: one generalist
173
+ for a small diff, the generalist and one focused reviewer for a medium
174
+ one, the generalist and up to three for a large or risky one. Findings are
175
+ merged into classes, checked by a refuter, fixed, and rechecked once. Then
176
+ write the review report as an HTML file and emit it:
177
+
178
+ outerlayer emit artifact <report>.html --for code-review-ran --caption "Review of the change and what it found"
179
+
180
+ 5. **Sync and pull request.** Read [references/release.md](references/release.md)
181
+ on entry and run it. A release agent rebases, gates and pushes. Open the
182
+ pull request. Then run `outerlayer work pr <number>` so the item links the
183
+ pull request. When the issue has criteria, a fresh agent judges them with
184
+ [references/criteria-judge.md](references/criteria-judge.md), and the
185
+ judge's HTML report is emitted:
186
+
187
+ outerlayer emit artifact <judge-report>.html --for acceptance-criteria --caption "Each criterion judged against its test"
188
+
189
+ Watch CI until every check has concluded.
190
+ 6. **Manual test and evidence.** Read [references/evidence.md](references/evidence.md)
191
+ when the pull request opens, and run it. Capture the proof each criterion
192
+ declares, emit it, then end the run with the final report. The report
193
+ lists every candidate issue and every assumption.
@@ -0,0 +1,243 @@
1
+ # Briefs for every agent outside the review panel
2
+
3
+ ## Contents
4
+
5
+ - How to use this file
6
+ - Shared rules
7
+ - Implementer brief
8
+ - Merger brief
9
+ - Fix brief
10
+ - Release brief: sync and push
11
+ - Release brief: CI and mergeability
12
+ - Evidence brief
13
+
14
+ ## How to use this file
15
+
16
+ SKILL.md says which agent runs in which phase. This file says what each one
17
+ is told. The panel's reviewer, refuter and mutation briefs are in
18
+ [review-briefs.md](review-briefs.md). Paste every brief verbatim. The
19
+ orchestrator adds only the run's own values, such as file paths, the branch,
20
+ the issue reference and the pull request number, and may append a short focus
21
+ addendum that adds emphasis. It never edits or trims a brief.
22
+
23
+ Each brief names the report file the agent writes. The agent writes the full
24
+ report there and returns a summary of at most ten lines.
25
+
26
+ ## Shared rules
27
+
28
+ Paste this block at the top of every brief in this file.
29
+
30
+ Read the repository's instructions file first and hold your work to it.
31
+ Write every text a person will read, such as reports, commit messages and
32
+ candidate issue bodies, in short sentences with plain words, leading with the
33
+ outcome.
34
+
35
+ Never create an issue in any tracker. Work that belongs in its own issue
36
+ becomes a candidate issue in your report: a title, a body and the target
37
+ repository. The user decides whether to file it.
38
+
39
+ Never ask the user a question. If you need a ruling, state your assumption in
40
+ your report and continue.
41
+
42
+ Run every long command, such as the full gate, the mutation tool or the
43
+ integration suite, in the foreground with the maximum timeout. Never
44
+ background one, and never end a turn waiting on a notification.
45
+
46
+ Every agent in a run shares one memory cap. Run only the test files for the
47
+ code you change or check, one at a time. Run a whole package's suite, a
48
+ coverage run, a repository-wide typecheck or the full gate only where this
49
+ brief names it. Two of those at once overrun the cap, and the kernel kills
50
+ test workers that then look like flaky tests.
51
+
52
+ Put scratch copies and extra worktrees under the system temp directory.
53
+
54
+ Never bypass a hook. Do not use `--no-verify`, skip a hook, or disable one,
55
+ even when it fails. Fix what the hook reports.
56
+
57
+ Never change the identity a command runs under. Never unset or override an
58
+ `OUTERLAYER_*` variable or the key a command would use, and never rerun a
59
+ refused command as someone else. A refusal is a result: write it in your
60
+ report with the command, and leave that step undone.
61
+
62
+ A session never records a pass or fail on an artifact.
63
+
64
+ Write your full report to the path this brief names. Then return at most ten
65
+ lines. Anything you leave out of the file is lost to the next stage.
66
+
67
+ ## Implementer brief
68
+
69
+ The shared rules apply. Your inputs are the issue, the branch and the
70
+ instructions-file pointer.
71
+
72
+ The issue is the definition of done. When it has criteria, each one is a
73
+ requirement. Make every test cite the criterion id it proves, in the form the
74
+ team uses.
75
+
76
+ Write the first commit red, with no implementation. It holds one test that
77
+ drives the real chain, from input through storage to read-back, with nothing
78
+ mocked at the store boundary, plus one test per criterion written from the
79
+ criterion's own words. Read every failure. Each test must fail because the
80
+ behavior is absent, not because of a typo or a missing import. Then implement
81
+ until green. When the issue has no criteria, write tests from its description
82
+ in the same order.
83
+
84
+ Version any new database migration with the current UTC time, never a round or
85
+ guessed one. Two open branches that guess the same time both pass their own
86
+ checks, and the second to merge breaks the build for everyone.
87
+
88
+ While iterating, run only the tests of the packages the diff touches. Never
89
+ run the full gate first to see where things stand. When the last change is in,
90
+ sync with the base branch and run the repository's own pre-push checks and
91
+ tests, as its instructions file names them, in the foreground, once. You are
92
+ not done until they pass. If they cannot be made to pass, stop and escalate.
93
+ Never hand a red branch to review.
94
+
95
+ Do not run a mutation tool here. A mutation checker runs beside the review.
96
+
97
+ A field that crosses package boundaries needs an assertion at every seam it
98
+ crosses: writer, relay and reader.
99
+
100
+ A defect that also reproduces on the base branch is out of scope. It is a
101
+ candidate issue, never a fix here.
102
+
103
+ Close your report with a "Rules that misled you" section. List every sentence
104
+ in the instructions file or a skill you had to work around: the file path, the
105
+ line, the sentence quoted verbatim, and `broken` when its instruction fails or
106
+ `wrong` when it says the wrong thing. Write "none" when there were none.
107
+
108
+ ## Merger brief
109
+
110
+ The shared rules apply. Your inputs are the round's report files.
111
+
112
+ Merge the findings into classes. A class is a claim about a pattern, not a
113
+ single site. For each class, build an inventory of every instance in the
114
+ repository, checked with a search. The reports name the instances they
115
+ happened to see. You name the rest.
116
+
117
+ Take the union, never a vote. A finding one reviewer raised and four missed
118
+ survives to verification exactly as one four raised does. Agreement is
119
+ telemetry, not evidence.
120
+
121
+ A finding without its reproduction is a note, not a class. Collect notes from
122
+ every report into a closing "Notes" section and never give one a class id.
123
+
124
+ Write `build-reports/round-<n>-merged.md` with one section per class: the
125
+ claim, its severity and kind, the full inventory, which reviewers found it,
126
+ and the rule citation when a report carried one. Two reports citing the same
127
+ sentence for one class are one citation. Ids are `r<round>-c<k>` for a class,
128
+ `r<round>-n<k>` for a note, `r<round>-i<k>` for a candidate issue and
129
+ `r<round>-x<k>` for a rule a report listed, each `k` counting from 1 within its
130
+ kind. A listed rule is its own entry with its citation and no severity.
131
+
132
+ Close the file with an area table: the classes grouped by the files in their
133
+ inventories, two classes in one area when any file appears in both. Return the
134
+ class table only: id, severity, kind, instance count, reviewers, rule cited or
135
+ none, and fix group.
136
+
137
+ ## Fix brief
138
+
139
+ The shared rules apply. Your input is the list of refuter verdict files, and
140
+ the mutation report when the round had one. Read them. Nobody pastes the
141
+ findings to you.
142
+
143
+ Fix classes, not instances. For each class, repair every instance in its
144
+ inventory and prove nothing is left with a search whose empty result you
145
+ quote. Commit each refuter's failing test together with the fix that turns it
146
+ green.
147
+
148
+ Your brief may name a fix group: a worktree path, a group id and only that
149
+ group's verdict files. Then work in that worktree and touch only the files in
150
+ your classes' inventories and the tests beside them. A fix that needs another
151
+ file is reported as not fixed with the reason `crosses group`. In group mode,
152
+ finish with every verification test in your group green, run one file at a
153
+ time. Run no typecheck, package suite or gate, because other groups run beside
154
+ you.
155
+
156
+ A confirmed finding that also reproduces on the base branch is a candidate
157
+ issue, unless its repair stays inside lines this branch already changes.
158
+
159
+ With no group named, end the pass with the repository's own checks green
160
+ again, the verification tests passing, and the comments and docs next to
161
+ changed behavior brought back in line.
162
+
163
+ Write `build-reports/round-<n>-fix.md`, or `round-<n>-fix-<group>.md` in
164
+ group mode. Name every briefed item and its outcome: fixed, with the green
165
+ test and the hunks, or not fixed, with the reason. An item the report does not
166
+ name counts as not fixed. A green suite is not a report.
167
+
168
+ ## Release brief: sync and push
169
+
170
+ The shared rules apply. Your inputs are the state file path, the branch and
171
+ the instructions-file pointer. You run no write command against the tracker
172
+ or the git host beyond the push.
173
+
174
+ Fetch the base branch, then rebase the branch onto it. Merge instead once the
175
+ branch is pushed and someone else may hold it. Resolve every conflict on the
176
+ branch, never in a web editor.
177
+
178
+ A migration this branch adds whose version is at or below the newest one on
179
+ the base branch is renamed to the current UTC time, with every reference to
180
+ its file updated.
181
+
182
+ Run the repository's own pre-push checks and tests, as its instructions file
183
+ names them, in the foreground with the maximum timeout. Leave any mutation
184
+ tool off: the review ran it over these lines already. Then run any check the
185
+ repository runs only in CI, when the diff touches its domain, each in the
186
+ foreground, and quote every result in your report. A check this machine cannot
187
+ run is named in the report with the reason. A check that fails is a red gate,
188
+ and you never push over it. Then push. The hooks run, and you never bypass
189
+ them.
190
+
191
+ When your brief says this is a resync of a branch already pushed and reviewed,
192
+ the surface is the merge's diff, not the branch's. Run the extra checks only
193
+ when that diff reaches their domain, and say which ones the resync did not
194
+ need and why.
195
+
196
+ Write `build-reports/release.md` naming every conflict you resolved and every
197
+ file the sync moved that the review had already cleared. The orchestrator acts
198
+ on that second list, so do not omit a file because the move looked harmless.
199
+
200
+ ## Release brief: CI and mergeability
201
+
202
+ The shared rules apply. Your inputs are the branch and the pull request
203
+ number.
204
+
205
+ Watch the pull request's checks until they reach a conclusive state, by
206
+ watching them or by a bounded poll. Any check that waits on results recorded
207
+ after the session ends stays pending for the whole run. Leave it out of the
208
+ wait. CI is conclusive when every other check has finished. Right after a pull
209
+ request opens, "no checks reported" is ambiguous, not green. Query again over
210
+ a short window before believing it.
211
+
212
+ Confirm mergeability, and query again while it reads unknown.
213
+
214
+ Write the quoted command output to `build-reports/release.md`. Return at most
215
+ ten lines naming the CI conclusion, the mergeability and every failed check.
216
+
217
+ ## Evidence brief
218
+
219
+ The shared rules apply, and the emitting-evidence skill governs the emit
220
+ mechanics. Your inputs are the issue, the pull request number, the criteria
221
+ that declare a proof form, and the instructions-file pointer.
222
+
223
+ Run the app and drive the feature by hand. A screenshot or video is taken in a
224
+ real browser against the running application, reached through its own routes
225
+ by a signed-in user, inside its real shell and theme. Never capture a test
226
+ render, a jsdom dump or a standalone file of a component. It shows neither what
227
+ a user sees nor that the page works. Seed what the page needs through the
228
+ application's own paths. When a state cannot be reached that way, the criterion
229
+ is not evidenced, and `evidence.md` says why. Before emitting, look at each
230
+ capture: it shows the application's navigation and real rows.
231
+
232
+ For each criterion, put the capture in a script under
233
+ `build-reports/evidence/` and run the script. A later fix can invalidate what
234
+ you captured, and regenerating it must be one command. The scripts go when the
235
+ run's report directory goes. The emitted artifacts outlive the run.
236
+
237
+ Emit each artifact with `outerlayer emit artifact ... --for <criterion id>
238
+ --pr <number>`. Always pass `--pr`. Regenerate with `--replaces` any evidence
239
+ the tree has since invalidated. Then run `outerlayer sync`.
240
+
241
+ Write `build-reports/evidence.md` naming every criterion with its artifact, or
242
+ with the reason none could be produced. A criterion you could not evidence is
243
+ a result, not a gap to leave silent.
@@ -0,0 +1,91 @@
1
+ # Criteria judge
2
+
3
+ Read this when the release phase spawns you. You are a fresh-context agent,
4
+ never the implementer. You judge whether each criterion in the issue is
5
+ actually proven by the test that cites it, not merely joined to one by id. For
6
+ a criterion that declares an artifact proof, you note that an artifact is
7
+ needed.
8
+
9
+ ## Contents
10
+
11
+ - Inputs
12
+ - Steps
13
+ - Rubric
14
+ - Output
15
+ - Guardrails
16
+
17
+ ## Inputs
18
+
19
+ The orchestrator gives you the worktree path (run every command there), the
20
+ base ref, the pull request number, the path of the build's state file, and the
21
+ path to write your report to. On a large issue it gives you one group of criteria instead.
22
+ Judge only that group. Other judges handle the rest.
23
+
24
+ ## Steps
25
+
26
+ 1. Read the criteria from the build's state file, under `criteria.list`. They
27
+ are the item's recorded list. If the state file lacks `criteria`, read the
28
+ `criteria` array in `criteria.json` in the build-reports directory. When
29
+ `criteria.source` is `item`, the item already had a recorded list that no
30
+ command returns, so `list` is empty. Judge `criteria.issueReading` instead,
31
+ and say in the report that it is the issue's reading, whose ids may differ
32
+ from the item's recorded list. Each has an id
33
+ and a text, and the text is usually a Given, When and Then. Judge every criterion you were given,
34
+ including one no line of the diff touched. A criterion with `proof: null`
35
+ is proven by a citing test or any artifact bound to its id.
36
+ 2. For each criterion, find its citing tests. Search the test files for its
37
+ id. A raw text match is not yet a citation. The convention is a comment such
38
+ as `// proves <id>` directly above the test, or a test that names the id.
39
+ An id that appears only as sample data a test uses as input, never as a proof
40
+ claim, is not a citation. Skip that hit.
41
+ 3. A criterion that declares `(proof: screenshot)`, `(proof: video)`,
42
+ `(proof: report)`, `(proof: log)` or `(proof: file)` is `not-judged`, with
43
+ the reason "artifact proof; an artifact bound to this id must be attached to
44
+ the pull request, which this judge cannot see from the worktree".
45
+ 4. A criterion with no proof annotation and no citing test is `not-satisfied`,
46
+ with the reason "no test cites this criterion".
47
+ 5. Otherwise, read every citing test's block, the test itself plus its
48
+ enclosing group names, and judge it against the rubric.
49
+
50
+ ## Rubric
51
+
52
+ A criterion is satisfied only when a test asserts the observable outcome its
53
+ Then clause states, under the Given and When conditions it states, with an
54
+ assertion that would fail if that outcome did not hold.
55
+
56
+ It is not satisfied when the test only cites the id, asserts something weaker
57
+ than the stated outcome, mocks away the very behavior the criterion is about,
58
+ or has no assertion bearing on the outcome, even when it cites the id.
59
+
60
+ When unsure, record `not-satisfied` and state the doubt in the reason. A silent
61
+ pass is worse than a wrong-looking fail a person can overrule.
62
+
63
+ ## Output
64
+
65
+ Write an HTML report to the path you were given. It has one row per criterion:
66
+ the id, the verdict (`satisfied`, `not-satisfied` or `not-judged`), and a
67
+ reason of one to three sentences, with no preamble, that quotes the decisive
68
+ assertion or names what is missing. Put a summary line at the top with the
69
+ count of each verdict. The report must open in a browser as a single file with
70
+ no external assets.
71
+
72
+ Then print a table of id, verdict and the first line of the reason, and end
73
+ with exactly one line: `result: pass` or `result: fail`. Fail when any
74
+ criterion is `not-satisfied`.
75
+
76
+ ## Guardrails
77
+
78
+ Judge every criterion at the same depth. Read the cited test's assertions and
79
+ compare them to the criterion's own Then clause, every time. A well-worded
80
+ citing comment that restates the criterion is a hint about what to look for,
81
+ never a substitute for reading the assertion beneath it. A citation can be
82
+ wrong or stale. Never shortcut to "many others looked fine, so the rest
83
+ probably are."
84
+
85
+ A long run is expected, and is no reason to sample. Work through the list in
86
+ order, and rewrite the report after each batch so a run that stops partway
87
+ still leaves a readable report.
88
+
89
+ Edit no file but the report. Do not fix a test, do not touch the issue, and do
90
+ not run `outerlayer emit`. That step belongs to the orchestrator once it holds
91
+ the report.
@@ -0,0 +1,42 @@
1
+ # Phase 6: manual test and evidence
2
+
3
+ Read this when the pull request opens. The evidence agent starts then, in the
4
+ same wave as the CI watch. Everything after it in this file runs once CI is
5
+ conclusive and the evidence agent has returned. This is the last phase, and it
6
+ ends the run: it deletes the state file and the report directory. The evidence
7
+ brief is in [agent-briefs.md](agent-briefs.md). Read every value from an
8
+ earlier phase from the state file.
9
+
10
+ ## The phase
11
+
12
+ The orchestrator does not run the app, drive a browser or emit an artifact
13
+ itself. Spawn one evidence agent in a fresh context on the Evidence brief, with
14
+ the issue, the pull request number, the criteria that declare a proof form, the
15
+ instructions-file pointer and the emitting-evidence skill.
16
+
17
+ A criterion that declares `proof: screenshot` needs a screenshot bound to its
18
+ id. One that declares `proof: video` needs a video. A screenshot never
19
+ satisfies `proof: video`. A criterion with no proof form is proven by its
20
+ test, and needs no artifact.
21
+
22
+ When a commit lands on the branch after the evidence agent captured, decide for
23
+ each capture from that commit's diff. A capture is stale when the diff touches
24
+ a file the capture's subject runs through: the feature's own files, the shared
25
+ code they import, a migration or schema behind them, or a test the capture is a
26
+ log of. A sync that brings in only files the feature never reaches leaves the
27
+ capture valid. Spawn the evidence agent again with the stale list. It reruns
28
+ their scripts and emits each with `--replaces`. The final report names the
29
+ commit and every capture with its call, kept or regenerated. A commit that
30
+ lands after this call reopens it. Never drive the browser yourself.
31
+
32
+ Re-read the pull request's evidence comment. Resolve or escalate every row that
33
+ needs attention. A row is resolved when a re-read shows it, not when the fixing
34
+ command exits. If no evidence comment exists, report that observation with the
35
+ time, never a belief.
36
+
37
+ End the report to the user with every candidate issue from every stage: title,
38
+ one line and target repository. Or write "no candidate issues". Hand them over.
39
+ Do not ask which to file, and do not file any. The user takes the ones worth
40
+ filing to `/spec`. List every assumption an agent stated.
41
+
42
+ Then delete the state file and the `build-reports/` directory.
@@ -0,0 +1,74 @@
1
+ # Phase 5: sync and pull request
2
+
3
+ Run this once the review loop reaches zero confirmed blocking findings. The
4
+ release briefs are in [agent-briefs.md](agent-briefs.md). Read every value
5
+ from an earlier phase, such as the tier verdict or the branch, from the state
6
+ file, not from the transcript.
7
+
8
+ ## Sync and push
9
+
10
+ The orchestrator does not sync, run the gate, push or watch CI itself. Spawn
11
+ one release agent in a fresh context on the "sync and push" brief, with the
12
+ state file path, the branch and the instructions-file pointer. A check that
13
+ fails there has the standing of a red gate and returns to the review phase's
14
+ remaining budget. If its report names files the sync moved that the review had
15
+ already cleared, run one check round on the rebased diff before going on.
16
+
17
+ ## Open the pull request
18
+
19
+ Open a scoped pull request that closes the issue. The body carries the
20
+ problem, the change, how to verify it, and, when the change alters what a user
21
+ sees or does, a link to each documentation file added or changed. It carries
22
+ no telemetry, round tables or reviewer roles. Those go in the final report.
23
+
24
+ Run the reader pass on the title and body before opening, and after any later
25
+ edit. Add an "Unverified, non-blocking" section to the review report when
26
+ classes were left unrefuted.
27
+
28
+ The moment the pull request exists, declare it to the work item from this
29
+ session:
30
+
31
+ outerlayer work pr <number>
32
+
33
+ A pull request belongs to a work item only through a declaration from a
34
+ session on that item. Nothing else links it, not the issue it closes and not
35
+ its branch. Without the declaration the item shows no pull request, no
36
+ evaluation and no evidence. Check that the command reports the link as created
37
+ or already present before going on. A CLI that does not know the command needs
38
+ refreshing, and the final report names that as a blocker.
39
+
40
+ ## Judge the criteria
41
+
42
+ When the issue has criteria, spawn one fresh agent on
43
+ [criteria-judge.md](criteria-judge.md), with the worktree path, the base ref,
44
+ the pull request number, the path of the build's state file, and a report
45
+ path under `build-reports/`. Above forty
46
+ criteria, spawn one judge per group of about forty in the same wave, each with
47
+ its own report path, and merge their results by concatenation.
48
+
49
+ The judge writes an HTML report with one verdict per criterion. Emit it:
50
+
51
+ outerlayer emit artifact <judge-report>.html --for acceptance-criteria --caption "Each criterion judged against the test that cites it"
52
+
53
+ Then run `outerlayer sync`. List every unsatisfied criterion in the final
54
+ report. A criterion that declares a proof form such as a screenshot or video is
55
+ proven only by an artifact bound to it on the pull request. The judge cannot
56
+ see that, so confirm each has an artifact of the declared kind, and emit any
57
+ that is missing in the evidence phase. When the issue has no criteria, no judge
58
+ runs and no judge report is emitted.
59
+
60
+ ## Watch CI
61
+
62
+ With the pull request open, spawn two agents in the same wave. The release
63
+ agent runs again on the "CI and mergeability" brief with the pull request
64
+ number. The evidence agent starts as [evidence.md](evidence.md) describes. The
65
+ manual test needs only the pull request number, and CI never reads the
66
+ captures, so the two waits overlap.
67
+
68
+ Act on the release agent's return. A failed check has the standing of a red
69
+ gate: fix it within the review phase's remaining budget or escalate. Never
70
+ carry it into phase 6. A persistent unknown mergeability escalates. A
71
+ conflict means one more sync through the release agent, told it is a resync so
72
+ it gates the merge's diff and not the branch's. A second conflict escalates.
73
+ Any fix or sync that lands after the evidence agent captured goes through
74
+ phase 6's late-commit rule.