@outerlayer/cli 0.1.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (143) hide show
  1. package/CHANGELOG.md +134 -0
  2. package/README.md +238 -77
  3. package/dist/agent-setup-IFH3FB5X.js +197 -0
  4. package/dist/build-TZUUTRGH.js +9 -0
  5. package/dist/build-info.json +1 -1
  6. package/dist/check-WQP2CZDW.js +94 -0
  7. package/dist/chunk-2E54Z3P5.js +232 -0
  8. package/dist/chunk-3TT7RQYB.js +55 -0
  9. package/dist/chunk-3YPKXE5L.js +1683 -0
  10. package/dist/{chunk-HNCCD2IX.js → chunk-5DLBAXY5.js} +209 -61
  11. package/dist/{chunk-R5KBGVII.js → chunk-5QRP3MVS.js} +30 -14
  12. package/dist/chunk-5Y3QQRUH.js +300 -0
  13. package/dist/chunk-6Y63SXFO.js +383 -0
  14. package/dist/chunk-774HEQL6.js +178 -0
  15. package/dist/chunk-7A6RVXPZ.js +2233 -0
  16. package/dist/chunk-7YIOOFVJ.js +71 -0
  17. package/dist/chunk-ABIWDUMS.js +104 -0
  18. package/dist/chunk-AUFG23AA.js +534 -0
  19. package/dist/chunk-B2G7JHHB.js +42 -0
  20. package/dist/chunk-BCJNHMZT.js +34 -0
  21. package/dist/chunk-BCLJSQCV.js +56 -0
  22. package/dist/chunk-BFESLKPP.js +61 -0
  23. package/dist/chunk-BJ3KTMDH.js +63 -0
  24. package/dist/chunk-BKO6JEZI.js +47 -0
  25. package/dist/chunk-BOLTI6LR.js +37 -0
  26. package/dist/chunk-CBPPJSAR.js +89 -0
  27. package/dist/chunk-CBZB6XY6.js +102 -0
  28. package/dist/{chunk-RCQXYLMO.js → chunk-DVDEBNQJ.js} +18 -18
  29. package/dist/{work-pr-cmd-GOYSPEN2.js → chunk-DVYHCMB2.js} +10 -25
  30. package/dist/{chunk-TFUIDMOB.js → chunk-EABW6AJQ.js} +12 -1
  31. package/dist/{chunk-KFYJV2ZG.js → chunk-EMY4I27X.js} +1 -1
  32. package/dist/{chunk-NTTPJV35.js → chunk-F47JBIAW.js} +3 -1
  33. package/dist/chunk-F6GFZXZ3.js +60 -0
  34. package/dist/{chunk-YOOSBOKS.js → chunk-F7CU5ABH.js} +1 -1
  35. package/dist/chunk-FAJSS2WN.js +1357 -0
  36. package/dist/chunk-FS2VYGZT.js +1764 -0
  37. package/dist/chunk-FV2CCGXK.js +9 -0
  38. package/dist/{chunk-BQB3U2VD.js → chunk-G7FSV7JX.js} +5113 -3729
  39. package/dist/{emit-cmd-TWSYYEZI.js → chunk-H5NCNVDA.js} +68 -17
  40. package/dist/chunk-HE2D4EY5.js +83 -0
  41. package/dist/chunk-L2DBRKRA.js +94 -0
  42. package/dist/chunk-LF6MLJCY.js +38 -0
  43. package/dist/chunk-M5POMKKW.js +1051 -0
  44. package/dist/{mcp-install-cmd-FDQH6SEN.js → chunk-MQ3IPHIZ.js} +35 -18
  45. package/dist/{chunk-D77LS3UI.js → chunk-N5FOE5PS.js} +33 -4
  46. package/dist/chunk-N6LURUEF.js +64 -0
  47. package/dist/chunk-NYC54VBD.js +1084 -0
  48. package/dist/chunk-OGFGZQA3.js +23 -0
  49. package/dist/chunk-QOZKPLVJ.js +226 -0
  50. package/dist/chunk-RA3O54FC.js +30 -0
  51. package/dist/chunk-TBT347UY.js +41 -0
  52. package/dist/{chunk-3VQ4XXQA.js → chunk-UFFXNXLQ.js} +115 -71
  53. package/dist/chunk-USO2DKBD.js +421 -0
  54. package/dist/chunk-W4SQCOZU.js +206 -0
  55. package/dist/{chunk-A3WLZX2F.js → chunk-WIOZAJ2W.js} +15 -2
  56. package/dist/chunk-WO2BXCTQ.js +83 -0
  57. package/dist/chunk-WZBHBH3J.js +20 -0
  58. package/dist/{chunk-IWIUYLDR.js → chunk-XCCLFVXM.js} +118 -189
  59. package/dist/{context-materialize-J6CRQ7O7.js → chunk-XDDW4FRS.js} +134 -26
  60. package/dist/chunk-YYYXRJUW.js +101 -0
  61. package/dist/chunk-ZM2IMMYN.js +76 -0
  62. package/dist/chunk-ZMYPLFG3.js +71 -0
  63. package/dist/chunk-ZNA27WEV.js +470 -0
  64. package/dist/{cli-EFJIG2VT.js → cli-NHUXABYH.js} +597 -965
  65. package/dist/{paths-D2VGWWFI.js → cli-build-K56DK4DS.js} +1 -1
  66. package/dist/config-XVJYZ4IQ.js +9 -0
  67. package/dist/connect-cmd-OOG4TPU7.js +16 -0
  68. package/dist/context-adopt-UMA4O5NY.js +105 -0
  69. package/dist/context-materialize-G2FUFC72.js +19 -0
  70. package/dist/dist-RW4UPPC2.js +6 -0
  71. package/dist/docker-NEGDLU6D.js +7 -0
  72. package/dist/doctor-53F2PYY7.js +43 -0
  73. package/dist/doctor-UFVL4PZY.js +426 -0
  74. package/dist/{emit-artifact-cmd-KPPD3YLO.js → emit-artifact-cmd-KTYW2TN2.js} +25 -19
  75. package/dist/emit-cmd-DCERAU27.js +9 -0
  76. package/dist/emit-criteria-cmd-VAUUSAYD.js +166 -0
  77. package/dist/{emit-finding-cmd-VR5VEGXJ.js → emit-finding-cmd-UR4ID45Z.js} +35 -30
  78. package/dist/{emit-result-cmd-YOA5BJKL.js → emit-result-cmd-FDTUKGKE.js} +35 -61
  79. package/dist/exec-client-4XGXV3XN.js +8 -0
  80. package/dist/guest-init-GENRHNP7.js +8 -0
  81. package/dist/hook-fast-EYENPK6F.js +9 -0
  82. package/dist/{hook-wrap-fast-KXYNX3AD.js → hook-wrap-fast-NPLHJXFK.js} +2 -2
  83. package/dist/host-key-KDYZ5ECC.js +7 -0
  84. package/dist/import-capture-cmd-AVZ4VIVJ.js +68 -0
  85. package/dist/{import-ruler-cmd-7LH2QOMG.js → import-ruler-cmd-GUHL6IY5.js} +1 -1
  86. package/dist/index.js +4 -4
  87. package/dist/init-DXBCOAJK.js +36 -0
  88. package/dist/init-YEXK35CF.js +221 -0
  89. package/dist/init-cmd-NHR7PRSQ.js +135 -0
  90. package/dist/install-cmd-7WXOIFDO.js +55 -0
  91. package/dist/lima-XN65D7GN.js +55 -0
  92. package/dist/login-browser-7XGSV7MA.js +10 -0
  93. package/dist/{config-POF7DEQW.js → logout-cmd-QAUFDJ6J.js} +3 -1
  94. package/dist/{logs-TRNPQM42.js → logs-7RYUH5A6.js} +1 -1
  95. package/dist/loop-VZOLL4WQ.js +43 -0
  96. package/dist/machine-WSG52J75.js +35 -0
  97. package/dist/mcp-install-cmd-RLK4SO3N.js +9 -0
  98. package/dist/{mcp-serve-cmd-57EZZOTL.js → mcp-serve-cmd-JPTP5FMS.js} +20 -9
  99. package/dist/paths-OYKMVYJP.js +6 -0
  100. package/dist/{pidfile-PTW76F56.js → pidfile-UZRH774M.js} +2 -3
  101. package/dist/real-deps-SU24ZA2K.js +21 -0
  102. package/dist/relay-L76HDX72.js +46 -0
  103. package/dist/settings-J2652U5N.js +7 -0
  104. package/dist/starter-pack-LIZYMKYQ.js +10 -0
  105. package/dist/{status-JHVSZYC5.js → status-2UZKKNB7.js} +31 -12
  106. package/dist/{statusline-fast-3C5OXHDD.js → statusline-fast-SVTMBMD7.js} +3 -2
  107. package/dist/sync-cmd-VPPYCNNM.js +28 -0
  108. package/dist/version-BWM6VLDI.js +6 -0
  109. package/dist/{watch-NRLXJUST.js → watch-5C4ZBBGO.js} +20 -7
  110. package/dist/{work-claim-cmd-HXB7K27H.js → work-claim-cmd-LSO2B2EX.js} +34 -19
  111. package/dist/work-cmd-UOQO2AD4.js +17 -0
  112. package/dist/work-comment-cmd-WF3EOLAE.js +106 -0
  113. package/dist/{work-launch-YNNCKIH3.js → work-launch-YI4CJDAP.js} +2 -1
  114. package/dist/work-open-pr-cmd-4SSQZMYG.js +156 -0
  115. package/dist/work-pr-cmd-6AN6RRQJ.js +16 -0
  116. package/package.json +13 -3
  117. package/skill-pack/maintained/amend/SKILL.md +104 -0
  118. package/skill-pack/maintained/emitting-evidence/SKILL.md +99 -0
  119. package/skill-pack/maintained/emitting-evidence/references/agents-snippet.md +20 -0
  120. package/skill-pack/maintained/outerlayer/SKILL.md +55 -0
  121. package/skill-pack/maintained/reporting-findings/SKILL.md +148 -0
  122. package/skill-pack/template/build/SKILL.md +193 -0
  123. package/skill-pack/template/build/references/agent-briefs.md +243 -0
  124. package/skill-pack/template/build/references/criteria-judge.md +91 -0
  125. package/skill-pack/template/build/references/evidence.md +42 -0
  126. package/skill-pack/template/build/references/release.md +74 -0
  127. package/skill-pack/template/build/references/review-briefs.md +275 -0
  128. package/skill-pack/template/build/references/review-loop.md +158 -0
  129. package/skill-pack/template/build/scripts/record-criteria.mjs +235 -0
  130. package/skill-pack/template/spec/SKILL.md +84 -0
  131. package/skill-pack/template/writing-specs/SKILL.md +134 -0
  132. package/dist/chunk-JJP7YLMN.js +0 -25
  133. package/dist/chunk-KKEU6FFR.js +0 -589
  134. package/dist/chunk-OZ7C3XUE.js +0 -34
  135. package/dist/chunk-WQ6VGRGZ.js +0 -150
  136. package/dist/emit-commit-credit-cmd-656BRAQ4.js +0 -121
  137. package/dist/hook-fast-I2XTHDE6.js +0 -8
  138. package/dist/import-capture-cmd-EUMBGIV3.js +0 -176
  139. package/dist/init-PTBITAUO.js +0 -103
  140. package/dist/login-cmd-IRX6LZT7.js +0 -62
  141. package/dist/loop-BWE2BS6J.js +0 -646
  142. package/dist/sync-cmd-S7Q6DWU3.js +0 -15
  143. package/dist/work-cmd-ABP6XDBP.js +0 -16
@@ -0,0 +1,275 @@
1
+ # Reviewer, refuter and mutation briefs
2
+
3
+ ## Contents
4
+
5
+ - Tier composition
6
+ - Shared preamble
7
+ - Security reviewer
8
+ - Failure-path reviewer
9
+ - Test-strength reviewer
10
+ - Contracts reviewer
11
+ - Hygiene reviewer
12
+ - Generalist brief
13
+ - Refuter brief
14
+ - Mutation brief
15
+
16
+ Each reviewer's prompt is the shared preamble, then exactly one reviewer
17
+ block, verbatim, plus an optional focus addendum. The reviewers split the
18
+ search by procedure. Two reviewers with the same procedure find the same
19
+ things, so never assign one twice.
20
+
21
+ ## Tier composition
22
+
23
+ This file is the closed set of briefs the orchestrator can spawn from. The
24
+ tier verdict picks a subset of the five focused reviewers plus the
25
+ generalist. The orchestrator may append a short focus addendum that adds
26
+ emphasis. It may not edit, trim or replace a brief, and it may not tell a
27
+ reviewer to skip part of its territory. The generalist is on every panel. Its
28
+ addendum names the territory no focused reviewer covers that round. The
29
+ refuter and mutation briefs run on whatever the panel produces.
30
+
31
+ ## Shared preamble
32
+
33
+ Your brief names a bundle file: the diff, the issue, the criteria, the
34
+ changed files and the test files at head. Read it in full first. Open a
35
+ source file only where the bundle is not enough. Never re-derive the diff with
36
+ git.
37
+
38
+ Never ask the user a question. If you need a ruling, state your assumption in
39
+ your report and continue.
40
+
41
+ You are a cold adversarial reviewer. You know nothing about how this change
42
+ was built, and that is deliberate. Judge only the branch diff, the issue (its
43
+ criteria, when it has them, are the definition of done), and the
44
+ repository's instructions file, which you read first and hold the change to.
45
+ Your job is to break the change within your assigned territory.
46
+
47
+ Report only findings you have confirmed by reading the code or running
48
+ something. Drop any suspicion that does not survive checking. For each finding
49
+ give: severity (blocker, major, minor or nit); kind (behavior, security, data
50
+ integrity, proof strength, or hygiene); file and line; a concrete failure
51
+ scenario; its reproduction; and a recommended fix. Proof strength means a test
52
+ or check that can be silently gutted. Hygiene means comment truth, dead code
53
+ and copy polish.
54
+
55
+ The reproduction is a failing test you wrote, a command and its output, or the
56
+ exact input and the code path that mishandles it. A finding you can only
57
+ argue for is a note. List notes in a closing "Notes" section. Kind decides
58
+ what blocks the merge, so classify honestly: a hygiene label on a behavior bug
59
+ ships the bug. When a finding is one instance of a repeatable class, name the
60
+ class so the merge step can inventory it.
61
+
62
+ When a finding relates to a sentence in the instructions file or a skill,
63
+ cite it: the file path, the line, the sentence quoted verbatim, and one word
64
+ for the relation. Use `broken` when the rule's own instruction fails, such as
65
+ a path that no longer exists. Use `wrong` when the rule says the wrong thing.
66
+ Use `none` when the rule is sound and the change broke it. A rule you found
67
+ broken or wrong with no code defect behind it is its own finding. List it in a
68
+ closing "Rules" section with the same citation.
69
+
70
+ Stay inside the change. A defect whose repair lies outside the issue's goal,
71
+ or outside the files the diff touches and the code that calls them, is a
72
+ candidate issue in your report, however severe. The exception is a behavior
73
+ or security defect the diff itself introduced.
74
+
75
+ Do not modify any code. Run only the test files for the code you review, one
76
+ at a time. Never run a package's suite, a coverage run, a repository-wide
77
+ typecheck or the full gate. The other reviewers run beside you inside one
78
+ memory cap, and parallel suites get test workers killed that then look like
79
+ flaky tests. Put scratch copies under the system temp directory.
80
+
81
+ Never create an issue in any tracker. A defect that also reproduces on the
82
+ base branch goes in your report as a candidate issue: a title, a body and a
83
+ repository.
84
+
85
+ Write each finding in plain English, the way you would explain the bug aloud
86
+ to the person who wrote it: the failure first, short sentences, everyday words.
87
+
88
+ Write your full report to the path your brief names. Return a summary of at
89
+ most ten lines: counts by severity and by kind, the id and one-line title of
90
+ every blocker, and the report path.
91
+
92
+ ## Security reviewer
93
+
94
+ Trace every new read and write to the credential it runs under, in order.
95
+
96
+ - Ordering: is the caller's permission established before any privileged
97
+ read? A denied caller must trigger zero privileged calls, not
98
+ privileged-then-filtered.
99
+ - Parity: does the new surface reveal anything a sibling surface gates? An
100
+ unauthorized caller must see no more than on the closest existing surface.
101
+ - Tenancy or ownership: every query is scoped by the resolved request owner.
102
+ No identity comes from unscoped claims. Cross-owner reads are impossible by
103
+ construction.
104
+ - Disclosure: what reaches world-readable surfaces versus private ones? Names,
105
+ repository identities and internal URLs each need a reason to cross that
106
+ line.
107
+ - Injection: every value rendered into markdown, HTML, SQL or shell,
108
+ including values from stores users write to, is escaped, stripped or typed
109
+ at the boundary.
110
+
111
+ ## Failure-path reviewer
112
+
113
+ For every external interaction (network, subprocess, store), list the
114
+ complete failure taxonomy and walk each branch.
115
+
116
+ - Result-shaped failures: errors that arrive as a typed non-ok return, not a
117
+ throw. The classic bug is a loud thrown path and a silent typed path that
118
+ renders the healthy state.
119
+ - Ambiguous errors: one error shape with two meanings, such as a 404 that
120
+ means both "no access" and "does not exist". What tells them apart, and
121
+ what happens when the wrong meaning is assumed?
122
+ - Silent degrades: every catch block and fallback must show a degraded
123
+ state, log enough to diagnose, or both.
124
+ - Bounds: every loop over remote data has a cap, and the cap bounds calls,
125
+ not just collected items. Check exactly-at-cap, empty-at-boundary and depth
126
+ limits.
127
+ - Partial states: a failure after step 2 of 4. What is on disk, what is
128
+ rendered, and what does the next run see?
129
+
130
+ ## Test-strength reviewer
131
+
132
+ For every criterion, find its citing test and ask: what is the smallest
133
+ change to production code that still passes it?
134
+
135
+ - Fake-seam blindness: a fake client that returns fields whatever the query
136
+ says proves nothing about the query. Pin each data-feeding request with a
137
+ query-text assertion or a recorded request, or test it against the real
138
+ store.
139
+ - Structural pinning: behavior claimed absent ("X is never called for Y")
140
+ needs a present spy asserting zero calls. A missing method skips quietly.
141
+ - Seam correctness: business behavior is tested below the UI, UI contracts at
142
+ the component level, store-resident behavior against the real store.
143
+ - Assertion strength: exact shapes over permissive matchers. Error-path tests
144
+ assert the message or status a user would see, not only that something
145
+ threw.
146
+ - Mutation survivors: for each new conditional and object literal, flip or
147
+ empty it. If no test fails, write the assertion that would.
148
+
149
+ ## Contracts reviewer
150
+
151
+ The diff's edges are where it meets everything it did not change.
152
+
153
+ - Both sides of every shared contract: a constant, schema version or type used
154
+ across packages must have writer and reader agreeing after the change. Check
155
+ the consumers, not just the definition.
156
+ - Removed and renamed identifiers: search the whole repository, hidden
157
+ directories included, for every deleted or renamed export, prop, key and
158
+ environment variable. Check tests, docs and configuration.
159
+ - Generated surfaces: API specs, generated types and codegen outputs.
160
+ Regenerate and diff. Drift is a finding.
161
+ - Gates and floors: every check whose input the diff changes, such as
162
+ coverage thresholds, unused-export checks and numbering collisions with
163
+ files landed on the base branch since the branch was cut.
164
+ - Lifecycle wedges: state one component writes and another reads. Can a
165
+ non-canonical state persist where the writer never rewrites it and the
166
+ reader never accepts it?
167
+
168
+ ## Hygiene reviewer
169
+
170
+ Run each sweep across the repository, not only the diff. The diff tells you
171
+ which classes to sweep. The repository tells you every instance.
172
+
173
+ - Comment truth: every comment the diff adds, edits or makes stale must be
174
+ true of the code as it now is. Search for verbatim twins the diff missed.
175
+ - Comment policy: no issue numbers, no change narration and no phase names in
176
+ production comments.
177
+ - Dead code: branches the diff's own refactors made unreachable, and exports
178
+ nothing imports.
179
+ - Copy edge cases: plurals, and empty or degenerate values rendered into
180
+ user-visible strings, such as an empty code span or a dangling separator.
181
+ Check list joins at one item and at three or more.
182
+
183
+ ## Generalist brief
184
+
185
+ The shared preamble applies. In place of an assigned territory, review the
186
+ full diff with no restriction, and favor the seams no focused reviewer on this
187
+ panel names as its own. Your addendum says which reviewers ran.
188
+
189
+ In the check round, your surface is the fix diff and its blast radius, the fix
190
+ reports and the confirmed classes' verdict files. Answer three questions with
191
+ evidence for every class. Is it resolved, shown by its verification test? Does
192
+ a fix report account for it? A class no report names counts as not fixed. Did
193
+ the hunks that fix it regress anything in those hunks or their blast radius?
194
+ Search for every export or identifier they change and check its consumers.
195
+
196
+ You are the one reader who sees every fix at once. Hunt for a fix for one
197
+ class that undoes another, a helper corrected on one call path and left
198
+ inverted on a sibling, and a state one fix makes reachable that another fix's
199
+ reader cannot handle. Do not re-review surface the fixes did not touch. Name
200
+ every class by the id the merged file gave it. A new defect the check confirms
201
+ gets a fresh id, `r2-c<k>` counting from 1, never a reused one.
202
+
203
+ ## Refuter brief
204
+
205
+ You are a refuter with a kill mandate. You receive a claimed finding: its
206
+ severity, kind, location and failure scenario, and for a class, the full
207
+ instance inventory. You do not receive the finder's reasoning. You also get
208
+ the branch diff and the repository. Write your full verdict to the path your
209
+ brief names, and return one line per class: the id, refuted or confirmed, and
210
+ the severity and kind you assign.
211
+
212
+ A class is a claim about its pattern. Where instances differ, prune the
213
+ inventory to the instances the claim holds for. Do not kill or confirm the
214
+ whole class on one instance.
215
+
216
+ Your job is to disprove the finding. Show the scenario cannot occur, is
217
+ already handled, misreads the code, or is not a defect under the repository's
218
+ documented rules. Attempt the strongest refutation first. If the finding can
219
+ be checked, check it: write the failing test or reproduce the behavior. An
220
+ empirical result outranks your reasoning and the finder's.
221
+
222
+ A confirmed finding ships as that failing test. Write it where its suite
223
+ lives, name it after the behavior it pins, and hand it over verbatim. It is
224
+ the brief the implementer works from and the regression test the fix keeps.
225
+ Run that test file alone, never its package's suite. Where a test is
226
+ impossible, say why in one line. "I did not write one" is not a reason.
227
+
228
+ Verdict: refuted with the disproof, or confirmed with the evidence and the
229
+ severity and kind you would assign. Either may differ from the finder's. Kind
230
+ gates what blocks the merge, so correcting a wrong kind is part of the job.
231
+ A class that cites a rule is checked at the citation too. Open the cited path
232
+ and confirm the quoted sentence is there. A quote you cannot find is dropped,
233
+ and your line says so. When you remain uncertain after honest effort, default
234
+ to refuted.
235
+
236
+ Judge each class on its own. Attempt each refutation separately and write one
237
+ verdict file per class. Sharing an area is a reason to read the code once,
238
+ never a reason to let one verdict colour another.
239
+
240
+ ## Mutation brief
241
+
242
+ Run this only when the repository has a mutation testing tool for changed
243
+ lines. Otherwise skip it and say so in the review report.
244
+
245
+ You are the mutation checker. Your inputs are the branch diff against its
246
+ base, the issue, the criteria with their citing tests, and the instructions
247
+ pointer. Work only over the lines the diff adds or changes.
248
+
249
+ Work in the scratch worktree your brief names, and leave it in place. When
250
+ your brief names none, create one under the system temp directory, install
251
+ dependencies, and give its path in your report. Reviewers are reading the
252
+ shared tree while you run, so no mutation may touch it.
253
+
254
+ 1. Run the repository's changed-lines mutation tool once over the whole diff.
255
+ Every survivor it reports is a finding. Never rerun it per file, never
256
+ lower its budget, and never run a whole suite or a coverage run. A run
257
+ that ends over budget is reported as over budget with the mutants it
258
+ scored. One rerun is allowed, and only when the tool reports no result
259
+ because a workspace dependency could not be resolved: build the
260
+ dependencies, then run once more.
261
+ 2. Read the tool's report. For each criterion whose production lines were
262
+ mutated, its citing test must appear among the killers of a mutant on those
263
+ lines. A criterion whose citing test killed nothing there is a finding that
264
+ names the criterion as unproven. This is a read of the report, not a test
265
+ run.
266
+ 3. For lines the tool never mutates, such as SQL, shell, markup or
267
+ configuration, break by hand each predicate, guard, cap and assignment the
268
+ diff introduces. Invert it, widen it, drop a conjunct, or ignore the field.
269
+ Judge each break by running the one test file that should catch it. If
270
+ nothing fails, that is a finding.
271
+
272
+ Every finding has kind proof strength and names the line, the mutation
273
+ applied, and the test that should have failed. Write the full report to the
274
+ path in your brief. Return at most ten lines: the score and whether the run
275
+ stayed in budget, counts by step, and the criterion ids left unproven.
@@ -0,0 +1,158 @@
1
+ # Phase 4: the review loop
2
+
3
+ Run this after the implementer's branch is green and before anything is
4
+ pushed. The reviewer, refuter and mutation briefs are in
5
+ [review-briefs.md](review-briefs.md). The merger and fix briefs are in
6
+ [agent-briefs.md](agent-briefs.md). Spawn every brief verbatim.
7
+
8
+ ## Tier
9
+
10
+ Before spawning anyone, record a tier verdict in the state file: the tier,
11
+ the reviewers, the round budget, the limit on agents in flight, and one or
12
+ two sentences on the diff's shape and risk. No verdict means stop and
13
+ escalate. Read the tier off the finished diff, never off the issue. Count the
14
+ diff's non-test production lines and note what it touches.
15
+
16
+ "Floor" means risky surfaces: authorization, tenancy or access control, data
17
+ access, schema or migrations, secrets and environment handling, and any
18
+ surface that runs with elevated privileges. A diff that touches one is `full`
19
+ whatever its size.
20
+
21
+ | Item and diff | Tier | Reviewers | Judge |
22
+ |---|---|---|---|
23
+ | No criteria | small | one generalist | no judge runs |
24
+ | Under ~40 lines, off the floor | small | one generalist | when criteria exist |
25
+ | Up to ~800 lines, off the floor | medium | generalist + one focused reviewer | when criteria exist |
26
+ | Floor, or larger | full | generalist + up to three | when criteria exist |
27
+
28
+ An item with no criteria is `small` whatever its diff, and no judge runs.
29
+
30
+ - **Small.** One generalist reads the full diff. No focused reviewer and no
31
+ mutation agent. The orchestrator runs the repository's mutation tool itself
32
+ when it has one, and a survivor on a new line is a proof-strength finding.
33
+ There is no merger: the orchestrator reads the one report and writes the
34
+ class table. One fix pass. The orchestrator reruns each confirmed class's
35
+ test and the full gate itself. Round budget one. Only blocker and major
36
+ findings block. A minor is fixed when its repair stays inside files the
37
+ diff already touches, and otherwise goes in the final report, unverified.
38
+ - **Medium.** The generalist plus one focused reviewer: the one whose
39
+ territory holds the diff's largest surface. Use the contracts reviewer for a
40
+ changed shared interface, the test-strength reviewer for a change that is
41
+ mostly tests, the failure-path reviewer for new failure paths, and the
42
+ hygiene reviewer otherwise. No mutation agent. Round budget two, where round
43
+ two is the check below and never a second panel.
44
+ - **Full.** The generalist plus at most three focused reviewers. On a risky
45
+ surface, the security reviewer and the failure-path reviewer are always two
46
+ of them. One mutation checker, when the repository has a mutation tool.
47
+ Round budget two.
48
+
49
+ More reviewers than this find more claims, but the extra ones are refuted at
50
+ the same rate. What grows is the refuter wave and the fix pass behind them.
51
+
52
+ The limit on agents in flight is twelve unless the rationale argues
53
+ otherwise, and never above twenty. It counts every agent a stage spawns. A
54
+ stage with more work runs in successive waves, each finishing before the next
55
+ starts. A round that finds more blocking classes than the budget can verify
56
+ is not a scheduling problem. Stop and escalate: a person decides whether the
57
+ change is still worth fixing.
58
+
59
+ Models: use one model for every review role unless the team says otherwise.
60
+ A reviewer on a risky surface, or the refuter of a blocking class, never
61
+ moves to a weaker model than the rest.
62
+
63
+ ## Panel
64
+
65
+ Write `build-reports/round-1-bundle.md` first: the branch diff against its
66
+ base, the issue body, the criteria, the changed-file list, and the list of
67
+ test files at head. Every reviewer reads the bundle first and opens source
68
+ files only where it is not enough.
69
+
70
+ Spawn the tier's reviewers in parallel, blind to each other and cold. Each
71
+ gets the bundle path, the instructions-file pointer, the shared preamble and
72
+ one reviewer brief from [review-briefs.md](review-briefs.md). A short focus
73
+ addendum may add emphasis inside the reviewer's territory. It never removes
74
+ territory or edits the brief. The generalist's addendum names any territory
75
+ no other reviewer covers this round. Each reviewer writes its full report to
76
+ `build-reports/round-1-<role>.md` and returns the ten-line summary.
77
+
78
+ The run keeps one scratch worktree for every agent that needs an isolated
79
+ tree. Create it once under the system temp directory, install once, record its
80
+ path in the state file, hand it to one agent at a time, and delete it at the
81
+ end of this phase.
82
+
83
+ ## Merge
84
+
85
+ One merger agent, on the Merger brief, reads the round's reports and returns
86
+ the class table only. A class is a claim about a pattern, so findings that
87
+ share a kind and a repair are one class.
88
+
89
+ A finding is a class only when it carries its reproduction: a failing test, a
90
+ command and its output, or the exact input and the code path that mishandles
91
+ it. A finding with reasoning alone is a note. Notes go at the end of the
92
+ merged file, and no refuter or fix agent sees them. The final report lists
93
+ them.
94
+
95
+ A finding whose repair lies outside the issue's goal, or outside the files the
96
+ diff touches and the code that calls them, is not a class. It is a candidate
97
+ issue. The exception is a behavior or security defect the diff itself
98
+ introduced.
99
+
100
+ More than twenty classes from one round means the merge over-split. The
101
+ merger gets one pass to consolidate.
102
+
103
+ ## Verify
104
+
105
+ Group the classes by area: two classes share an area when a file appears in
106
+ both. Spawn one refuter per area, at most three, each holding every class in
107
+ its area. Each refuter reads the claims and the code, not the finders'
108
+ reasoning, and writes one verdict file per class, returning one line per
109
+ class: the id, refuted or confirmed, and the severity and kind it assigns.
110
+
111
+ A confirmed finding arrives as the refuter's failing test. Where a test is
112
+ impossible, such as a comment that lies, the brief says so in one line.
113
+
114
+ A confirmed hygiene class of minor severity or below gets no refuter. It is
115
+ fixed when its repair stays inside the diff's files, and otherwise recorded,
116
+ unverified. A nit gets no refuter and goes to an "Unverified, non-blocking"
117
+ section that is never dropped.
118
+
119
+ Under `full`, spawn the mutation checker in the same wave, in the scratch
120
+ worktree. Each surviving mutant is a confirmed class of kind proof strength
121
+ and goes to the fix stage without a refuter. The fix stage waits for both.
122
+
123
+ ## Fix
124
+
125
+ The implementer runs again on the Fix brief. Its input is the list of refuter
126
+ verdict files and the mutation report, never pasted findings. It fixes
127
+ classes, not instances. With one group, or fewer than eight confirmed classes,
128
+ one fix agent works on the branch. Otherwise partition by area into at most
129
+ three groups. Each group's agent works in its own worktree on a branch from
130
+ the feature branch head, and the groups are merged into the feature branch in
131
+ turn. A conflict means the partition was wrong: re-run that group on the
132
+ merged head.
133
+
134
+ ## Check
135
+
136
+ Round two is a check and never a panel. Rerun every confirmed class's test
137
+ yourself, in one command, then run the full gate. Then spawn one generalist
138
+ over the fix hunks and the code they call. It reconciles each class against
139
+ its fix report, attacks the fix hunks and their blast radius, and hunts for a
140
+ fix for one class that undoes another. It writes `build-reports/round-2-check.md`.
141
+
142
+ A new class the check confirms gets one more fix pass, and the orchestrator
143
+ reruns its test and the gate as proof. A class still open after that ends the
144
+ budget: escalate. Repeated fix rounds add defects nearly as fast as they
145
+ remove them.
146
+
147
+ ## Stop
148
+
149
+ A confirmed finding blocks when its severity is minor or above and its kind
150
+ is behavior, security, data integrity or proof strength. Under `small`, only
151
+ major or above blocks. Confirmed hygiene findings are fixed in the round and
152
+ never block. Done means zero confirmed blocking findings within the round
153
+ budget. Budget spent without convergence means escalate.
154
+
155
+ Finish by writing the review report as one HTML file: the tier and why, the
156
+ classes with their verdicts, what was fixed, and an "Unverified,
157
+ non-blocking" section for anything left unrefuted. Emit it with
158
+ `outerlayer emit artifact <report>.html --for code-review-ran`.
@@ -0,0 +1,235 @@
1
+ #!/usr/bin/env node
2
+ // Reads an issue body, finds the acceptance criteria, writes them as the
3
+ // list `outerlayer emit criteria` takes, records the list, and notes the
4
+ // outcome in the build's state file. Needs only Node.
5
+ //
6
+ // node record-criteria.mjs --item <number> --issue-file <path>
7
+ //
8
+ // It never edits the issue, never replaces a list a person recorded (the
9
+ // gateway answers 409 and the item's list stands), and records nothing when
10
+ // the issue holds neither shape.
11
+ import { spawnSync } from "node:child_process";
12
+ import { mkdirSync, readFileSync, writeFileSync } from "node:fs";
13
+ import { dirname, join } from "node:path";
14
+ import { pathToFileURL } from "node:url";
15
+
16
+ const PROOF_KINDS = ["video", "screenshot", "report", "log", "file"];
17
+ const BULLET = /^\s*(?:[-*+]|\d+[.)])\s+(.*)$/;
18
+ const HEADING = /^\s{0,3}(#{1,6})\s+(.*?)\s*#*\s*$/;
19
+ const ID_ITEM = /^`([A-Za-z0-9_.:-]{1,64})`\s*(.*)$/;
20
+ const LABEL = /^\s{0,3}(?:#{1,6}\s+)?(?:\*\*|__)?(acceptance(?: criteria)?|done when)\s*:?\s*(?:\*\*|__)?\s*:?\s*$/i;
21
+
22
+ /** The lines with the contents of fenced code blocks, and the fences, blanked: a quoted example is not a section. */
23
+ function blankFencedLines(lines) {
24
+ let fence = null;
25
+ return lines.map((line) => {
26
+ const opener = /^\s{0,3}(`{3,}|~{3,})/.exec(line);
27
+ if (fence === null) {
28
+ if (opener === null) return line;
29
+ fence = opener[1];
30
+ return "";
31
+ }
32
+ if (opener !== null && opener[1][0] === fence[0] && opener[1].length >= fence.length) fence = null;
33
+ return "";
34
+ });
35
+ }
36
+
37
+ /** A bullet and the indented lines continuing it, as one string. */
38
+ function collectItems(lines, start) {
39
+ const items = [];
40
+ let i = start;
41
+ for (; i < lines.length; i++) {
42
+ const line = lines[i];
43
+ const bullet = BULLET.exec(line);
44
+ if (bullet) {
45
+ items.push(bullet[1].trim());
46
+ } else if (line.trim() === "") {
47
+ // a blank line ends the list unless another bullet follows
48
+ const next = lines.slice(i + 1).find((l) => l.trim() !== "");
49
+ if (next === undefined || !BULLET.test(next)) break;
50
+ } else if (/^\s+\S/.test(line) && items.length > 0) {
51
+ items[items.length - 1] += ` ${line.trim()}`;
52
+ } else {
53
+ break;
54
+ }
55
+ }
56
+ return items;
57
+ }
58
+
59
+ function stripCheckbox(text) {
60
+ return text.replace(/^\[[ xX]\]\s+/, "");
61
+ }
62
+
63
+ /** The id-bearing shape: an `## Acceptance criteria` section whose every item starts with a backticked id. */
64
+ function readIdSection(lines) {
65
+ const at = lines.findIndex((l) => {
66
+ const h = HEADING.exec(l);
67
+ return h !== null && h[1].length === 2 && /^acceptance criteria$/i.test(h[2]);
68
+ });
69
+ if (at === -1) return null;
70
+ const items = [];
71
+ for (let i = at + 1; i < lines.length; i++) {
72
+ const h = HEADING.exec(lines[i]);
73
+ if (h !== null && h[1].length <= 2) break;
74
+ const bullet = BULLET.exec(lines[i]);
75
+ if (bullet) items.push(bullet[1].trim());
76
+ else if (/^\s+\S/.test(lines[i]) && items.length > 0) items[items.length - 1] += ` ${lines[i].trim()}`;
77
+ }
78
+ if (items.length === 0) return null;
79
+ const parsed = items.map((item) => ID_ITEM.exec(item));
80
+ const withId = parsed.filter((m) => m !== null && m[2].trim() !== "").length;
81
+ if (withId === 0) return null;
82
+ if (withId < parsed.length) return { mixed: true };
83
+ return {
84
+ criteria: parsed.map((m) => {
85
+ const kind = /\(proof:\s*([a-z]+)\s*\)/i.exec(m[2])?.[1]?.toLowerCase();
86
+ return { id: m[1], text: m[2].trim(), proof: PROOF_KINDS.includes(kind) ? kind : null };
87
+ }),
88
+ };
89
+ }
90
+
91
+ /** The plain shape: an Acceptance / Acceptance criteria / Done when label, then a bullet list. */
92
+ function readLabelledBullets(lines) {
93
+ for (let i = 0; i < lines.length; i++) {
94
+ if (!LABEL.test(lines[i])) continue;
95
+ let start = i + 1;
96
+ while (start < lines.length && lines[start].trim() === "") start++;
97
+ const items = collectItems(lines, start).map(stripCheckbox).filter((t) => t !== "");
98
+ if (items.length > 0) return items;
99
+ }
100
+ return null;
101
+ }
102
+
103
+ /**
104
+ * Finds the criteria in an issue body.
105
+ * @returns {{shape: "ids"|"bullets"|"none", criteria: Array<{id:string,text:string,proof:string|null}>, note?: string}}
106
+ */
107
+ export function parseCriteria(body, item) {
108
+ const lines = blankFencedLines(String(body).split(/\r?\n/));
109
+ const ids = readIdSection(lines);
110
+ if (ids?.criteria) return { shape: "ids", criteria: ids.criteria };
111
+ const bullets = readLabelledBullets(lines);
112
+ if (ids?.mixed && bullets) {
113
+ return {
114
+ shape: "none",
115
+ criteria: [],
116
+ note: "some criteria carry an id and some do not; recording nothing rather than renumbering the written ids",
117
+ };
118
+ }
119
+ if (bullets) {
120
+ return {
121
+ shape: "bullets",
122
+ criteria: bullets.map((text, n) => ({ id: `AC-${item}-${String(n + 1).padStart(2, "0")}`, text, proof: null })),
123
+ };
124
+ }
125
+ return { shape: "none", criteria: [] };
126
+ }
127
+
128
+ /** Sorts a failed `emit criteria` into "the item already has a person's list" and any other refusal. */
129
+ export function classifyEmitFailure(message) {
130
+ // Only the CLI's replace refusal reads "refused (409):". Other answers that
131
+ // carry a 409, such as a lost race, read "emit failed (409):".
132
+ return /\brefused \(409\):/.test(message) ? "item-list-stands" : "refused";
133
+ }
134
+
135
+ /** Builds the `emit` function that runs `outerlayer emit criteria`; `spawn` is `spawnSync`, or a stand-in in tests. */
136
+ export function makeEmit(spawn, item) {
137
+ return (file) => {
138
+ const run = spawn("outerlayer", ["emit", "criteria", file, "--item", item], { encoding: "utf8", timeout: 120000 });
139
+ if (run.status === 0) return { ok: true };
140
+ if (run.error) return { ok: false, message: `could not run outerlayer emit criteria: ${run.error.message}` };
141
+ if (run.signal) return { ok: false, message: `outerlayer emit criteria was stopped by ${run.signal}` };
142
+ return { ok: false, message: `${run.stderr ?? ""}${run.stdout ?? ""}`.replace(/\s+/g, " ").trim() || "outerlayer emit criteria failed" };
143
+ };
144
+ }
145
+
146
+ function readState(path) {
147
+ try {
148
+ return JSON.parse(readFileSync(path, "utf8"));
149
+ } catch {
150
+ return {};
151
+ }
152
+ }
153
+
154
+ /**
155
+ * Records the issue's criteria and notes the outcome in the state file.
156
+ * `emit(file)` runs `outerlayer emit criteria <file>` and returns
157
+ * `{ ok: true }` or `{ ok: false, message }`.
158
+ */
159
+ export async function recordCriteria({ body, item, reportsDir, statePath, emit }) {
160
+ const found = parseCriteria(body, item);
161
+ let outcome;
162
+ if (found.shape === "none") {
163
+ outcome = { source: "none", recorded: false, count: 0, list: [], ...(found.note ? { note: found.note } : {}) };
164
+ } else {
165
+ mkdirSync(reportsDir, { recursive: true });
166
+ const file = join(reportsDir, "criteria.json");
167
+ writeFileSync(file, `${JSON.stringify({ criteria: found.criteria }, null, 2)}\n`);
168
+ const result = await emit(file);
169
+ const base = { count: found.criteria.length, list: found.criteria };
170
+ if (result.ok) {
171
+ outcome = { source: found.shape === "ids" ? "issue-ids" : "issue-bullets", recorded: true, ...base };
172
+ } else if (classifyEmitFailure(result.message) === "item-list-stands") {
173
+ // No command returns the item's own list, so `list` stays empty; the issue's reading goes under `issueReading`.
174
+ outcome = {
175
+ source: "item",
176
+ recorded: false,
177
+ count: found.criteria.length,
178
+ list: [],
179
+ issueReading: found.criteria,
180
+ note: "the item already has a recorded list; it stands",
181
+ };
182
+ } else {
183
+ outcome = {
184
+ source: found.shape === "ids" ? "issue-ids" : "issue-bullets",
185
+ recorded: false,
186
+ ...base,
187
+ refusal: result.message,
188
+ };
189
+ }
190
+ }
191
+ mkdirSync(dirname(statePath), { recursive: true });
192
+ writeFileSync(statePath, `${JSON.stringify({ ...readState(statePath), criteria: outcome }, null, 2)}\n`);
193
+ return outcome;
194
+ }
195
+
196
+ export function summarize(outcome) {
197
+ const n = outcome.count;
198
+ const noun = n === 1 ? "criterion" : "criteria";
199
+ if (outcome.source === "none") return `no acceptance criteria found; none recorded${outcome.note ? ` (${outcome.note})` : ""}`;
200
+ if (outcome.source === "item") return `the item already has a recorded list; it stands, and the ${n} ${noun} read from the issue were not recorded`;
201
+ const shape = outcome.source === "issue-ids" ? "an Acceptance criteria section with ids" : "bullets under an Acceptance heading";
202
+ return outcome.recorded
203
+ ? `recorded ${n} ${noun} from ${shape}`
204
+ : `read ${n} ${noun} from ${shape} but the record was refused: ${outcome.refusal}`;
205
+ }
206
+
207
+ async function main(argv) {
208
+ const arg = (name) => {
209
+ const at = argv.indexOf(name);
210
+ return at === -1 ? undefined : argv[at + 1];
211
+ };
212
+ const item = arg("--item");
213
+ const issueFile = arg("--issue-file");
214
+ if (!item || !/^\d+$/.test(item) || !issueFile) {
215
+ process.stderr.write("usage: record-criteria.mjs --item <number> --issue-file <path>\n");
216
+ return 2;
217
+ }
218
+ const git = spawnSync("git", ["rev-parse", "--absolute-git-dir"], { encoding: "utf8" });
219
+ if (git.status !== 0) {
220
+ process.stderr.write(`not inside a git repository${git.error ? `: ${git.error.message}` : ""}\n`);
221
+ return 2;
222
+ }
223
+ const gitDir = git.stdout.trim();
224
+ const outcome = await recordCriteria({
225
+ body: readFileSync(issueFile, "utf8"),
226
+ item,
227
+ reportsDir: join(gitDir, "build-reports"),
228
+ statePath: join(gitDir, "build-state.json"),
229
+ emit: makeEmit(spawnSync, item),
230
+ });
231
+ process.stdout.write(`${summarize(outcome)}\n`);
232
+ return 0;
233
+ }
234
+
235
+ if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) process.exitCode = await main(process.argv.slice(2));