@mrciphersmith/keryx 0.2.73 → 0.2.75

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. package/dist/cli.js +33536 -33010
  2. package/package.json +1 -1
  3. package/src/gdskills/bundled/skills/core/reviewer-skill-creator/SKILL.md +214 -0
  4. package/src/gdskills/bundled/skills/orchestration/flow-orchestrator/SKILL.md +48 -1
  5. package/src/gdskills/bundled/skills/orchestration/flow-orchestrator/input-contract.schema.json +70 -4
  6. package/src/gdskills/bundled/skills/planning/consistency-checker/SKILL.codex.md +1 -1
  7. package/src/gdskills/bundled/skills/planning/consistency-checker/SKILL.cursor.md +1 -1
  8. package/src/gdskills/bundled/skills/planning/consistency-checker/SKILL.md +1 -1
  9. package/src/gdskills/bundled/skills/planning/patterns-researcher/SKILL.codex.md +1 -1
  10. package/src/gdskills/bundled/skills/planning/patterns-researcher/SKILL.cursor.md +1 -1
  11. package/src/gdskills/bundled/skills/planning/patterns-researcher/SKILL.md +1 -1
  12. package/src/gdskills/bundled/skills/planning/planner/SKILL.codex.md +1 -1
  13. package/src/gdskills/bundled/skills/planning/planner/SKILL.cursor.md +1 -1
  14. package/src/gdskills/bundled/skills/planning/planner/SKILL.md +1 -1
  15. package/src/gdskills/bundled/skills/planning/problem-definer/SKILL.codex.md +1 -1
  16. package/src/gdskills/bundled/skills/planning/problem-definer/SKILL.cursor.md +1 -1
  17. package/src/gdskills/bundled/skills/planning/problem-definer/SKILL.md +1 -1
  18. package/src/gdskills/bundled/skills/planning/project-discovery/SKILL.codex.md +1 -1
  19. package/src/gdskills/bundled/skills/planning/project-discovery/SKILL.cursor.md +1 -1
  20. package/src/gdskills/bundled/skills/planning/project-discovery/SKILL.md +1 -1
  21. package/src/gdskills/bundled/skills/planning/spec-writer/SKILL.codex.md +1 -1
  22. package/src/gdskills/bundled/skills/planning/spec-writer/SKILL.cursor.md +1 -1
  23. package/src/gdskills/bundled/skills/planning/spec-writer/SKILL.md +1 -1
  24. package/src/gdskills/bundled/skills/planning/stack-advisor/SKILL.codex.md +1 -1
  25. package/src/gdskills/bundled/skills/planning/stack-advisor/SKILL.cursor.md +1 -1
  26. package/src/gdskills/bundled/skills/planning/stack-advisor/SKILL.md +1 -1
  27. package/src/gdskills/bundled/skills/review/review-clean-code/SKILL.md +33 -1
  28. package/src/gdskills/bundled/skills/review/review-layout/SKILL.md +217 -0
  29. package/src/gdskills/bundled/skills/review/review-logic/SKILL.md +26 -0
  30. package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.md +304 -2
  31. package/src/gdskills/bundled/skills/review/review-pr-feedback/SKILL.md +644 -113
  32. package/src/gdskills/bundled/skills/review/review-pr-feedback/input-contract.schema.json +79 -0
  33. package/src/gdskills/bundled/skills/review/review-pr-feedback/output-contract.schema.json +375 -0
  34. package/src/gdskills/bundled/skills/review/review-testing-practices/SKILL.md +111 -1
  35. package/src/gdskills/bundled/skills/review/review-verifier/SKILL.md +25 -1
@@ -0,0 +1,79 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://keryx.local/schemas/review-pr-feedback/input-contract.schema.json",
4
+ "title": "ReviewPrFeedbackInput",
5
+ "type": "object",
6
+ "additionalProperties": false,
7
+ "required": [
8
+ "pr_url"
9
+ ],
10
+ "properties": {
11
+ "pr_url": {
12
+ "type": "string",
13
+ "minLength": 1,
14
+ "description": "GitHub PR URL (https://github.com/owner/repo/pull/123), owner/repo#123, or #123 when the git remote resolves the repository."
15
+ },
16
+ "fix": {
17
+ "type": "boolean",
18
+ "default": false,
19
+ "description": "Run the execution steps: dispatch flow-orchestrator with the fix plan, merge into the reviewed PR's head branch, and answer every comment. Never inferred \u2014 absent this flag the skill produces a plan and stops."
20
+ },
21
+ "context_doc": {
22
+ "type": "string",
23
+ "description": "Path to a job context document, e.g. <JOBS_ROOT>/<job>/ai/context.md. Optional and non-blocking."
24
+ },
25
+ "comment_ids": {
26
+ "type": "array",
27
+ "items": {
28
+ "type": "string"
29
+ },
30
+ "default": [],
31
+ "description": "Restrict the run to these collected comment ids. Every excluded comment is still listed, with that as its reason \u2014 a filter that removes silently reads as nobody having commented."
32
+ },
33
+ "max_fix_rounds": {
34
+ "type": "integer",
35
+ "minimum": 1,
36
+ "description": "Ceiling on review/fix rounds, passed to flow-orchestrator as a constraint. It can only lower that skill's own attempt budget, never raise it."
37
+ },
38
+ "operator_confirmed": {
39
+ "type": "object",
40
+ "additionalProperties": false,
41
+ "required": [
42
+ "confirmed_by",
43
+ "confirmed_at",
44
+ "plan_digest"
45
+ ],
46
+ "description": "REQUIRED whenever `fix` is true and this skill is running under dispatch. A subagent has no user to answer the Step 9 confirmation, and --fix merges third-party review comments into somebody's pull request: without this the run is BLOCKED, never defaulted. What `plan_digest` is worth is stated on the field itself; do not infer a replay defence from its presence here.",
47
+ "properties": {
48
+ "confirmed_by": {
49
+ "type": "string",
50
+ "minLength": 1
51
+ },
52
+ "confirmed_at": {
53
+ "type": "string",
54
+ "minLength": 1
55
+ },
56
+ "plan_digest": {
57
+ "type": "string",
58
+ "minLength": 1,
59
+ "description": "A record of WHICH plan the human said they read \u2014 not a control. Nothing in this tree computes or verifies a digest, and the schema constrains it only to a non-empty string, so an agent composing the dispatch chooses the value and a plan mutated after approval carries the same one. `operator_confirmed`'s PRESENCE is enforced; this field's VALUE is not. Making it a control means hashing the rendered plan and having the receiver recompute over what it got."
60
+ }
61
+ }
62
+ }
63
+ },
64
+ "if": {
65
+ "required": [
66
+ "fix"
67
+ ],
68
+ "properties": {
69
+ "fix": {
70
+ "const": true
71
+ }
72
+ }
73
+ },
74
+ "then": {
75
+ "required": [
76
+ "operator_confirmed"
77
+ ]
78
+ }
79
+ }
@@ -0,0 +1,375 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://keryx.local/schemas/review-pr-feedback/output-contract.schema.json",
4
+ "title": "ReviewPrFeedbackOutput",
5
+ "type": "object",
6
+ "additionalProperties": false,
7
+ "required": [
8
+ "status",
9
+ "mode",
10
+ "pr",
11
+ "head_sha",
12
+ "collected",
13
+ "verdicts",
14
+ "plan_items",
15
+ "summary",
16
+ "screen_status",
17
+ "screened",
18
+ "excluded_for_injection",
19
+ "filtered"
20
+ ],
21
+ "properties": {
22
+ "status": {
23
+ "type": "string",
24
+ "enum": [
25
+ "DONE",
26
+ "DONE_WITH_CONCERNS",
27
+ "NEEDS_CONTEXT",
28
+ "BLOCKED"
29
+ ]
30
+ },
31
+ "mode": {
32
+ "type": "string",
33
+ "enum": [
34
+ "analyze",
35
+ "fix"
36
+ ]
37
+ },
38
+ "pr": {
39
+ "type": "string",
40
+ "description": "<owner>/<repo>#<n>"
41
+ },
42
+ "head_sha": {
43
+ "type": "string",
44
+ "description": "The commit the collection and every verdict are true of. A run that cannot name it is stale, never fresh."
45
+ },
46
+ "collected": {
47
+ "type": "integer",
48
+ "minimum": 0
49
+ },
50
+ "filtered": {
51
+ "type": "array",
52
+ "default": [],
53
+ "items": {
54
+ "type": "object",
55
+ "additionalProperties": false,
56
+ "required": [
57
+ "comment",
58
+ "reason"
59
+ ],
60
+ "properties": {
61
+ "comment": {
62
+ "type": "string"
63
+ },
64
+ "reason": {
65
+ "type": "string"
66
+ }
67
+ }
68
+ },
69
+ "description": "Everything the collection or comment_ids removed, with the reason. Never empty by omission."
70
+ },
71
+ "verdicts": {
72
+ "type": "object",
73
+ "additionalProperties": false,
74
+ "properties": {
75
+ "valid": {
76
+ "type": "integer",
77
+ "minimum": 0
78
+ },
79
+ "valid-wider": {
80
+ "type": "integer",
81
+ "minimum": 0
82
+ },
83
+ "already-fixed": {
84
+ "type": "integer",
85
+ "minimum": 0
86
+ },
87
+ "not-reproducible": {
88
+ "type": "integer",
89
+ "minimum": 0
90
+ },
91
+ "disagree": {
92
+ "type": "integer",
93
+ "minimum": 0
94
+ },
95
+ "out-of-scope": {
96
+ "type": "integer",
97
+ "minimum": 0
98
+ },
99
+ "needs-clarification": {
100
+ "type": "integer",
101
+ "minimum": 0
102
+ },
103
+ "unverified": {
104
+ "type": "integer",
105
+ "minimum": 0
106
+ }
107
+ }
108
+ },
109
+ "plan_items": {
110
+ "type": "integer",
111
+ "minimum": 0
112
+ },
113
+ "fix": {
114
+ "type": [
115
+ "object",
116
+ "null"
117
+ ],
118
+ "additionalProperties": false,
119
+ "description": "Present in fix mode only. Null in analyze mode: no branch, no PR, no merge, no reply.",
120
+ "required": [
121
+ "flow_id",
122
+ "flow_status",
123
+ "merged_into",
124
+ "operator_confirmed"
125
+ ],
126
+ "properties": {
127
+ "flow_id": {
128
+ "type": "string"
129
+ },
130
+ "flow_status": {
131
+ "type": "string",
132
+ "enum": [
133
+ "initialized",
134
+ "in_progress",
135
+ "implemented",
136
+ "done",
137
+ "blocked",
138
+ "failed"
139
+ ]
140
+ },
141
+ "fix_pr_url": {
142
+ "type": [
143
+ "string",
144
+ "null"
145
+ ]
146
+ },
147
+ "merged_into": {
148
+ "type": [
149
+ "string",
150
+ "null"
151
+ ],
152
+ "description": "The reviewed PR's head branch. Any other value is a fix that did not land inside the pull request it answers."
153
+ },
154
+ "merge_sha": {
155
+ "type": [
156
+ "string",
157
+ "null"
158
+ ],
159
+ "description": "The merge commit. Its presence is what makes the round's exit condition enforceable: a recorded merge REQUIRES remaining_findings.blocker/major/minor to be 0. The threshold was prose in two skills and nothing refused an output that merged over three blockers."
160
+ },
161
+ "review_rounds": {
162
+ "type": "integer",
163
+ "minimum": 0
164
+ },
165
+ "remaining_findings": {
166
+ "type": "object",
167
+ "additionalProperties": false,
168
+ "properties": {
169
+ "blocker": {
170
+ "type": "integer",
171
+ "minimum": 0
172
+ },
173
+ "major": {
174
+ "type": "integer",
175
+ "minimum": 0
176
+ },
177
+ "minor": {
178
+ "type": "integer",
179
+ "minimum": 0
180
+ },
181
+ "info": {
182
+ "type": "integer",
183
+ "minimum": 0
184
+ }
185
+ },
186
+ "description": "A merge requires blocker, major and minor to be zero. info does not hold the loop."
187
+ },
188
+ "operator_confirmed": {
189
+ "type": "object",
190
+ "additionalProperties": false,
191
+ "required": [
192
+ "confirmed_by",
193
+ "confirmed_at",
194
+ "plan_digest"
195
+ ],
196
+ "description": "Who authorised the merge and the replies, and against which plan. REQUIRED in fix mode and NOT nullable: the field exists to separate an approved run from an assumed one, and both an omitted field and a null one are the assumed run wearing the approved one's shape. An interactive run records the operator it asked; there is no merge without somebody behind it.",
197
+ "properties": {
198
+ "confirmed_by": {
199
+ "type": "string",
200
+ "minLength": 1
201
+ },
202
+ "confirmed_at": {
203
+ "type": "string",
204
+ "minLength": 1
205
+ },
206
+ "plan_digest": {
207
+ "type": "string",
208
+ "minLength": 1
209
+ }
210
+ }
211
+ }
212
+ },
213
+ "if": {
214
+ "required": [
215
+ "merge_sha"
216
+ ],
217
+ "properties": {
218
+ "merge_sha": {
219
+ "type": "string",
220
+ "minLength": 1
221
+ }
222
+ }
223
+ },
224
+ "then": {
225
+ "properties": {
226
+ "remaining_findings": {
227
+ "required": [
228
+ "blocker",
229
+ "major",
230
+ "minor"
231
+ ],
232
+ "properties": {
233
+ "blocker": {
234
+ "const": 0
235
+ },
236
+ "major": {
237
+ "const": 0
238
+ },
239
+ "minor": {
240
+ "const": 0
241
+ }
242
+ }
243
+ }
244
+ }
245
+ }
246
+ },
247
+ "replies": {
248
+ "type": [
249
+ "object",
250
+ "null"
251
+ ],
252
+ "additionalProperties": false,
253
+ "description": "Present in fix mode only, after the merge.",
254
+ "properties": {
255
+ "posted": {
256
+ "type": "integer",
257
+ "minimum": 0
258
+ },
259
+ "escalated": {
260
+ "type": "array",
261
+ "items": {
262
+ "type": "string"
263
+ },
264
+ "default": []
265
+ },
266
+ "backlog": {
267
+ "type": "array",
268
+ "items": {
269
+ "type": "string"
270
+ },
271
+ "default": []
272
+ }
273
+ }
274
+ },
275
+ "action_items": {
276
+ "type": "array",
277
+ "items": {
278
+ "type": "string"
279
+ },
280
+ "default": []
281
+ },
282
+ "learning_proposal": {
283
+ "type": [
284
+ "string",
285
+ "null"
286
+ ],
287
+ "description": "Path to the proposal keryx review learn wrote. Proposed, never applied \u2014 applying is the caller's step."
288
+ },
289
+ "needs_context": {
290
+ "type": "array",
291
+ "items": {
292
+ "type": "string"
293
+ },
294
+ "default": []
295
+ },
296
+ "summary": {
297
+ "type": "string"
298
+ },
299
+ "screened": {
300
+ "type": "integer",
301
+ "minimum": 0,
302
+ "description": "Comments passed through `keryx security check-input`. REQUIRED: absent and 0 are different claims. Whether the screen ran at all is `screen_status`, not this \u2014 a fallback run reports `screened: 0` with `screen_status: unavailable`, and that pairing is enforced."
303
+ },
304
+ "excluded_for_injection": {
305
+ "type": "array",
306
+ "items": {
307
+ "type": "string"
308
+ },
309
+ "default": [],
310
+ "description": "Comment ids carrying a prompt-injection finding. They produce no plan item and skip Steps 6 and 7; they are still answered in Step 10."
311
+ },
312
+ "screen_status": {
313
+ "type": "string",
314
+ "enum": [
315
+ "ran",
316
+ "unavailable"
317
+ ],
318
+ "description": "Whether the injection screen ran at all. REQUIRED, because `screened: 0` is ambiguous on its own: the documented fallback path \u2014 keryx CLI unavailable \u2014 screens nothing and would otherwise emit exactly what a run that screened and found nothing emits. `unavailable` obliges the report to carry the line saying so."
319
+ }
320
+ },
321
+ "if": {
322
+ "required": [
323
+ "mode"
324
+ ],
325
+ "properties": {
326
+ "mode": {
327
+ "const": "fix"
328
+ }
329
+ }
330
+ },
331
+ "then": {
332
+ "required": [
333
+ "fix"
334
+ ],
335
+ "properties": {
336
+ "fix": {
337
+ "type": "object"
338
+ }
339
+ }
340
+ },
341
+ "else": {
342
+ "properties": {
343
+ "fix": {
344
+ "type": "null"
345
+ },
346
+ "replies": {
347
+ "type": "null"
348
+ }
349
+ }
350
+ },
351
+ "allOf": [
352
+ {
353
+ "if": {
354
+ "required": [
355
+ "screen_status"
356
+ ],
357
+ "properties": {
358
+ "screen_status": {
359
+ "const": "unavailable"
360
+ }
361
+ }
362
+ },
363
+ "then": {
364
+ "properties": {
365
+ "screened": {
366
+ "const": 0
367
+ },
368
+ "excluded_for_injection": {
369
+ "maxItems": 0
370
+ }
371
+ }
372
+ }
373
+ }
374
+ ]
375
+ }
@@ -6,11 +6,13 @@ description: |
6
6
  against repository-local testing conventions: co-location, network mocking,
7
7
  MSW-style boundary mocks, behaviour assertions, deterministic waits,
8
8
  smoke/full split, locator priority, and screenshot-test discipline.
9
+ Runs a bounded mutation pass over the gates the diff added, so a suite that
10
+ cannot fail is found by executing it rather than by reading it.
9
11
  Dispatched by review-orchestrator for --testing-practices,
10
12
  --project-conventions, --all, or changed test/e2e/story files.
11
13
  metadata:
12
14
  author: "MrCipherSmith"
13
- version: "1.0.0"
15
+ version: "1.1.0"
14
16
  category: "review"
15
17
  license: "MIT"
16
18
  ---
@@ -36,6 +38,64 @@ If the repository has local test documentation, cite the relevant convention in
36
38
 
37
39
  ---
38
40
 
41
+ ## The mutation pass — run this before you read anything
42
+
43
+ Reading a suite tells you whether it is *well written*. It does not tell you
44
+ whether it can **fail**. Those are different questions, and the second one is the
45
+ one this reviewer exists to answer.
46
+
47
+ So: before the checklist, for every guard, filter, conditional or branch the diff
48
+ **added or changed**, delete it, run the nearest suite, record the result, revert.
49
+
50
+ ```
51
+ for each gate G added or changed in scope A:
52
+ delete G -> run the nearest suite -> record red/green -> revert
53
+ ```
54
+
55
+ **A gate whose deletion leaves the suite green is a finding**, however correct the
56
+ gate itself is. State it as: *this gate is unpinned; no fixture builds the input
57
+ that separates it from the condition beside it.* It is mandatory when the gate was
58
+ added in answer to an earlier round — that is where a regression costs most and
59
+ where coverage is most often assumed rather than written.
60
+
61
+ Publish the whole table, survivors and casualties both. The reds are what makes
62
+ the greens mean something; a table with only survivors reads as cherry-picking.
63
+
64
+ | mutation | result |
65
+ |---|---|
66
+ | delete the `(half.weight ?? 0) > 0` filter | **18/18 green** |
67
+ | `controlsScore` -> `report.score` | 2 fail |
68
+ | drop `Math.min(100, …)` | 1 fail |
69
+
70
+ ### Why this is your job and not the verifier's
71
+
72
+ `review-verifier` also executes — but it is **delete-only**: it confirms or refutes
73
+ findings that already exist. A surviving mutation is not a verdict on a finding,
74
+ it *is* the finding, and nothing downstream of you can produce one. If you skip
75
+ this pass the class is unreachable for the whole round.
76
+
77
+ ### Bounds
78
+
79
+ - Mutate only gates inside **scope A**, one at a time, reverting each before the
80
+ next. Never leave a mutation in the tree.
81
+ - Run the **nearest** suite, not the full one. This is not a `test:mutation`
82
+ sweep — those are slow, opt-in, and answer a different question.
83
+ - Cap it: the gates the diff added, plus any gate a previous round asked for. If
84
+ that set is larger than your budget, take the previous-round gates first and say
85
+ in your summary how many you did not reach. A stated cap is a result; silence
86
+ reads as a clean sweep.
87
+ - If you cannot run the suite at all, say so and skip the pass. Do **not** report
88
+ a mutation you did not run — a predicted result is `info` at best, and this
89
+ reviewer has no use for one.
90
+
91
+ ### This does not replace reading the tests
92
+
93
+ Assertion duplication and mutation survival are independent: a suite can be free
94
+ of duplication, correctly tiered, properly mocked — and still pin nothing. Run the
95
+ pass, then work the checklist.
96
+
97
+ ---
98
+
39
99
  ## Checklist
40
100
 
41
101
  ### Test Location and Tiers
@@ -65,6 +125,39 @@ If the repository has local test documentation, cite the relevant convention in
65
125
  - Store-internal timing and concurrency can be unit-tested with lower-level mocks when that is
66
126
  the clean seam.
67
127
 
128
+ ### Fixture Shape Against the Producer
129
+
130
+ - For every fixture whose name or comment claims a scenario — `drift-only`, `legacy
131
+ report`, `never run` — find the **producer** (the backend class, the builder, the
132
+ API contract) and compare **every** field, not only the one the test pins.
133
+ - A fixture that calls itself X while carrying the shape of Y covers Y, leaves X
134
+ untested, and reads to the next author as coverage of X. That is worse than no
135
+ fixture: it is coverage that lies.
136
+ - Cheapest tell: the fixture sets one field to the scenario's value and inherits
137
+ the rest from a default builder written for a different scenario.
138
+ - Name the field that disagrees and the producer line that settles it. If the
139
+ producer lives in another repository, pin the SHA you read it at. If you cannot
140
+ read it, the finding is `info` and says what would settle it.
141
+ - A value the producer **cannot emit** (a score of 40 where the backend rounds to
142
+ tens, two weights summing to 200) is the same class: the test pins a state that
143
+ does not exist, so it cannot catch the state that does.
144
+
145
+ ### Branch Under Test
146
+
147
+ - When the code under test chooses between sources — a preferred one and a
148
+ fallback — establish which branch the fixture actually drives.
149
+ - A comment of the form *"this copy is the authoritative one, the other is a
150
+ fallback"* is a direct instruction that the test must go through the preferred
151
+ branch. Assertions that all ride the fallback while the preferred source is
152
+ authoritative leave the suite green in exactly the failure the comment warns
153
+ about.
154
+ - Look for the preference in **the code under test, not in the test**. The test
155
+ cannot tell you which branch it takes; the selector can.
156
+ - Same shape, one level up: a test that asserts only falsy state — "not open",
157
+ "not present", "no toast" — passes trivially when the code path never ran. Each
158
+ such test needs one positive artifact proving the path was entered before the
159
+ negative assertions mean anything.
160
+
68
161
  ### Browser and E2E Determinism
69
162
 
70
163
  - Tests create their own artifacts with unique names and assert on those artifacts.
@@ -158,9 +251,26 @@ conditions land under that rubric.
158
251
  |---|---|---|
159
252
  | A test that passes while the behaviour it names is broken | `blocker` | The acceptance criterion is unimplemented — the test only claims otherwise |
160
253
  | Real-network leak; shared-data mutation across tests; fixed sleeps; asserting on a backend race | `major` | Named trigger (the run) and named outcome (flake or a false pass) |
254
+ | A gate the diff added survives deletion with the suite green | `minor` | The code is correct; the cost is that the next edit to it is unprotected |
255
+ | A fixture whose shape contradicts the scenario it names, or pins a value the producer cannot emit | `minor` | Same — plus the next author reads it as coverage it does not provide |
256
+ | Assertions ride the fallback branch while the code names another authoritative | `minor` | Same |
161
257
  | Substrate choice, smoke tagging, locator priority, co-location | `minor` | The suite is correct; the cost is to whoever maintains it |
162
258
  | A convention preference with no effect on determinism or signal | `info` | Shared laws 1 and 2 |
163
259
 
164
260
  A flaky test is `major`, not `blocker`: it wastes time, but it does not ship a
165
261
  defect. A test that cannot fail does — which is why it is the one `blocker` here.
166
262
 
263
+ The three mutation-pass rows sit at `minor` deliberately, and the reasoning is
264
+ the canonical rubric rather than modesty: a coverage gap names no runtime trigger
265
+ and no user-visible outcome, so it cannot be `major` under the boundary test. What
266
+ makes them worth reporting is not severity but **evidence class** — each one is a
267
+ command that ran, which is the rarest thing in a review and the only kind of
268
+ finding a later round cannot argue away. Report them with the mutation and its
269
+ result, and they survive adjudication intact. Report them as "coverage looks
270
+ thin" and they are correctly discarded.
271
+
272
+ The moment a surviving mutation is accompanied by a **reachable input that makes
273
+ the unpinned code do the wrong thing**, that is a separate finding at the severity
274
+ its own outcome earns — usually `major`. The coverage gap and the defect are two
275
+ findings, not one, and conflating them loses whichever the adjudicator disbelieves.
276
+
@@ -17,7 +17,7 @@ triggers:
17
17
  - "verification pass"
18
18
  metadata:
19
19
  author: "MrCipherSmith"
20
- version: "1.0.0"
20
+ version: "1.1.0"
21
21
  category: "review"
22
22
  compatible_harnesses: "cursor,codex,zed,opencode,claude"
23
23
  license: "MIT"
@@ -130,6 +130,30 @@ A finding is `confirmed` when the procedure **reproduced the defect**, and
130
130
  command you ran would not have failed either way, you have not verified anything —
131
131
  use `unverifiable` and say what you ran.
132
132
 
133
+ #### `refuted` by unreachability is not the end of the thread
134
+
135
+ There is one refutation shape that closes a finding and leaves a question standing:
136
+ the input the finding needs is **unreachable given the current behaviour of a
137
+ producer you do not own** — another repository, a service, a generated client.
138
+
139
+ The harm claim is genuinely dead and `refuted` is the correct verdict. But an
140
+ invariant held by another repo's code plus a comment on ours is exactly what breaks
141
+ silently when the contract moves, and the question the refutation raises is a
142
+ different one:
143
+
144
+ > *Does any test build the shape that was in question?*
145
+
146
+ Record that successor question in your **prose summary**, naming the producer and
147
+ the SHA your check relied on. It is prose, never a claim — you delete, you do not
148
+ add — and the orchestrator picks it up for the next round or routes it to
149
+ `review-testing-practices`.
150
+
151
+ The failure this exists to stop is recorded: a reviewer asked "is this broken?",
152
+ a verifier correctly answered "no, the backend never emits that", and the thread
153
+ ended. The shape was untested, stayed untested, and the coverage gap was found by a
154
+ different reviewer two rounds later. *Is it broken* and *is it pinned* are two
155
+ questions; refuting the first says nothing about the second.
156
+
133
157
  ### 2. `site-check` — do the named sites exist?
134
158
 
135
159
  A `blocker` or `major` carries `class_scope.sites`: every location holding the