@mrciphersmith/keryx 0.2.70 → 0.2.72

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (192) hide show
  1. package/dist/cli.js +25237 -17483
  2. package/docs/README.md +54 -0
  3. package/docs/requirements/shared-agent-context/README.md +104 -0
  4. package/package.json +3 -2
  5. package/src/gdskills/bundled/rules/core/code-review-learned-profile.mdc +81 -0
  6. package/src/gdskills/bundled/rules/core/jobs-documentation.mdc +1 -1
  7. package/src/gdskills/bundled/rules/core/model-selection.mdc +184 -31
  8. package/src/gdskills/bundled/rules/core/review-strict-profile.mdc +8 -4
  9. package/src/gdskills/bundled/rules/core/skills-storage-workflow.mdc +36 -0
  10. package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.codex.md +1 -1
  11. package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.cursor.md +1 -1
  12. package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.md +2 -1
  13. package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.opencode.md +1 -1
  14. package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.zed.md +1 -1
  15. package/src/gdskills/bundled/skills/orchestration/context-collector/SKILL.codex.md +3 -3
  16. package/src/gdskills/bundled/skills/orchestration/context-collector/SKILL.cursor.md +3 -3
  17. package/src/gdskills/bundled/skills/orchestration/context-collector/SKILL.md +3 -3
  18. package/src/gdskills/bundled/skills/orchestration/context-collector/SKILL.opencode.md +3 -3
  19. package/src/gdskills/bundled/skills/orchestration/context-collector/SKILL.zed.md +3 -3
  20. package/src/gdskills/bundled/skills/orchestration/context-collector/orchestrator-prompt.md +2 -2
  21. package/src/gdskills/bundled/skills/orchestration/feature-analyzer/SKILL.codex.md +4 -4
  22. package/src/gdskills/bundled/skills/orchestration/feature-analyzer/SKILL.cursor.md +4 -4
  23. package/src/gdskills/bundled/skills/orchestration/feature-analyzer/SKILL.md +4 -4
  24. package/src/gdskills/bundled/skills/orchestration/feature-analyzer/SKILL.opencode.md +4 -4
  25. package/src/gdskills/bundled/skills/orchestration/feature-analyzer/SKILL.zed.md +4 -4
  26. package/src/gdskills/bundled/skills/orchestration/feature-dev/SKILL.codex.md +2 -2
  27. package/src/gdskills/bundled/skills/orchestration/feature-dev/SKILL.cursor.md +2 -2
  28. package/src/gdskills/bundled/skills/orchestration/feature-dev/SKILL.md +3 -3
  29. package/src/gdskills/bundled/skills/orchestration/flow-orchestrator/SKILL.md +80 -21
  30. package/src/gdskills/bundled/skills/orchestration/issue-analyzer/SKILL.codex.md +1 -1
  31. package/src/gdskills/bundled/skills/orchestration/issue-analyzer/SKILL.cursor.md +1 -1
  32. package/src/gdskills/bundled/skills/orchestration/issue-analyzer/SKILL.md +1 -1
  33. package/src/gdskills/bundled/skills/orchestration/issue-analyzer/SKILL.opencode.md +1 -1
  34. package/src/gdskills/bundled/skills/orchestration/issue-analyzer/SKILL.zed.md +1 -1
  35. package/src/gdskills/bundled/skills/orchestration/job-documenter/SKILL.codex.md +2 -2
  36. package/src/gdskills/bundled/skills/orchestration/job-documenter/SKILL.cursor.md +2 -2
  37. package/src/gdskills/bundled/skills/orchestration/job-documenter/SKILL.md +3 -2
  38. package/src/gdskills/bundled/skills/orchestration/job-documenter/SKILL.opencode.md +2 -2
  39. package/src/gdskills/bundled/skills/orchestration/job-documenter/SKILL.zed.md +2 -2
  40. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.codex.md +997 -509
  41. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.cursor.md +997 -509
  42. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.md +968 -513
  43. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.opencode.md +997 -509
  44. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.zed.md +997 -509
  45. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/input-contract.schema.json +38 -28
  46. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/orchestrator-prompt.md +98 -66
  47. package/src/gdskills/bundled/skills/orchestration/job-orchestrator/output-contract.schema.json +27 -5
  48. package/src/gdskills/bundled/skills/orchestration/task-implementer/SKILL.codex.md +20 -2
  49. package/src/gdskills/bundled/skills/orchestration/task-implementer/SKILL.cursor.md +20 -2
  50. package/src/gdskills/bundled/skills/orchestration/task-implementer/SKILL.md +21 -2
  51. package/src/gdskills/bundled/skills/orchestration/task-implementer/SKILL.opencode.md +20 -2
  52. package/src/gdskills/bundled/skills/orchestration/task-implementer/SKILL.zed.md +20 -2
  53. package/src/gdskills/bundled/skills/orchestration/task-implementer/input-contract.schema.json +1 -1
  54. package/src/gdskills/bundled/skills/planning/autodoc-analyst/SKILL.md +2 -1
  55. package/src/gdskills/bundled/skills/planning/autodoc-architect/SKILL.md +3 -1
  56. package/src/gdskills/bundled/skills/planning/autodoc-assembler/SKILL.md +2 -1
  57. package/src/gdskills/bundled/skills/planning/autodoc-orchestrator/SKILL.md +2 -1
  58. package/src/gdskills/bundled/skills/planning/autodoc-scanner/SKILL.md +2 -1
  59. package/src/gdskills/bundled/skills/planning/autodoc-writer/SKILL.md +2 -1
  60. package/src/gdskills/bundled/skills/planning/brainstorm/SKILL.codex.md +1 -1
  61. package/src/gdskills/bundled/skills/planning/brainstorm/SKILL.cursor.md +1 -1
  62. package/src/gdskills/bundled/skills/planning/brainstorm/SKILL.md +1 -1
  63. package/src/gdskills/bundled/skills/planning/consistency-checker/SKILL.codex.md +1 -1
  64. package/src/gdskills/bundled/skills/planning/consistency-checker/SKILL.cursor.md +1 -1
  65. package/src/gdskills/bundled/skills/planning/consistency-checker/SKILL.md +2 -1
  66. package/src/gdskills/bundled/skills/planning/docpack-orchestrator/SKILL.md +1 -1
  67. package/src/gdskills/bundled/skills/planning/docpack-review/SKILL.md +1 -1
  68. package/src/gdskills/bundled/skills/planning/interview/SKILL.codex.md +1 -1
  69. package/src/gdskills/bundled/skills/planning/interview/SKILL.cursor.md +1 -1
  70. package/src/gdskills/bundled/skills/planning/interview/SKILL.md +1 -1
  71. package/src/gdskills/bundled/skills/planning/interviewer/SKILL.codex.md +1 -1
  72. package/src/gdskills/bundled/skills/planning/interviewer/SKILL.cursor.md +1 -1
  73. package/src/gdskills/bundled/skills/planning/interviewer/SKILL.md +1 -1
  74. package/src/gdskills/bundled/skills/planning/patterns-researcher/SKILL.codex.md +1 -1
  75. package/src/gdskills/bundled/skills/planning/patterns-researcher/SKILL.cursor.md +1 -1
  76. package/src/gdskills/bundled/skills/planning/patterns-researcher/SKILL.md +2 -1
  77. package/src/gdskills/bundled/skills/planning/planner/SKILL.codex.md +1 -1
  78. package/src/gdskills/bundled/skills/planning/planner/SKILL.cursor.md +1 -1
  79. package/src/gdskills/bundled/skills/planning/planner/SKILL.md +2 -1
  80. package/src/gdskills/bundled/skills/planning/prd-creator/SKILL.codex.md +1 -1
  81. package/src/gdskills/bundled/skills/planning/prd-creator/SKILL.cursor.md +1 -1
  82. package/src/gdskills/bundled/skills/planning/prd-creator/SKILL.md +1 -1
  83. package/src/gdskills/bundled/skills/planning/prd-creator/SKILL.opencode.md +1 -1
  84. package/src/gdskills/bundled/skills/planning/prd-creator/SKILL.zed.md +1 -1
  85. package/src/gdskills/bundled/skills/planning/problem-definer/SKILL.codex.md +1 -1
  86. package/src/gdskills/bundled/skills/planning/problem-definer/SKILL.cursor.md +1 -1
  87. package/src/gdskills/bundled/skills/planning/problem-definer/SKILL.md +2 -1
  88. package/src/gdskills/bundled/skills/planning/project-discovery/SKILL.codex.md +1 -1
  89. package/src/gdskills/bundled/skills/planning/project-discovery/SKILL.cursor.md +1 -1
  90. package/src/gdskills/bundled/skills/planning/project-discovery/SKILL.md +2 -1
  91. package/src/gdskills/bundled/skills/planning/spec-writer/SKILL.codex.md +1 -1
  92. package/src/gdskills/bundled/skills/planning/spec-writer/SKILL.cursor.md +1 -1
  93. package/src/gdskills/bundled/skills/planning/spec-writer/SKILL.md +2 -1
  94. package/src/gdskills/bundled/skills/planning/stack-advisor/SKILL.codex.md +1 -1
  95. package/src/gdskills/bundled/skills/planning/stack-advisor/SKILL.cursor.md +1 -1
  96. package/src/gdskills/bundled/skills/planning/stack-advisor/SKILL.md +2 -1
  97. package/src/gdskills/bundled/skills/platform/claude-md-management/SKILL.codex.md +1 -1
  98. package/src/gdskills/bundled/skills/platform/claude-md-management/SKILL.cursor.md +1 -1
  99. package/src/gdskills/bundled/skills/platform/claude-md-management/SKILL.md +1 -1
  100. package/src/gdskills/bundled/skills/platform/hookify/SKILL.codex.md +1 -1
  101. package/src/gdskills/bundled/skills/platform/hookify/SKILL.cursor.md +1 -1
  102. package/src/gdskills/bundled/skills/platform/hookify/SKILL.md +1 -1
  103. package/src/gdskills/bundled/skills/quality/changelog/SKILL.codex.md +1 -1
  104. package/src/gdskills/bundled/skills/quality/changelog/SKILL.cursor.md +1 -1
  105. package/src/gdskills/bundled/skills/quality/changelog/SKILL.md +1 -1
  106. package/src/gdskills/bundled/skills/quality/commit/SKILL.codex.md +1 -1
  107. package/src/gdskills/bundled/skills/quality/commit/SKILL.cursor.md +1 -1
  108. package/src/gdskills/bundled/skills/quality/commit/SKILL.md +1 -1
  109. package/src/gdskills/bundled/skills/quality/db-migrate/SKILL.codex.md +1 -1
  110. package/src/gdskills/bundled/skills/quality/db-migrate/SKILL.cursor.md +1 -1
  111. package/src/gdskills/bundled/skills/quality/db-migrate/SKILL.md +1 -1
  112. package/src/gdskills/bundled/skills/quality/dependency-update/SKILL.codex.md +1 -1
  113. package/src/gdskills/bundled/skills/quality/dependency-update/SKILL.cursor.md +1 -1
  114. package/src/gdskills/bundled/skills/quality/dependency-update/SKILL.md +1 -1
  115. package/src/gdskills/bundled/skills/quality/deploy/SKILL.codex.md +1 -1
  116. package/src/gdskills/bundled/skills/quality/deploy/SKILL.cursor.md +1 -1
  117. package/src/gdskills/bundled/skills/quality/deploy/SKILL.md +1 -1
  118. package/src/gdskills/bundled/skills/quality/metaproject-security/SKILL.md +1 -1
  119. package/src/gdskills/bundled/skills/quality/perf-check/SKILL.codex.md +1 -1
  120. package/src/gdskills/bundled/skills/quality/perf-check/SKILL.cursor.md +1 -1
  121. package/src/gdskills/bundled/skills/quality/perf-check/SKILL.md +1 -1
  122. package/src/gdskills/bundled/skills/quality/pr/SKILL.codex.md +1 -1
  123. package/src/gdskills/bundled/skills/quality/pr/SKILL.cursor.md +1 -1
  124. package/src/gdskills/bundled/skills/quality/pr/SKILL.md +1 -1
  125. package/src/gdskills/bundled/skills/quality/pr-issue-documenter/SKILL.codex.md +1 -1
  126. package/src/gdskills/bundled/skills/quality/pr-issue-documenter/SKILL.cursor.md +1 -1
  127. package/src/gdskills/bundled/skills/quality/pr-issue-documenter/SKILL.md +1 -1
  128. package/src/gdskills/bundled/skills/quality/pr-issue-documenter/SKILL.opencode.md +1 -1
  129. package/src/gdskills/bundled/skills/quality/pr-issue-documenter/SKILL.zed.md +1 -1
  130. package/src/gdskills/bundled/skills/quality/push/SKILL.codex.md +1 -1
  131. package/src/gdskills/bundled/skills/quality/push/SKILL.cursor.md +1 -1
  132. package/src/gdskills/bundled/skills/quality/push/SKILL.md +1 -1
  133. package/src/gdskills/bundled/skills/quality/security-audit/SKILL.codex.md +1 -1
  134. package/src/gdskills/bundled/skills/quality/security-audit/SKILL.cursor.md +1 -1
  135. package/src/gdskills/bundled/skills/quality/security-audit/SKILL.md +1 -1
  136. package/src/gdskills/bundled/skills/quality/test-gen/SKILL.codex.md +1 -1
  137. package/src/gdskills/bundled/skills/quality/test-gen/SKILL.cursor.md +1 -1
  138. package/src/gdskills/bundled/skills/quality/test-gen/SKILL.md +1 -1
  139. package/src/gdskills/bundled/skills/quality/tests-creator/SKILL.codex.md +1 -1
  140. package/src/gdskills/bundled/skills/quality/tests-creator/SKILL.cursor.md +1 -1
  141. package/src/gdskills/bundled/skills/quality/tests-creator/SKILL.md +1 -1
  142. package/src/gdskills/bundled/skills/quality/tests-creator/SKILL.opencode.md +1 -1
  143. package/src/gdskills/bundled/skills/quality/tests-creator/SKILL.zed.md +1 -1
  144. package/src/gdskills/bundled/skills/review/code-ai-review/SKILL.codex.md +1 -1
  145. package/src/gdskills/bundled/skills/review/code-ai-review/SKILL.cursor.md +1 -1
  146. package/src/gdskills/bundled/skills/review/code-ai-review/SKILL.md +3 -3
  147. package/src/gdskills/bundled/skills/review/code-ai-review/SKILL.opencode.md +1 -1
  148. package/src/gdskills/bundled/skills/review/code-ai-review/SKILL.zed.md +1 -1
  149. package/src/gdskills/bundled/skills/review/code-learned-review/SKILL.codex.md +252 -0
  150. package/src/gdskills/bundled/skills/review/code-learned-review/SKILL.cursor.md +252 -0
  151. package/src/gdskills/bundled/skills/review/code-learned-review/SKILL.md +243 -0
  152. package/src/gdskills/bundled/skills/review/code-learned-review/SKILL.opencode.md +252 -0
  153. package/src/gdskills/bundled/skills/review/code-learned-review/SKILL.zed.md +252 -0
  154. package/src/gdskills/bundled/skills/review/code-mobx-store-review/SKILL.codex.md +1 -1
  155. package/src/gdskills/bundled/skills/review/code-mobx-store-review/SKILL.cursor.md +1 -1
  156. package/src/gdskills/bundled/skills/review/code-mobx-store-review/SKILL.md +3 -2
  157. package/src/gdskills/bundled/skills/review/code-mobx-store-review/SKILL.opencode.md +1 -1
  158. package/src/gdskills/bundled/skills/review/code-mobx-store-review/SKILL.zed.md +1 -1
  159. package/src/gdskills/bundled/skills/review/code-style-review/SKILL.codex.md +1 -1
  160. package/src/gdskills/bundled/skills/review/code-style-review/SKILL.cursor.md +1 -1
  161. package/src/gdskills/bundled/skills/review/code-style-review/SKILL.md +2 -2
  162. package/src/gdskills/bundled/skills/review/code-style-review/SKILL.opencode.md +1 -1
  163. package/src/gdskills/bundled/skills/review/code-style-review/SKILL.zed.md +1 -1
  164. package/src/gdskills/bundled/skills/review/review-architecture/SKILL.md +38 -11
  165. package/src/gdskills/bundled/skills/review/review-backend/SKILL.md +49 -15
  166. package/src/gdskills/bundled/skills/review/review-clean-code/SKILL.md +50 -13
  167. package/src/gdskills/bundled/skills/review/review-core-boundaries/SKILL.md +35 -3
  168. package/src/gdskills/bundled/skills/review/review-flow-graph/SKILL.md +34 -3
  169. package/src/gdskills/bundled/skills/review/review-frontend/SKILL.md +71 -30
  170. package/src/gdskills/bundled/skills/review/review-frontend-conventions/SKILL.md +35 -4
  171. package/src/gdskills/bundled/skills/review/review-highload/SKILL.md +50 -16
  172. package/src/gdskills/bundled/skills/review/review-logic/SKILL.md +41 -13
  173. package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.md +599 -30
  174. package/src/gdskills/bundled/skills/review/review-orchestrator/reviewer-finding.schema.json +7 -0
  175. package/src/gdskills/bundled/skills/review/review-orchestrator/verification-claim.schema.json +78 -0
  176. package/src/gdskills/bundled/skills/review/review-performance/SKILL.md +44 -14
  177. package/src/gdskills/bundled/skills/review/review-pr-feedback/SKILL.md +44 -19
  178. package/src/gdskills/bundled/skills/review/review-regression/SKILL.md +185 -0
  179. package/src/gdskills/bundled/skills/review/review-security-code/SKILL.md +45 -14
  180. package/src/gdskills/bundled/skills/review/review-style/SKILL.md +27 -7
  181. package/src/gdskills/bundled/skills/review/review-testing-practices/SKILL.md +36 -4
  182. package/src/gdskills/bundled/skills/review/review-verifier/SKILL.md +276 -0
  183. package/src/gdskills/bundled/skills/shared/git-merge-base.md +1 -1
  184. package/src/gdskills/contracts/review-finding.schema.json +119 -1
  185. package/src/gdskills/contracts/subagent-dispatch.schema.json +59 -3
  186. package/src/gdskills/bundled/rules/core/code-review-b091-profile.mdc +0 -48
  187. package/src/gdskills/bundled/skills/review/code-b091-review/SKILL.codex.md +0 -209
  188. package/src/gdskills/bundled/skills/review/code-b091-review/SKILL.cursor.md +0 -209
  189. package/src/gdskills/bundled/skills/review/code-b091-review/SKILL.md +0 -208
  190. package/src/gdskills/bundled/skills/review/code-b091-review/SKILL.opencode.md +0 -209
  191. package/src/gdskills/bundled/skills/review/code-b091-review/SKILL.zed.md +0 -209
  192. package/src/gdskills/bundled/skills/review/review-strict/SKILL.md +0 -328
@@ -1,5 +1,6 @@
1
1
  ---
2
2
  name: review-style
3
+ model_tier: light
3
4
  description: |
4
5
  Use when: reviewing code for style, naming conventions, readability, and DRY violations —
5
6
  without touching logic, architecture, security, or performance. Covers "review style",
@@ -7,7 +8,6 @@ description: |
7
8
  with --style flag.
8
9
  NOT for: logic bugs, architectural violations, security vulnerabilities, performance
9
10
  anti-patterns, or any finding that could cause a functional regression.
10
- version: "1.0.0"
11
11
  triggers:
12
12
  - "review style"
13
13
  - "style review"
@@ -18,8 +18,8 @@ metadata:
18
18
  author: "MrCipherSmith"
19
19
  version: "1.0.0"
20
20
  category: "review"
21
+ compatible_harnesses: "cursor,codex,zed,opencode,claude"
21
22
  license: "MIT"
22
- compatibility: "cursor,codex,zed,opencode,claude"
23
23
  ---
24
24
 
25
25
  # Review — Style, Naming & Readability
@@ -105,7 +105,7 @@ Only review code changed in scope. Do not flag style issues in lines outside the
105
105
  - File names match the primary export: `UserService` → `user.service.ts`
106
106
  - Test files: `*.spec.ts` or `*.test.ts` next to the subject file
107
107
 
108
- Flag naming issues as `minor` unless the name is actively misleading (e.g., `isLoading` that is actually a count), which is `major`.
108
+ Naming severity is set by the Style laws below and by the canonical rubric in `review-orchestrator` — not here. A second ruling in the same file is the defect this flow removed from `review-highload` and `review-frontend`.
109
109
 
110
110
  ---
111
111
 
@@ -198,18 +198,38 @@ Do not flag DRY violations for code outside the diff even if legacy duplication
198
198
 
199
199
  ## Iron Laws
200
200
 
201
+ ### Shared laws (every reviewer)
202
+
203
+ 1. **A claim of runtime harm with no reproducible path is `info`.** If you cannot
204
+ name the input, call, or condition that reaches the code, you have an
205
+ observation, not a finding. Report it as `info` and say what would settle it.
206
+ 2. **Never flag the theoretical.** The path you describe must exist in the code
207
+ under review. Do not report a safe API because it could be misused, or a
208
+ pattern because it is often wrong elsewhere.
209
+ 3. **One finding per class, not one per occurrence.** When the same shape appears
210
+ at several sites, report it once and list every site. Ten findings that are one
211
+ finding hide the other nine problems.
212
+
213
+ Severity levels are defined once, in `review-orchestrator/SKILL.md` →
214
+ **Severity (canonical)**. This reviewer does not restate them: `blocker` is the
215
+ four merge-blocking shapes named there and nothing else, and the `major`/`minor`
216
+ boundary is the trigger-and-outcome test.
217
+
218
+ ### Style laws
219
+
201
220
  | Rule | Rationale |
202
221
  |------|-----------|
203
- | Style findings are **never** blockers unless they cause a functional bug | Style is a quality concern, not a safety gate |
204
- | Maximum severity for pure style is `major` (only for actively misleading names or circular imports) | Most style is `minor` or `info` |
222
+ | Style findings are **never** `blocker` | None of the four merge-blocking shapes is reachable from a style observation. This reviewer's finding format omits `blocker` for that reason |
223
+ | A naming issue is `minor`; it reaches `major` only when the name has already produced an observable wrong outcome at a call site you can name | Identical to `review-clean-code` law 2, deliberately: the same condition must not carry two severities |
224
+ | Duplication is `minor`, reported once with every site listed | Identical to `review-clean-code` law 3 |
205
225
  | Never flag issues handled by the project's autoformatter (indentation, trailing spaces, bracket style) | Linter/formatter owns that; double-flagging creates noise |
206
- | Do not expand scope to architectural or logic concerns | Stay in style lane; hand off to the right reviewer |
226
+ | Do not expand scope to architectural or logic concerns | Stay in style lane; hand off to the right reviewer. A circular import that fails at runtime is `review-architecture`'s finding, not a style one |
207
227
 
208
228
  ---
209
229
 
210
230
  ## Orchestrated Review Contract
211
231
 
212
- When dispatched by `review-orchestrator`, follow the provided `reviewer-input.schema.json` payload. Return a `REVIEW_RESULT` object compatible with `skills/review-orchestrator/reviewer-finding.schema.json`, then a concise markdown summary. Keep findings evidence-based, include concrete `suggested_fix` for every blocker/major, and return `NEEDS_CONTEXT` instead of guessing when required context is missing.
232
+ When dispatched by `review-orchestrator`, follow the provided `reviewer-input.schema.json` payload. Return a `REVIEW_RESULT` object compatible with `skills/review/review-orchestrator/reviewer-finding.schema.json`, then a concise markdown summary. Keep findings evidence-based, include concrete `suggested_fix` for every blocker/major, and return `NEEDS_CONTEXT` instead of guessing when required context is missing.
213
233
 
214
234
  ---
215
235
 
@@ -1,5 +1,6 @@
1
1
  ---
2
2
  name: review-testing-practices
3
+ model_tier: standard
3
4
  description: |
4
5
  Use when reviewing unit, integration, Storybook, component, or e2e tests
5
6
  against repository-local testing conventions: co-location, network mocking,
@@ -85,9 +86,30 @@ If the repository has local test documentation, cite the relevant convention in
85
86
 
86
87
  ---
87
88
 
89
+ ## Iron Laws
90
+
91
+ ### Shared laws (every reviewer)
92
+
93
+ 1. **A claim of runtime harm with no reproducible path is `info`.** If you cannot
94
+ name the input, call, or condition that reaches the code, you have an
95
+ observation, not a finding. Report it as `info` and say what would settle it.
96
+ 2. **Never flag the theoretical.** The path you describe must exist in the code
97
+ under review. Do not report a safe API because it could be misused, or a
98
+ pattern because it is often wrong elsewhere.
99
+ 3. **One finding per class, not one per occurrence.** When the same shape appears
100
+ at several sites, report it once and list every site. Ten findings that are one
101
+ finding hide the other nine problems.
102
+
103
+ Severity levels are defined once, in `review-orchestrator/SKILL.md` →
104
+ **Severity (canonical)**. This reviewer does not restate them: `blocker` is the
105
+ four merge-blocking shapes named there and nothing else, and the `major`/`minor`
106
+ boundary is the trigger-and-outcome test.
107
+
108
+ ---
109
+
88
110
  ## Orchestrated Review Contract
89
111
 
90
- When dispatched by `review-orchestrator`, follow the provided `reviewer-input.schema.json` payload. Return a `REVIEW_RESULT` object compatible with `skills/review-orchestrator/reviewer-finding.schema.json`, then a concise markdown summary. Keep findings evidence-based, include concrete `suggested_fix` for every blocker/major, and return `NEEDS_CONTEXT` instead of guessing when required context is missing.
112
+ When dispatched by `review-orchestrator`, follow the provided `reviewer-input.schema.json` payload. Return a `REVIEW_RESULT` object compatible with `skills/review/review-orchestrator/reviewer-finding.schema.json`, then a concise markdown summary. Keep findings evidence-based, include concrete `suggested_fix` for every blocker/major, and return `NEEDS_CONTEXT` instead of guessing when required context is missing.
91
113
 
92
114
  ---
93
115
 
@@ -128,7 +150,17 @@ observation is theatre, not rigour.
128
150
  - **Fix**: concrete test rewrite or fixture/handler change
129
151
  ```
130
152
 
131
- Severity guidance: real-network leaks, shared-data mutation, fixed sleeps, and backend-race
132
- assertions are usually `major` or `blocker`; substrate choice and smoke tagging are usually
133
- `minor` unless they make CI flaky.
153
+ Severity comes from **Severity (canonical)** in `review-orchestrator/SKILL.md`.
154
+ This reviewer keeps no rubric of its own; what follows is where its recurring
155
+ conditions land under that rubric.
156
+
157
+ | Condition | Severity | Why, under the canonical rubric |
158
+ |---|---|---|
159
+ | A test that passes while the behaviour it names is broken | `blocker` | The acceptance criterion is unimplemented — the test only claims otherwise |
160
+ | Real-network leak; shared-data mutation across tests; fixed sleeps; asserting on a backend race | `major` | Named trigger (the run) and named outcome (flake or a false pass) |
161
+ | Substrate choice, smoke tagging, locator priority, co-location | `minor` | The suite is correct; the cost is to whoever maintains it |
162
+ | A convention preference with no effect on determinism or signal | `info` | Shared laws 1 and 2 |
163
+
164
+ A flaky test is `major`, not `blocker`: it wastes time, but it does not ship a
165
+ defect. A test that cannot fail does — which is why it is the one `blocker` here.
134
166
 
@@ -0,0 +1,276 @@
1
+ ---
2
+ name: review-verifier
3
+ model_tier: light
4
+ description: |
5
+ Use when: consolidated findings from other reviewers need to be checked before they are
6
+ reported — by RUNNING something that fails if the finding is real, or by confirming the
7
+ sites the finding named actually exist. Covers "verify the findings", "check these
8
+ findings", "review --verify", or dispatched by review-orchestrator as Wave C.
9
+ This reviewer can only DELETE. It cannot raise a severity, add a finding, or change a
10
+ finding's text.
11
+ NOT for: first-pass review (run domain reviewers first); re-scoring findings by re-reading
12
+ them, which is the operation this skill replaced and which is measured to degrade accuracy.
13
+ triggers:
14
+ - "verify findings"
15
+ - "review --verify"
16
+ - "check these findings"
17
+ - "verification pass"
18
+ metadata:
19
+ author: "MrCipherSmith"
20
+ version: "1.0.0"
21
+ category: "review"
22
+ compatible_harnesses: "cursor,codex,zed,opencode,claude"
23
+ license: "MIT"
24
+ ---
25
+
26
+ # Review — Verifier
27
+
28
+ Wave C. Reads the findings the domain reviewers produced and tries to **make each
29
+ one fail on purpose**. Emits one verdict per finding it checked. It removes; it
30
+ never adds and never sharpens.
31
+
32
+ ---
33
+
34
+ ## What this replaced, and why it is not coming back
35
+
36
+ This slot used to hold `review-strict`: a meta-pass that re-read the consolidated
37
+ findings and adjusted their severity, under an elevation table biased 3:1 toward
38
+ escalation, **with no new evidence**. It was removed rather than improved,
39
+ because the operation it performed is measured to make accuracy worse:
40
+
41
+ - **GPT-4 on GSM8K across self-correction rounds: 95.5 → 91.5 → 89.0.**
42
+ **GPT-3.5 on CommonSenseQA: 75.8 → 38.1.** Among the answers that changed,
43
+ correct → incorrect exceeded incorrect → correct (Huang et al., *Large Language
44
+ Models Cannot Self-Correct Reasoning Yet*, ICLR 2024, arXiv:2310.01798).
45
+ - **Self-Refine (arXiv:2303.17651) shows the same shape from the other side:
46
+ +49.2 on dialogue response generation, +0.2 on maths.** Self-refinement gains
47
+ live on subjective tasks and vanish on verifiable reasoning. Deciding whether a
48
+ null-guard is missing is verifiable reasoning.
49
+
50
+ So: re-reading a finding and changing what happens to it, without running
51
+ anything, is not a rigour pass. It is a coin flip weighted toward more findings.
52
+ If you are about to reintroduce it because it looks obviously useful — it looked
53
+ obviously useful the first time, and the numbers above are what happened.
54
+
55
+ ## Why this one is different
56
+
57
+ It does not re-read. It runs something.
58
+
59
+ - Verification that **executes** rejects **85–96% of false reports**, against
60
+ **4–15% unaided**, while finding **30–44% more true bugs** (AnyPoC,
61
+ arXiv:2604.11950).
62
+ - Meta's TestGen-LLM funnel: **75% build → 57% build and pass → 25% improve
63
+ coverage** — and the surviving quarter reaches **73% human acceptance**
64
+ (arXiv:2402.09171). A hard filter that discards three quarters of its own
65
+ output is what makes the remainder trustworthy.
66
+ - **80+ agents unanimously endorsed a padding-oracle vulnerability that did not
67
+ exist. A single empirical test killed it.** Consensus cannot detect a
68
+ hallucination its members share, so this skill never votes, never polls other
69
+ reviewers, and never treats agreement as evidence.
70
+
71
+ ---
72
+
73
+ ## Workflow
74
+
75
+ ```
76
+ review-verifier Progress:
77
+ - [ ] Step 1: Read the consolidated findings (global_id, reviewer, class_scope, evidence)
78
+ - [ ] Step 2: Drop every finding raised by yourself — you may not verify those
79
+ - [ ] Step 3: For each remaining finding, choose the strongest method that is actually available
80
+ - [ ] Step 4: Run it. Record the command and its output verbatim
81
+ - [ ] Step 5: Emit one claim per finding checked, conforming to verification-claim.schema.json
82
+ - [ ] Step 6: Emit nothing else — no new findings, no severity, no rewrites
83
+ ```
84
+
85
+ ---
86
+
87
+ ## Input Contract
88
+
89
+ | Field | Type | Required | Description |
90
+ |-------|------|----------|-------------|
91
+ | `findings` | array | yes | The consolidated findings, each conforming to `review-finding.schema.json`. `global_id` and `reviewer` must be present. |
92
+ | `verifier` | string | yes | Your own name. Recorded on every claim. |
93
+ | `branch` / `base_sha` | string | no | So a command can be run against the change under review. |
94
+ | `verification_mode` | string | no | `off` \| `annotate` \| `filter`. Informational for you — the mode is applied by the merge, not by you. Default `annotate`. |
95
+
96
+ ---
97
+
98
+ ## Step 3 — choosing a method
99
+
100
+ Three methods, strongest first. **Take the strongest one that is genuinely
101
+ available, not the strongest one you can describe.**
102
+
103
+ ### 1. `execution` — run something that fails if the finding is real
104
+
105
+ This is the method that works, and in this repository it is nearly always
106
+ available: keryx is a Bun project, so `bun test <file>`, `bun run typecheck`, or a
107
+ three-line script is cheap and immediate.
108
+
109
+ The test is not "does the suite pass". It is: **construct the situation the
110
+ finding claims is broken, and see whether it breaks.**
111
+
112
+ ```bash
113
+ # Finding: "createManagedReviewPackage writes a package even when the contract gate fails"
114
+ bun test src/review/managed.test.ts -t "refuses" # does the guard actually fire?
115
+
116
+ # Finding: "this guard is asserted against a synthetic value, so it cannot fail"
117
+ # Delete the guarded line and re-run. If the test stays green, the finding is CONFIRMED.
118
+
119
+ # Finding: "the type allows undefined here"
120
+ bun run typecheck
121
+ ```
122
+
123
+ Record the command and the relevant output. "I ran the tests and they passed" is
124
+ not evidence; the command you ran and its output is — `bun test <file>` with
125
+ the counts it actually printed, not a count quoted from somewhere else. (This
126
+ sentence used to quote a fixed number, which went stale twice in one day.)
127
+
128
+ A finding is `confirmed` when the procedure **reproduced the defect**, and
129
+ `refuted` when the procedure **that would have shown the defect did not**. If the
130
+ command you ran would not have failed either way, you have not verified anything —
131
+ use `unverifiable` and say what you ran.
132
+
133
+ ### 2. `site-check` — do the named sites exist?
134
+
135
+ A `blocker` or `major` carries `class_scope.sites`: every location holding the
136
+ shape, and how the set was enumerated. Check the list.
137
+
138
+ ```bash
139
+ keryx ctx rg "ensureKeryxConfigDir\(" src
140
+ ```
141
+
142
+ Weaker than execution, and the reason is worth keeping in mind: this establishes
143
+ that the code the finding describes is there. It does not establish that the
144
+ behaviour the finding claims occurs. A site list that is right about locations and
145
+ wrong about consequences passes this check.
146
+
147
+ - Sites named, sites found → the finding is about real code. Usually
148
+ `unverifiable` unless the check itself settles the claim.
149
+ - Sites named, sites **absent** → `refuted`, and the evidence is the search that
150
+ returned nothing.
151
+ - The enumeration is demonstrably incomplete → that is not a refutation. The
152
+ finding is understated, and understatement is not yours to correct. Record
153
+ `confirmed` if you established the shape exists, and say so in the evidence.
154
+
155
+ ### 3. `reasoning` — you could not do either
156
+
157
+ **Capped at `unverifiable`. It can never be `confirmed`, and it can never be
158
+ `refuted`.**
159
+
160
+ The cap is not conservatism. Reasoning alone produces no new evidence, and a
161
+ verdict is a claim about evidence; a "verified" verdict reached by re-reading is
162
+ exactly the `review-strict` operation whose numbers are at the top of this file.
163
+ `refuted` is capped for the same reason in the other direction: it is the one
164
+ verdict that removes a finding, so granting it to the weakest method would
165
+ reinstate the removed pass with the sign flipped.
166
+
167
+ Nothing checkable is lost by this. "The line this finding cites does not exist" is
168
+ a `site-check`, not reasoning. `reasoning` is the residual — the cases where you
169
+ ran nothing and looked up nothing — and the honest thing for the residual to say
170
+ is *I could not verify this*.
171
+
172
+ The merge enforces the cap: a `reasoning` claim carrying `confirmed` or `refuted`
173
+ is rewritten to `unverifiable`, and the attempt is recorded in the review record.
174
+
175
+ ---
176
+
177
+ ## Iron Laws
178
+
179
+ | Rule | Why |
180
+ |------|-----|
181
+ | **You can only delete.** No new findings, no severity changes, no edits to a finding's text. | The merge builds the record from the ORIGINAL finding and takes only your verdict. A claim carrying anything else is discarded whole, including its verdict. |
182
+ | **Never verify your own finding.** | The reviewer that raised a finding is the one actor whose agreement carries no information about it. Refused by the merge, by name. |
183
+ | **Reasoning alone is `unverifiable`.** | See above. Capped in code, not by this instruction. |
184
+ | **Every verdict cites what you ran.** | An unevidenced claim is discarded and the finding stays as reported. |
185
+ | **Never poll, never vote, never count agreement.** | 80+ agents unanimously endorsed a vulnerability that did not exist. |
186
+ | **`unverifiable` is a normal answer.** | Expect it to be the majority. Reaching for `refuted` to look productive is how a true blocker gets deleted. |
187
+ | **Not checking a finding removes nothing.** | A finding with no claim is reported unchanged. Absence is not a verdict. |
188
+
189
+ ---
190
+
191
+ ## Output Contract
192
+
193
+ ```
194
+ STATUS: DONE | NEEDS_CONTEXT | BLOCKED
195
+ ```
196
+
197
+ - `DONE` — every finding you were given was either checked or explicitly left alone.
198
+ - `NEEDS_CONTEXT` — you cannot run anything (no working tree, no test command); say what is missing.
199
+ - `BLOCKED` — the findings input is unreadable or carries no `global_id`.
200
+
201
+ There is no `DONE_WITH_CONCERNS`: a verifier has no concerns of its own.
202
+
203
+ Return one object conforming to
204
+ `skills/review/review-orchestrator/verification-claim.schema.json`:
205
+
206
+ ````text
207
+ ```json keryx:verifications
208
+ {
209
+ "status": "DONE",
210
+ "verifier": "review-verifier",
211
+ "summary": "12 findings, 4 executed, 3 site-checked, 5 unverifiable",
212
+ "verifications": [
213
+ {
214
+ "finding": "2026-08-29-pr-273#F-001",
215
+ "verdict": "refuted",
216
+ "method": "execution",
217
+ "evidence": "bun test src/lib/config-dir.writers.test.ts -t 'group-writable' -> 3 pass, 0 fail; measured mode 0700 under umask 002, the finding read the wrong call site"
218
+ },
219
+ {
220
+ "finding": "2026-08-29-pr-273#F-004",
221
+ "verdict": "unverifiable",
222
+ "method": "reasoning",
223
+ "evidence": "no command distinguishes the two orderings without a scheduler hook; nothing was run"
224
+ }
225
+ ],
226
+ "stats": { "confirmed": 4, "refuted": 1, "unverifiable": 5, "not_checked": 2 }
227
+ }
228
+ ```
229
+ ````
230
+
231
+ Then a short markdown summary:
232
+
233
+ ```markdown
234
+ # Verification Report
235
+
236
+ ## Method mix
237
+ - execution: N
238
+ - site-check: N
239
+ - reasoning (capped to unverifiable): N
240
+ - not checked: N (with the reason for each)
241
+
242
+ ## Refuted
243
+ <[global_id] finding — the command that ran, and what it returned>
244
+
245
+ ## Confirmed
246
+ <[global_id] finding — the command that reproduced it>
247
+
248
+ ## Unverifiable
249
+ <[global_id] finding — what was attempted and why it settled nothing>
250
+ ```
251
+
252
+ ---
253
+
254
+ ## Scope Boundaries
255
+
256
+ | Concern | This skill | Use instead |
257
+ |---------|------------|-------------|
258
+ | Checking whether a reported finding is real | YES | — |
259
+ | Removing a finding an executed check refuted | YES | — |
260
+ | Raising a severity, adding a finding, rewriting one | **NO — structurally impossible** | report it as a new round |
261
+ | First-pass logic / frontend / backend review | NO | `review-logic`, `review-frontend`, `review-backend` |
262
+ | Deciding what became of a finding after the fix | NO | `keryx review complete --disposition` |
263
+ | Architectural commentary and opinion | NO | file it as a finding in a domain reviewer, with evidence |
264
+
265
+ ---
266
+
267
+ ## Red Flags
268
+
269
+ | Rationalization | Why it is wrong |
270
+ |----------------|-----------------|
271
+ | "I read it carefully and it's clearly correct — `confirmed`." | Reading is not a method. That is capped at `unverifiable`, in code. |
272
+ | "It's obviously a false positive, `refuted`." | Obvious to whom, on what evidence? 80+ agents found an obvious vulnerability that did not exist. |
273
+ | "The finding is understated; I'll bump it to blocker." | You cannot. The merge discards the whole claim and records the attempt. |
274
+ | "Most of my verdicts are `unverifiable`, that looks bad." | It is the expected shape. TestGen-LLM discards 75% of its own output and that is why the remainder is trusted. |
275
+ | "I also spotted something the reviewers missed." | Report it through a domain reviewer in a new round, with evidence. This pass cannot add. |
276
+ | "I raised this finding, so I know best whether it holds." | That is exactly the case AC9 forbids. |
@@ -2,7 +2,7 @@
2
2
 
3
3
  ## Purpose
4
4
  Reusable bash script for determining the review scope (merge-base) for review skills.
5
- Used by: code-ai-review, code-boss-review, code-mobx-store-review, code-style-review.
5
+ Used by: code-ai-review, code-learned-review, code-mobx-store-review, code-style-review.
6
6
 
7
7
  ## Script
8
8
 
@@ -16,7 +16,84 @@
16
16
  ],
17
17
  "properties": {
18
18
  "id": { "type": "string" },
19
+ "global_id": {
20
+ "title": "The finding's identity, unique across every review package",
21
+ "description": "`id` is the DISPLAY form (`F-001`) and is per-report: 15 of 43 distinct ids in the recorded corpus appear in more than one package, and `F-001` denotes six different findings, so joining a finding to a commit, a ledger row or a flow journal by `id` is unsafe for two thirds of it. `global_id` is the join key: `<reviewId>#<id>`, minted once when the finding is first recorded and CARRIED VERBATIM thereafter, so a finding re-reported in round N+1 keeps the key it was minted under in round N. Uniqueness comes from `reviewId` being a package directory name rather than from a registry, which is what makes the key recomputable from disk with no new state. Optional because 83 findings predate it and read back without one; a record written by the current pipeline always carries it. The `pattern` is not decoration: `mintGlobalFindingId` is the only writer inside this repository, but a producer outside it may supply any string, and a `global_id` that is not `<reviewId>#<id>` joins to nothing while looking like a key — the measurement in scripts/review-precision-baseline.ts splits on the separator to reach the package.",
22
+ "type": "string",
23
+ "minLength": 1,
24
+ "pattern": "^[^#]+#[^#]+$"
25
+ },
19
26
  "reviewer": { "type": "string" },
27
+ "disposition": {
28
+ "title": "What became of this finding, and the evidence for saying so",
29
+ "description": "Declared HERE and deliberately NOT in the bundled reviewer-finding.schema.json: a reviewer states what is wrong, it never states what became of the finding. The measured reason this exists: precision = acted-on / (acted-on + dismissed-incorrect) computed over the recorded corpus returned 53/53 = 100%, not because the reviewers were right but because nothing on disk could express a wrong finding. `classification` looked like a validity verdict and was assigned from the ingest MODE; decisions.md said the same sentence for every finding. Only `acted-on` and `dismissed-incorrect` say anything about reviewer accuracy, which is why the three non-accuracy dismissals are separate states rather than one `dismissed` bucket — collapsing them is what makes a dismissal rate meaningless.",
30
+ "type": "object",
31
+ "additionalProperties": false,
32
+ "required": ["state"],
33
+ "properties": {
34
+ "state": {
35
+ "type": "string",
36
+ "enum": [
37
+ "unknown",
38
+ "acted-on",
39
+ "dismissed-incorrect",
40
+ "dismissed-wont-fix",
41
+ "dismissed-out-of-scope",
42
+ "dismissed-deprioritised",
43
+ "answered-disagree"
44
+ ],
45
+ "default": "unknown",
46
+ "description": "`unknown` is the default and the reading of an absent `disposition`: nobody recorded an outcome. It is never counted as valid — an unknown counted as acted-on inflates the very figure this field exists to make measurable."
47
+ },
48
+ "evidence": {
49
+ "type": "string",
50
+ "minLength": 1,
51
+ "description": "Where the outcome is written down: the commit that closed it, the test that went red, the decision that dismissed it. A disposition without this is an assertion, and assertions are what produced a corpus whose findings all appear correct. `answered-disagree` is the one state that is NOT a dismissal: it records that our verifier refuted an external comment, which settles the code and settles nothing about the person who asked — the reply is still owed."
52
+ }
53
+ },
54
+ "if": {
55
+ "required": ["state"],
56
+ "properties": { "state": { "const": "unknown" } }
57
+ },
58
+ "else": { "required": ["evidence"] }
59
+ },
60
+ "verification": {
61
+ "title": "What an independent check found when it went looking for this finding",
62
+ "description": "Declared HERE and deliberately NOT in the bundled reviewer-finding.schema.json, on the same basis as `disposition`: a reviewer states what is wrong, and whether someone else could reproduce it is not something the reviewer knows. The rule is sharper for this field than for `disposition`, because the reviewer that raised a finding is the ONE actor forbidden to answer it (AC9) — declaring the property in the shape reviewers emit would put the forbidden field next to `severity` in every reviewer's output. This replaces `review-strict`, which re-read findings and adjusted severity with no new evidence: intrinsic self-correction is measured to degrade accuracy (GPT-4 on GSM8K 95.5 -> 91.5 -> 89.0 across rounds; GPT-3.5 on CommonSenseQA 75.8 -> 38.1; Huang et al., ICLR 2024, arXiv:2310.01798). An ABSENT verification means nobody checked — which is true of all 83 pre-contract findings on disk — and is never a reason to drop a finding.",
63
+ "type": "object",
64
+ "additionalProperties": false,
65
+ "required": ["verdict", "method", "evidence"],
66
+ "properties": {
67
+ "verdict": {
68
+ "type": "string",
69
+ "enum": ["confirmed", "refuted", "unverifiable"],
70
+ "description": "`unverifiable` is not a failure state; it is the honest majority answer when nothing could be run. Only `refuted` removes anything, and only under verification_mode: filter."
71
+ },
72
+ "method": {
73
+ "type": "string",
74
+ "enum": ["execution", "site-check", "reasoning"],
75
+ "description": "Strongest first. `execution` ran a command or test that FAILS IF THE FINDING IS REAL — verification that executes rejects 85-96% of false reports against 4-15% unaided while finding 30-44% more true bugs (AnyPoC, arXiv:2604.11950). `site-check` confirmed the class_scope sites exist. `reasoning` is the residual, and it is capped: see the conditional below."
76
+ },
77
+ "evidence": {
78
+ "type": "string",
79
+ "minLength": 1,
80
+ "description": "The command that ran and what it returned, or the search that enumerated the sites. A verdict with nothing behind it is the operation this replaces."
81
+ },
82
+ "verifier": {
83
+ "type": "string",
84
+ "minLength": 1,
85
+ "description": "Who checked. Beyond the three properties AC7 names, and present because AC9 is otherwise untraceable after the fact: the merge refuses a claim whose verifier equals the finding's `reviewer`, but a record that does not say who verified cannot be audited for that rule — the same defect as `reviewer` being hardcoded to `review-orchestrator` on all 83 recorded findings."
86
+ }
87
+ },
88
+ "if": {
89
+ "required": ["method"],
90
+ "properties": { "method": { "const": "reasoning" } }
91
+ },
92
+ "then": {
93
+ "properties": { "verdict": { "const": "unverifiable" } },
94
+ "description": "Reasoning alone produces no new evidence, so it can never reach `confirmed` (AC7). It cannot reach `refuted` either, and that extension is deliberate: `refuted` is the only verdict with a destructive consequence, so granting it to the weakest method would reinstate review-strict with the sign flipped. Anything checkable is `site-check` or `execution`; `reasoning` is what is left when nothing was run."
95
+ }
96
+ },
20
97
  "severity": { "type": "string", "enum": ["blocker", "major", "minor", "info"] },
21
98
  "file": { "type": ["string", "null"] },
22
99
  "line": { "type": ["integer", "null"], "minimum": 1 },
@@ -30,6 +107,36 @@
30
107
  "blocking_merge": { "type": "boolean", "default": false },
31
108
  "related_skill": { "type": ["string", "null"] },
32
109
  "learning_candidate": { "type": "boolean", "default": false },
110
+ "source": {
111
+ "title": "Who raised this finding: one of our reviewers, or somebody on the pull request",
112
+ "type": "string",
113
+ "enum": ["internal", "external"],
114
+ "default": "internal",
115
+ "description": "Absent reads as `internal`, which is every reviewer-emitted finding and all 83 pre-contract records; it is written only for `external`, on the same rule as `disposition` — a property present on every record says nothing. The value is not descriptive: three mechanisms branch on it. An external finding cannot be removed by the verifier (a machine deciding a human's question was invalid is not an answer), is not truncated by the per-reviewer findings cap, and blocks `flow complete` while it has no reply. A finding that lost this property would quietly acquire all three of the behaviours the criteria forbid."
116
+ },
117
+ "external_ref": {
118
+ "title": "The GitHub comment this finding is the record of",
119
+ "type": "object",
120
+ "additionalProperties": false,
121
+ "required": ["id", "author", "url", "submitted_at"],
122
+ "description": "Every property here exists to make ONE operation possible after a session restart: replying in the right place, to the right person, exactly once. `id` is namespaced (`review-comment:12`) because the three GitHub endpoints number independently and a bare integer collides across them.",
123
+ "properties": {
124
+ "id": { "type": "string", "minLength": 1 },
125
+ "author": { "type": "string", "minLength": 1 },
126
+ "url": { "type": "string", "minLength": 1 },
127
+ "path": { "type": ["string", "null"] },
128
+ "line": { "type": ["integer", "null"], "minimum": 1 },
129
+ "thread_id": {
130
+ "type": ["string", "null"],
131
+ "description": "The root comment of the review thread. Null for a review submission body and for a PR-level comment, neither of which has a thread — which is why the reply for those is a new top-level comment and is recorded as such rather than pretending to be threaded."
132
+ },
133
+ "submitted_at": {
134
+ "type": "string",
135
+ "minLength": 1,
136
+ "description": "What decides whether a later reply from someone else makes an already-handled comment new again."
137
+ }
138
+ }
139
+ },
33
140
  "class_scope": {
34
141
  "title": "Every site of this shape, and how the set was derived",
35
142
  "description": "A finding anchored to one file:line is a claim about one site. Repeatedly, a fix repaired that site and left its siblings — one writer of five, one operator instruction of four, six readers of eight — and the next review round found them. Required for blocker and major; optional below, because enumerating the class for every info finding is theatre.",
@@ -55,5 +162,16 @@
55
162
  "required": ["severity"],
56
163
  "properties": { "severity": { "enum": ["blocker", "major"] } }
57
164
  },
58
- "then": { "required": ["class_scope"] }
165
+ "then": { "required": ["class_scope"] },
166
+ "allOf": [
167
+ {
168
+ "title": "An external finding must say which comment it is the record of",
169
+ "description": "Kept in `allOf` rather than merged into the root `if`/`then` above, because that pair is pinned identical to the bundled reviewer-finding schema by review-finding-class-scope.test.ts — and `external_ref` deliberately does NOT belong there. A reviewer never emits an external finding; only the comment collector does. A `source: external` without a ref is a finding that can never be replied to, which is the one outcome AC13 calls unacceptable.",
170
+ "if": {
171
+ "required": ["source"],
172
+ "properties": { "source": { "const": "external" } }
173
+ },
174
+ "then": { "required": ["external_ref"] }
175
+ }
176
+ ]
59
177
  }
@@ -79,17 +79,73 @@
79
79
  "max_prompt_tokens": { "type": ["integer", "null"], "minimum": 1000 },
80
80
  "max_output_tokens": { "type": ["integer", "null"], "minimum": 500 },
81
81
  "max_runtime_ms": { "type": ["integer", "null"], "minimum": 1000 },
82
- "max_findings": { "type": ["integer", "null"], "minimum": 1 }
82
+ "max_findings": { "type": ["integer", "null"], "minimum": 1, "default": 10, "description": "Findings a reviewer may report. Per reviewer, not per review: a global cap on a fan-out is spent by whichever reviewer runs first. Blockers and merge-blocking findings are exempt and consume no budget. Applies to reported findings only, never to dismissal or refutation records - truncating those would rebuild the triage-survivor corpus that made precision unmeasurable." }
83
83
  }
84
84
  },
85
85
  "model": {
86
+ "description": "Which model runs this dispatch. Skills declare a `tier`; at dispatch time src/gdskills/model-tier.ts DISCOVERS the models runtime provider detection reported for the active provider, ranks them by size markers in their names, and places the tier relative to the session's own model — `standard` is the session model, `deep` a discovered model ranked above it, `light` one ranked below. No table of model ids exists anywhere in the resolution. When the candidates cannot be ranked, every tier keeps the session's model. `provider`/`model` remain available for an explicit lateral selection, and an omitted block inherits the parent verbatim. Two things fill this block. The interactive `spawn_subagent` tool does it in code: it builds the tier map with `buildTierMap` from its host's detection result, hands it to `resolveChildModel`, and records the outcome on the dispatch's run trace. An orchestrator authoring a dispatch document RUNS `keryx review tier`, which takes the §4.4 signals as flags and prints exactly the fields below, ready to paste (`--json` prints this object alone). It is a command rather than a documented function call because an orchestrator is an agent following prose and cannot call a TypeScript function: told to `decideDispatchModel`, it computed the tier in its head from a table, which is the mechanical step that belongs in code. When nothing can be worked out the command emits `inherit: true` instead of `provider`/`model` — the dispatch runs on the caller's own model, never a downgrade. Neither path is enforced by a reader — nothing validates a recorded resolution against the model that actually ran — so these fields are a record, not a gate.",
86
87
  "type": "object",
87
88
  "additionalProperties": false,
88
89
  "properties": {
89
90
  "provider": { "type": "string", "minLength": 1 },
90
91
  "model": { "type": "string", "minLength": 1 },
91
- "tier": { "type": "string", "minLength": 1 },
92
- "inherit": { "type": "boolean" }
92
+ "tier": {
93
+ "description": "`light` is mechanical, verifiable work (pre-filter, class_scope existence checks, comment collection, formatting a reply); `standard` is ordinary reviewing and implementation and is the default; `deep` is genuinely hard reasoning (regression review across a blast radius, a strategy change after a failed loop, synthesis across many findings). An ENUM, not a free string, and that is the enforcement: a skill cannot write a model name into this field. The older `cheap` spelling frozen into docs/requirements/keryx-multi-agent-engine/schemas/child-model-selection.schema.json is still READ by parseModelTier, but is not accepted here — new dispatches use the three names above.",
94
+ "type": "string",
95
+ "enum": ["light", "standard", "deep"]
96
+ },
97
+ "tier_reasons": {
98
+ "description": "The ordered rule ids `assignTier` applied, e.g. [\"base:standard\", \"floor:blast-radius\"]. Produced by `keryx review tier` from the signals passed as flags, never written by hand: a reason list an author composes records the story they tell about the tier rather than the rules that produced it. Recorded so a tier can be explained after the run instead of re-derived from a state nobody kept. Ids rather than prose, because they are compared across runs.",
99
+ "type": "array",
100
+ "items": { "type": "string", "minLength": 1 }
101
+ },
102
+ "tier_resolution": {
103
+ "description": "Which of three outcomes produced the model, as `resolveTierModel` reports it on `TierResolution.source`. `discovered` — a model reported by runtime provider detection was assigned to this tier. `session-ranked` — ranking WORKED and placed the tier at the session's own model (always so for `standard`; for `light`/`deep` when nothing discovered ranks below/above the session). `session-fallback` — ranking was refused, or the ranking carried no rank for the session model, so the session's model is used because nothing could be worked out. Collapsing the last two would hide whether the environment was understood. Recorded, never read back: it explains a finished run and gates nothing.",
104
+ "type": "string",
105
+ "enum": ["discovered", "session-ranked", "session-fallback"]
106
+ },
107
+ "model_discovery": {
108
+ "description": "What was on the table when the tier was resolved: which models runtime detection reported for the active provider, how the size-marker hints ordered them, and — when ranking was refused — why. `tier_resolution` alone says a fallback happened and never says what it fell back FROM.",
109
+ "type": "object",
110
+ "additionalProperties": false,
111
+ "required": ["provider", "candidates", "ranked", "session_rank", "fallback_reason"],
112
+ "properties": {
113
+ "provider": {
114
+ "description": "The provider whose catalogue was consulted. Always the session's own: candidates are never taken from a provider the parent holds no grant for.",
115
+ "type": "string"
116
+ },
117
+ "candidates": {
118
+ "description": "Every model discovered for that provider, de-duplicated, in the order detection reported it. Empty when nothing was discovered.",
119
+ "type": "array",
120
+ "items": { "type": "string", "minLength": 1 }
121
+ },
122
+ "ranked": {
123
+ "description": "The subset the hints could place, best first. A candidate absent from this list carried no size marker and was therefore not orderable — recorded by its absence rather than by a guessed rank.",
124
+ "type": "array",
125
+ "items": {
126
+ "type": "object",
127
+ "additionalProperties": false,
128
+ "required": ["model", "rank"],
129
+ "properties": {
130
+ "model": { "type": "string", "minLength": 1 },
131
+ "rank": { "description": "Ordinal only. Absolute values mean nothing; only comparisons between candidates are read.", "type": "integer" }
132
+ }
133
+ }
134
+ },
135
+ "session_rank": {
136
+ "description": "Where the session's own model sits in the same ordering, or null when the hints could not place it — which is itself a refusal to rank, because nothing can be called above or below an unplaced anchor.",
137
+ "type": ["integer", "null"]
138
+ },
139
+ "fallback_reason": {
140
+ "description": "Why ranking was refused, or null when it was not. Prose, because it is read by a person reconstructing a run and never compared across runs.",
141
+ "type": ["string", "null"]
142
+ }
143
+ }
144
+ },
145
+ "inherit": {
146
+ "description": "Run this dispatch on the caller's own model. `keryx review tier` writes it — in place of `provider`/`model`, never alongside them — when the environment could not be worked out: no persisted session selection, an unrankable catalogue, or an unrankable session model. It is the recorded form of \"never a downgrade\": the alternative is an empty `provider`/`model` pair, which is schema-invalid and still reads like an answer. A `tier` and `tier_reasons` may accompany it, because the tier was assigned from signals even when no model could be named for it.",
147
+ "type": "boolean"
148
+ }
93
149
  }
94
150
  },
95
151
  "runtime": {