@mrciphersmith/keryx 0.2.69 → 0.2.71
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +11136 -4863
- package/docs/README.md +54 -0
- package/docs/requirements/shared-agent-context/README.md +104 -0
- package/package.json +3 -2
- package/src/gdgraph/build-lang.test.ts +10 -3
- package/src/gdgraph/build.ts +54 -9
- package/src/gdgraph/import-kind.test.ts +205 -0
- package/src/gdgraph/query.ts +6 -1
- package/src/gdgraph/types.ts +34 -0
- package/src/gdskills/bundled/rules/core/model-selection.mdc +184 -31
- package/src/gdskills/bundled/rules/core/skills-storage-workflow.mdc +36 -0
- package/src/gdskills/bundled/rules/core/subagent-status-protocol.md +27 -1
- package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.opencode.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/code-verifier/SKILL.zed.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/context-collector/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/context-collector/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/context-collector/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/context-collector/SKILL.opencode.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/context-collector/SKILL.zed.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/feature-analyzer/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/feature-analyzer/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/feature-analyzer/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/feature-analyzer/SKILL.opencode.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/feature-analyzer/SKILL.zed.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/feature-dev/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/feature-dev/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/feature-dev/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/flow-orchestrator/SKILL.md +159 -20
- package/src/gdskills/bundled/skills/orchestration/issue-analyzer/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/issue-analyzer/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/issue-analyzer/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/issue-analyzer/SKILL.opencode.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/issue-analyzer/SKILL.zed.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/job-documenter/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/job-documenter/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/job-documenter/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/orchestration/job-documenter/SKILL.opencode.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/job-documenter/SKILL.zed.md +1 -1
- package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.codex.md +28 -3
- package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.cursor.md +28 -3
- package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.md +28 -3
- package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.opencode.md +28 -3
- package/src/gdskills/bundled/skills/orchestration/job-orchestrator/SKILL.zed.md +28 -3
- package/src/gdskills/bundled/skills/orchestration/task-implementer/SKILL.codex.md +20 -2
- package/src/gdskills/bundled/skills/orchestration/task-implementer/SKILL.cursor.md +20 -2
- package/src/gdskills/bundled/skills/orchestration/task-implementer/SKILL.md +22 -3
- package/src/gdskills/bundled/skills/orchestration/task-implementer/SKILL.opencode.md +20 -2
- package/src/gdskills/bundled/skills/orchestration/task-implementer/SKILL.zed.md +20 -2
- package/src/gdskills/bundled/skills/planning/autodoc-analyst/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/planning/autodoc-architect/SKILL.md +3 -1
- package/src/gdskills/bundled/skills/planning/autodoc-assembler/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/planning/autodoc-orchestrator/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/planning/autodoc-scanner/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/planning/autodoc-writer/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/planning/brainstorm/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/planning/brainstorm/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/planning/brainstorm/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/planning/consistency-checker/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/planning/consistency-checker/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/planning/consistency-checker/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/planning/docpack-orchestrator/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/planning/docpack-review/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/planning/interview/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/planning/interview/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/planning/interview/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/planning/interviewer/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/planning/interviewer/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/planning/interviewer/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/planning/patterns-researcher/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/planning/patterns-researcher/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/planning/patterns-researcher/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/planning/planner/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/planning/planner/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/planning/planner/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/planning/prd-creator/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/planning/prd-creator/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/planning/prd-creator/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/planning/prd-creator/SKILL.opencode.md +1 -1
- package/src/gdskills/bundled/skills/planning/prd-creator/SKILL.zed.md +1 -1
- package/src/gdskills/bundled/skills/planning/problem-definer/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/planning/problem-definer/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/planning/problem-definer/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/planning/project-discovery/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/planning/project-discovery/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/planning/project-discovery/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/planning/spec-writer/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/planning/spec-writer/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/planning/spec-writer/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/planning/stack-advisor/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/planning/stack-advisor/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/planning/stack-advisor/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/platform/claude-md-management/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/platform/claude-md-management/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/platform/claude-md-management/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/platform/hookify/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/platform/hookify/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/platform/hookify/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/changelog/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/quality/changelog/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/quality/changelog/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/commit/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/quality/commit/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/quality/commit/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/db-migrate/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/quality/db-migrate/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/quality/db-migrate/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/dependency-update/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/quality/dependency-update/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/quality/dependency-update/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/deploy/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/quality/deploy/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/quality/deploy/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/metaproject-security/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/perf-check/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/quality/perf-check/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/quality/perf-check/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/pr/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/quality/pr/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/quality/pr/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/pr-issue-documenter/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/quality/pr-issue-documenter/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/quality/pr-issue-documenter/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/pr-issue-documenter/SKILL.opencode.md +1 -1
- package/src/gdskills/bundled/skills/quality/pr-issue-documenter/SKILL.zed.md +1 -1
- package/src/gdskills/bundled/skills/quality/push/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/quality/push/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/quality/push/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/security-audit/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/quality/security-audit/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/quality/security-audit/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/test-gen/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/quality/test-gen/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/quality/test-gen/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/tests-creator/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/quality/tests-creator/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/quality/tests-creator/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/quality/tests-creator/SKILL.opencode.md +1 -1
- package/src/gdskills/bundled/skills/quality/tests-creator/SKILL.zed.md +1 -1
- package/src/gdskills/bundled/skills/review/code-ai-review/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/review/code-ai-review/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/review/code-ai-review/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/review/code-ai-review/SKILL.opencode.md +1 -1
- package/src/gdskills/bundled/skills/review/code-ai-review/SKILL.zed.md +1 -1
- package/src/gdskills/bundled/skills/review/code-b091-review/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/review/code-b091-review/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/review/code-b091-review/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/review/code-b091-review/SKILL.opencode.md +1 -1
- package/src/gdskills/bundled/skills/review/code-b091-review/SKILL.zed.md +1 -1
- package/src/gdskills/bundled/skills/review/code-mobx-store-review/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/review/code-mobx-store-review/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/review/code-mobx-store-review/SKILL.md +2 -1
- package/src/gdskills/bundled/skills/review/code-mobx-store-review/SKILL.opencode.md +1 -1
- package/src/gdskills/bundled/skills/review/code-mobx-store-review/SKILL.zed.md +1 -1
- package/src/gdskills/bundled/skills/review/code-style-review/SKILL.codex.md +1 -1
- package/src/gdskills/bundled/skills/review/code-style-review/SKILL.cursor.md +1 -1
- package/src/gdskills/bundled/skills/review/code-style-review/SKILL.md +1 -1
- package/src/gdskills/bundled/skills/review/code-style-review/SKILL.opencode.md +1 -1
- package/src/gdskills/bundled/skills/review/code-style-review/SKILL.zed.md +1 -1
- package/src/gdskills/bundled/skills/review/review-architecture/SKILL.md +37 -10
- package/src/gdskills/bundled/skills/review/review-backend/SKILL.md +48 -14
- package/src/gdskills/bundled/skills/review/review-clean-code/SKILL.md +49 -12
- package/src/gdskills/bundled/skills/review/review-core-boundaries/SKILL.md +34 -2
- package/src/gdskills/bundled/skills/review/review-flow-graph/SKILL.md +33 -2
- package/src/gdskills/bundled/skills/review/review-frontend/SKILL.md +70 -29
- package/src/gdskills/bundled/skills/review/review-frontend-conventions/SKILL.md +34 -3
- package/src/gdskills/bundled/skills/review/review-highload/SKILL.md +49 -15
- package/src/gdskills/bundled/skills/review/review-logic/SKILL.md +39 -11
- package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.md +659 -64
- package/src/gdskills/bundled/skills/review/review-orchestrator/reviewer-finding.schema.json +7 -0
- package/src/gdskills/bundled/skills/review/review-orchestrator/verification-claim.schema.json +78 -0
- package/src/gdskills/bundled/skills/review/review-performance/SKILL.md +43 -13
- package/src/gdskills/bundled/skills/review/review-pr-feedback/SKILL.md +8 -2
- package/src/gdskills/bundled/skills/review/review-regression/SKILL.md +185 -0
- package/src/gdskills/bundled/skills/review/review-security-code/SKILL.md +44 -13
- package/src/gdskills/bundled/skills/review/review-style/SKILL.md +26 -6
- package/src/gdskills/bundled/skills/review/review-testing-practices/SKILL.md +35 -3
- package/src/gdskills/bundled/skills/review/review-verifier/SKILL.md +276 -0
- package/src/gdskills/contracts/review-finding.schema.json +119 -1
- package/src/gdskills/contracts/subagent-dispatch.schema.json +59 -3
- package/src/gdskills/bundled/skills/review/review-strict/SKILL.md +0 -328
|
@@ -0,0 +1,276 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: review-verifier
|
|
3
|
+
model_tier: light
|
|
4
|
+
description: |
|
|
5
|
+
Use when: consolidated findings from other reviewers need to be checked before they are
|
|
6
|
+
reported — by RUNNING something that fails if the finding is real, or by confirming the
|
|
7
|
+
sites the finding named actually exist. Covers "verify the findings", "check these
|
|
8
|
+
findings", "review --verify", or dispatched by review-orchestrator as Wave C.
|
|
9
|
+
This reviewer can only DELETE. It cannot raise a severity, add a finding, or change a
|
|
10
|
+
finding's text.
|
|
11
|
+
NOT for: first-pass review (run domain reviewers first); re-scoring findings by re-reading
|
|
12
|
+
them, which is the operation this skill replaced and which is measured to degrade accuracy.
|
|
13
|
+
triggers:
|
|
14
|
+
- "verify findings"
|
|
15
|
+
- "review --verify"
|
|
16
|
+
- "check these findings"
|
|
17
|
+
- "verification pass"
|
|
18
|
+
metadata:
|
|
19
|
+
author: "MrCipherSmith"
|
|
20
|
+
version: "1.0.0"
|
|
21
|
+
category: "review"
|
|
22
|
+
compatible_harnesses: "cursor,codex,zed,opencode,claude"
|
|
23
|
+
license: "MIT"
|
|
24
|
+
---
|
|
25
|
+
|
|
26
|
+
# Review — Verifier
|
|
27
|
+
|
|
28
|
+
Wave C. Reads the findings the domain reviewers produced and tries to **make each
|
|
29
|
+
one fail on purpose**. Emits one verdict per finding it checked. It removes; it
|
|
30
|
+
never adds and never sharpens.
|
|
31
|
+
|
|
32
|
+
---
|
|
33
|
+
|
|
34
|
+
## What this replaced, and why it is not coming back
|
|
35
|
+
|
|
36
|
+
This slot used to hold `review-strict`: a meta-pass that re-read the consolidated
|
|
37
|
+
findings and adjusted their severity, under an elevation table biased 3:1 toward
|
|
38
|
+
escalation, **with no new evidence**. It was removed rather than improved,
|
|
39
|
+
because the operation it performed is measured to make accuracy worse:
|
|
40
|
+
|
|
41
|
+
- **GPT-4 on GSM8K across self-correction rounds: 95.5 → 91.5 → 89.0.**
|
|
42
|
+
**GPT-3.5 on CommonSenseQA: 75.8 → 38.1.** Among the answers that changed,
|
|
43
|
+
correct → incorrect exceeded incorrect → correct (Huang et al., *Large Language
|
|
44
|
+
Models Cannot Self-Correct Reasoning Yet*, ICLR 2024, arXiv:2310.01798).
|
|
45
|
+
- **Self-Refine (arXiv:2303.17651) shows the same shape from the other side:
|
|
46
|
+
+49.2 on dialogue response generation, +0.2 on maths.** Self-refinement gains
|
|
47
|
+
live on subjective tasks and vanish on verifiable reasoning. Deciding whether a
|
|
48
|
+
null-guard is missing is verifiable reasoning.
|
|
49
|
+
|
|
50
|
+
So: re-reading a finding and changing what happens to it, without running
|
|
51
|
+
anything, is not a rigour pass. It is a coin flip weighted toward more findings.
|
|
52
|
+
If you are about to reintroduce it because it looks obviously useful — it looked
|
|
53
|
+
obviously useful the first time, and the numbers above are what happened.
|
|
54
|
+
|
|
55
|
+
## Why this one is different
|
|
56
|
+
|
|
57
|
+
It does not re-read. It runs something.
|
|
58
|
+
|
|
59
|
+
- Verification that **executes** rejects **85–96% of false reports**, against
|
|
60
|
+
**4–15% unaided**, while finding **30–44% more true bugs** (AnyPoC,
|
|
61
|
+
arXiv:2604.11950).
|
|
62
|
+
- Meta's TestGen-LLM funnel: **75% build → 57% build and pass → 25% improve
|
|
63
|
+
coverage** — and the surviving quarter reaches **73% human acceptance**
|
|
64
|
+
(arXiv:2402.09171). A hard filter that discards three quarters of its own
|
|
65
|
+
output is what makes the remainder trustworthy.
|
|
66
|
+
- **80+ agents unanimously endorsed a padding-oracle vulnerability that did not
|
|
67
|
+
exist. A single empirical test killed it.** Consensus cannot detect a
|
|
68
|
+
hallucination its members share, so this skill never votes, never polls other
|
|
69
|
+
reviewers, and never treats agreement as evidence.
|
|
70
|
+
|
|
71
|
+
---
|
|
72
|
+
|
|
73
|
+
## Workflow
|
|
74
|
+
|
|
75
|
+
```
|
|
76
|
+
review-verifier Progress:
|
|
77
|
+
- [ ] Step 1: Read the consolidated findings (global_id, reviewer, class_scope, evidence)
|
|
78
|
+
- [ ] Step 2: Drop every finding raised by yourself — you may not verify those
|
|
79
|
+
- [ ] Step 3: For each remaining finding, choose the strongest method that is actually available
|
|
80
|
+
- [ ] Step 4: Run it. Record the command and its output verbatim
|
|
81
|
+
- [ ] Step 5: Emit one claim per finding checked, conforming to verification-claim.schema.json
|
|
82
|
+
- [ ] Step 6: Emit nothing else — no new findings, no severity, no rewrites
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
---
|
|
86
|
+
|
|
87
|
+
## Input Contract
|
|
88
|
+
|
|
89
|
+
| Field | Type | Required | Description |
|
|
90
|
+
|-------|------|----------|-------------|
|
|
91
|
+
| `findings` | array | yes | The consolidated findings, each conforming to `review-finding.schema.json`. `global_id` and `reviewer` must be present. |
|
|
92
|
+
| `verifier` | string | yes | Your own name. Recorded on every claim. |
|
|
93
|
+
| `branch` / `base_sha` | string | no | So a command can be run against the change under review. |
|
|
94
|
+
| `verification_mode` | string | no | `off` \| `annotate` \| `filter`. Informational for you — the mode is applied by the merge, not by you. Default `annotate`. |
|
|
95
|
+
|
|
96
|
+
---
|
|
97
|
+
|
|
98
|
+
## Step 3 — choosing a method
|
|
99
|
+
|
|
100
|
+
Three methods, strongest first. **Take the strongest one that is genuinely
|
|
101
|
+
available, not the strongest one you can describe.**
|
|
102
|
+
|
|
103
|
+
### 1. `execution` — run something that fails if the finding is real
|
|
104
|
+
|
|
105
|
+
This is the method that works, and in this repository it is nearly always
|
|
106
|
+
available: keryx is a Bun project, so `bun test <file>`, `bun run typecheck`, or a
|
|
107
|
+
three-line script is cheap and immediate.
|
|
108
|
+
|
|
109
|
+
The test is not "does the suite pass". It is: **construct the situation the
|
|
110
|
+
finding claims is broken, and see whether it breaks.**
|
|
111
|
+
|
|
112
|
+
```bash
|
|
113
|
+
# Finding: "createManagedReviewPackage writes a package even when the contract gate fails"
|
|
114
|
+
bun test src/review/managed.test.ts -t "refuses" # does the guard actually fire?
|
|
115
|
+
|
|
116
|
+
# Finding: "this guard is asserted against a synthetic value, so it cannot fail"
|
|
117
|
+
# Delete the guarded line and re-run. If the test stays green, the finding is CONFIRMED.
|
|
118
|
+
|
|
119
|
+
# Finding: "the type allows undefined here"
|
|
120
|
+
bun run typecheck
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
Record the command and the relevant output. "I ran the tests and they passed" is
|
|
124
|
+
not evidence; the command you ran and its output is — `bun test <file>` with
|
|
125
|
+
the counts it actually printed, not a count quoted from somewhere else. (This
|
|
126
|
+
sentence used to quote a fixed number, which went stale twice in one day.)
|
|
127
|
+
|
|
128
|
+
A finding is `confirmed` when the procedure **reproduced the defect**, and
|
|
129
|
+
`refuted` when the procedure **that would have shown the defect did not**. If the
|
|
130
|
+
command you ran would not have failed either way, you have not verified anything —
|
|
131
|
+
use `unverifiable` and say what you ran.
|
|
132
|
+
|
|
133
|
+
### 2. `site-check` — do the named sites exist?
|
|
134
|
+
|
|
135
|
+
A `blocker` or `major` carries `class_scope.sites`: every location holding the
|
|
136
|
+
shape, and how the set was enumerated. Check the list.
|
|
137
|
+
|
|
138
|
+
```bash
|
|
139
|
+
keryx ctx rg "ensureKeryxConfigDir\(" src
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
Weaker than execution, and the reason is worth keeping in mind: this establishes
|
|
143
|
+
that the code the finding describes is there. It does not establish that the
|
|
144
|
+
behaviour the finding claims occurs. A site list that is right about locations and
|
|
145
|
+
wrong about consequences passes this check.
|
|
146
|
+
|
|
147
|
+
- Sites named, sites found → the finding is about real code. Usually
|
|
148
|
+
`unverifiable` unless the check itself settles the claim.
|
|
149
|
+
- Sites named, sites **absent** → `refuted`, and the evidence is the search that
|
|
150
|
+
returned nothing.
|
|
151
|
+
- The enumeration is demonstrably incomplete → that is not a refutation. The
|
|
152
|
+
finding is understated, and understatement is not yours to correct. Record
|
|
153
|
+
`confirmed` if you established the shape exists, and say so in the evidence.
|
|
154
|
+
|
|
155
|
+
### 3. `reasoning` — you could not do either
|
|
156
|
+
|
|
157
|
+
**Capped at `unverifiable`. It can never be `confirmed`, and it can never be
|
|
158
|
+
`refuted`.**
|
|
159
|
+
|
|
160
|
+
The cap is not conservatism. Reasoning alone produces no new evidence, and a
|
|
161
|
+
verdict is a claim about evidence; a "verified" verdict reached by re-reading is
|
|
162
|
+
exactly the `review-strict` operation whose numbers are at the top of this file.
|
|
163
|
+
`refuted` is capped for the same reason in the other direction: it is the one
|
|
164
|
+
verdict that removes a finding, so granting it to the weakest method would
|
|
165
|
+
reinstate the removed pass with the sign flipped.
|
|
166
|
+
|
|
167
|
+
Nothing checkable is lost by this. "The line this finding cites does not exist" is
|
|
168
|
+
a `site-check`, not reasoning. `reasoning` is the residual — the cases where you
|
|
169
|
+
ran nothing and looked up nothing — and the honest thing for the residual to say
|
|
170
|
+
is *I could not verify this*.
|
|
171
|
+
|
|
172
|
+
The merge enforces the cap: a `reasoning` claim carrying `confirmed` or `refuted`
|
|
173
|
+
is rewritten to `unverifiable`, and the attempt is recorded in the review record.
|
|
174
|
+
|
|
175
|
+
---
|
|
176
|
+
|
|
177
|
+
## Iron Laws
|
|
178
|
+
|
|
179
|
+
| Rule | Why |
|
|
180
|
+
|------|-----|
|
|
181
|
+
| **You can only delete.** No new findings, no severity changes, no edits to a finding's text. | The merge builds the record from the ORIGINAL finding and takes only your verdict. A claim carrying anything else is discarded whole, including its verdict. |
|
|
182
|
+
| **Never verify your own finding.** | The reviewer that raised a finding is the one actor whose agreement carries no information about it. Refused by the merge, by name. |
|
|
183
|
+
| **Reasoning alone is `unverifiable`.** | See above. Capped in code, not by this instruction. |
|
|
184
|
+
| **Every verdict cites what you ran.** | An unevidenced claim is discarded and the finding stays as reported. |
|
|
185
|
+
| **Never poll, never vote, never count agreement.** | 80+ agents unanimously endorsed a vulnerability that did not exist. |
|
|
186
|
+
| **`unverifiable` is a normal answer.** | Expect it to be the majority. Reaching for `refuted` to look productive is how a true blocker gets deleted. |
|
|
187
|
+
| **Not checking a finding removes nothing.** | A finding with no claim is reported unchanged. Absence is not a verdict. |
|
|
188
|
+
|
|
189
|
+
---
|
|
190
|
+
|
|
191
|
+
## Output Contract
|
|
192
|
+
|
|
193
|
+
```
|
|
194
|
+
STATUS: DONE | NEEDS_CONTEXT | BLOCKED
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
- `DONE` — every finding you were given was either checked or explicitly left alone.
|
|
198
|
+
- `NEEDS_CONTEXT` — you cannot run anything (no working tree, no test command); say what is missing.
|
|
199
|
+
- `BLOCKED` — the findings input is unreadable or carries no `global_id`.
|
|
200
|
+
|
|
201
|
+
There is no `DONE_WITH_CONCERNS`: a verifier has no concerns of its own.
|
|
202
|
+
|
|
203
|
+
Return one object conforming to
|
|
204
|
+
`skills/review-orchestrator/verification-claim.schema.json`:
|
|
205
|
+
|
|
206
|
+
````text
|
|
207
|
+
```json keryx:verifications
|
|
208
|
+
{
|
|
209
|
+
"status": "DONE",
|
|
210
|
+
"verifier": "review-verifier",
|
|
211
|
+
"summary": "12 findings, 4 executed, 3 site-checked, 5 unverifiable",
|
|
212
|
+
"verifications": [
|
|
213
|
+
{
|
|
214
|
+
"finding": "2026-08-29-pr-273#F-001",
|
|
215
|
+
"verdict": "refuted",
|
|
216
|
+
"method": "execution",
|
|
217
|
+
"evidence": "bun test src/lib/config-dir.writers.test.ts -t 'group-writable' -> 3 pass, 0 fail; measured mode 0700 under umask 002, the finding read the wrong call site"
|
|
218
|
+
},
|
|
219
|
+
{
|
|
220
|
+
"finding": "2026-08-29-pr-273#F-004",
|
|
221
|
+
"verdict": "unverifiable",
|
|
222
|
+
"method": "reasoning",
|
|
223
|
+
"evidence": "no command distinguishes the two orderings without a scheduler hook; nothing was run"
|
|
224
|
+
}
|
|
225
|
+
],
|
|
226
|
+
"stats": { "confirmed": 4, "refuted": 1, "unverifiable": 5, "not_checked": 2 }
|
|
227
|
+
}
|
|
228
|
+
```
|
|
229
|
+
````
|
|
230
|
+
|
|
231
|
+
Then a short markdown summary:
|
|
232
|
+
|
|
233
|
+
```markdown
|
|
234
|
+
# Verification Report
|
|
235
|
+
|
|
236
|
+
## Method mix
|
|
237
|
+
- execution: N
|
|
238
|
+
- site-check: N
|
|
239
|
+
- reasoning (capped to unverifiable): N
|
|
240
|
+
- not checked: N (with the reason for each)
|
|
241
|
+
|
|
242
|
+
## Refuted
|
|
243
|
+
<[global_id] finding — the command that ran, and what it returned>
|
|
244
|
+
|
|
245
|
+
## Confirmed
|
|
246
|
+
<[global_id] finding — the command that reproduced it>
|
|
247
|
+
|
|
248
|
+
## Unverifiable
|
|
249
|
+
<[global_id] finding — what was attempted and why it settled nothing>
|
|
250
|
+
```
|
|
251
|
+
|
|
252
|
+
---
|
|
253
|
+
|
|
254
|
+
## Scope Boundaries
|
|
255
|
+
|
|
256
|
+
| Concern | This skill | Use instead |
|
|
257
|
+
|---------|------------|-------------|
|
|
258
|
+
| Checking whether a reported finding is real | YES | — |
|
|
259
|
+
| Removing a finding an executed check refuted | YES | — |
|
|
260
|
+
| Raising a severity, adding a finding, rewriting one | **NO — structurally impossible** | report it as a new round |
|
|
261
|
+
| First-pass logic / frontend / backend review | NO | `review-logic`, `review-frontend`, `review-backend` |
|
|
262
|
+
| Deciding what became of a finding after the fix | NO | `keryx review complete --disposition` |
|
|
263
|
+
| Architectural commentary and opinion | NO | file it as a finding in a domain reviewer, with evidence |
|
|
264
|
+
|
|
265
|
+
---
|
|
266
|
+
|
|
267
|
+
## Red Flags
|
|
268
|
+
|
|
269
|
+
| Rationalization | Why it is wrong |
|
|
270
|
+
|----------------|-----------------|
|
|
271
|
+
| "I read it carefully and it's clearly correct — `confirmed`." | Reading is not a method. That is capped at `unverifiable`, in code. |
|
|
272
|
+
| "It's obviously a false positive, `refuted`." | Obvious to whom, on what evidence? 80+ agents found an obvious vulnerability that did not exist. |
|
|
273
|
+
| "The finding is understated; I'll bump it to blocker." | You cannot. The merge discards the whole claim and records the attempt. |
|
|
274
|
+
| "Most of my verdicts are `unverifiable`, that looks bad." | It is the expected shape. TestGen-LLM discards 75% of its own output and that is why the remainder is trusted. |
|
|
275
|
+
| "I also spotted something the reviewers missed." | Report it through a domain reviewer in a new round, with evidence. This pass cannot add. |
|
|
276
|
+
| "I raised this finding, so I know best whether it holds." | That is exactly the case AC9 forbids. |
|
|
@@ -16,7 +16,84 @@
|
|
|
16
16
|
],
|
|
17
17
|
"properties": {
|
|
18
18
|
"id": { "type": "string" },
|
|
19
|
+
"global_id": {
|
|
20
|
+
"title": "The finding's identity, unique across every review package",
|
|
21
|
+
"description": "`id` is the DISPLAY form (`F-001`) and is per-report: 15 of 43 distinct ids in the recorded corpus appear in more than one package, and `F-001` denotes six different findings, so joining a finding to a commit, a ledger row or a flow journal by `id` is unsafe for two thirds of it. `global_id` is the join key: `<reviewId>#<id>`, minted once when the finding is first recorded and CARRIED VERBATIM thereafter, so a finding re-reported in round N+1 keeps the key it was minted under in round N. Uniqueness comes from `reviewId` being a package directory name rather than from a registry, which is what makes the key recomputable from disk with no new state. Optional because 83 findings predate it and read back without one; a record written by the current pipeline always carries it. The `pattern` is not decoration: `mintGlobalFindingId` is the only writer inside this repository, but a producer outside it may supply any string, and a `global_id` that is not `<reviewId>#<id>` joins to nothing while looking like a key — the measurement in scripts/review-precision-baseline.ts splits on the separator to reach the package.",
|
|
22
|
+
"type": "string",
|
|
23
|
+
"minLength": 1,
|
|
24
|
+
"pattern": "^[^#]+#[^#]+$"
|
|
25
|
+
},
|
|
19
26
|
"reviewer": { "type": "string" },
|
|
27
|
+
"disposition": {
|
|
28
|
+
"title": "What became of this finding, and the evidence for saying so",
|
|
29
|
+
"description": "Declared HERE and deliberately NOT in the bundled reviewer-finding.schema.json: a reviewer states what is wrong, it never states what became of the finding. The measured reason this exists: precision = acted-on / (acted-on + dismissed-incorrect) computed over the recorded corpus returned 53/53 = 100%, not because the reviewers were right but because nothing on disk could express a wrong finding. `classification` looked like a validity verdict and was assigned from the ingest MODE; decisions.md said the same sentence for every finding. Only `acted-on` and `dismissed-incorrect` say anything about reviewer accuracy, which is why the three non-accuracy dismissals are separate states rather than one `dismissed` bucket — collapsing them is what makes a dismissal rate meaningless.",
|
|
30
|
+
"type": "object",
|
|
31
|
+
"additionalProperties": false,
|
|
32
|
+
"required": ["state"],
|
|
33
|
+
"properties": {
|
|
34
|
+
"state": {
|
|
35
|
+
"type": "string",
|
|
36
|
+
"enum": [
|
|
37
|
+
"unknown",
|
|
38
|
+
"acted-on",
|
|
39
|
+
"dismissed-incorrect",
|
|
40
|
+
"dismissed-wont-fix",
|
|
41
|
+
"dismissed-out-of-scope",
|
|
42
|
+
"dismissed-deprioritised",
|
|
43
|
+
"answered-disagree"
|
|
44
|
+
],
|
|
45
|
+
"default": "unknown",
|
|
46
|
+
"description": "`unknown` is the default and the reading of an absent `disposition`: nobody recorded an outcome. It is never counted as valid — an unknown counted as acted-on inflates the very figure this field exists to make measurable."
|
|
47
|
+
},
|
|
48
|
+
"evidence": {
|
|
49
|
+
"type": "string",
|
|
50
|
+
"minLength": 1,
|
|
51
|
+
"description": "Where the outcome is written down: the commit that closed it, the test that went red, the decision that dismissed it. A disposition without this is an assertion, and assertions are what produced a corpus whose findings all appear correct. `answered-disagree` is the one state that is NOT a dismissal: it records that our verifier refuted an external comment, which settles the code and settles nothing about the person who asked — the reply is still owed."
|
|
52
|
+
}
|
|
53
|
+
},
|
|
54
|
+
"if": {
|
|
55
|
+
"required": ["state"],
|
|
56
|
+
"properties": { "state": { "const": "unknown" } }
|
|
57
|
+
},
|
|
58
|
+
"else": { "required": ["evidence"] }
|
|
59
|
+
},
|
|
60
|
+
"verification": {
|
|
61
|
+
"title": "What an independent check found when it went looking for this finding",
|
|
62
|
+
"description": "Declared HERE and deliberately NOT in the bundled reviewer-finding.schema.json, on the same basis as `disposition`: a reviewer states what is wrong, and whether someone else could reproduce it is not something the reviewer knows. The rule is sharper for this field than for `disposition`, because the reviewer that raised a finding is the ONE actor forbidden to answer it (AC9) — declaring the property in the shape reviewers emit would put the forbidden field next to `severity` in every reviewer's output. This replaces `review-strict`, which re-read findings and adjusted severity with no new evidence: intrinsic self-correction is measured to degrade accuracy (GPT-4 on GSM8K 95.5 -> 91.5 -> 89.0 across rounds; GPT-3.5 on CommonSenseQA 75.8 -> 38.1; Huang et al., ICLR 2024, arXiv:2310.01798). An ABSENT verification means nobody checked — which is true of all 83 pre-contract findings on disk — and is never a reason to drop a finding.",
|
|
63
|
+
"type": "object",
|
|
64
|
+
"additionalProperties": false,
|
|
65
|
+
"required": ["verdict", "method", "evidence"],
|
|
66
|
+
"properties": {
|
|
67
|
+
"verdict": {
|
|
68
|
+
"type": "string",
|
|
69
|
+
"enum": ["confirmed", "refuted", "unverifiable"],
|
|
70
|
+
"description": "`unverifiable` is not a failure state; it is the honest majority answer when nothing could be run. Only `refuted` removes anything, and only under verification_mode: filter."
|
|
71
|
+
},
|
|
72
|
+
"method": {
|
|
73
|
+
"type": "string",
|
|
74
|
+
"enum": ["execution", "site-check", "reasoning"],
|
|
75
|
+
"description": "Strongest first. `execution` ran a command or test that FAILS IF THE FINDING IS REAL — verification that executes rejects 85-96% of false reports against 4-15% unaided while finding 30-44% more true bugs (AnyPoC, arXiv:2604.11950). `site-check` confirmed the class_scope sites exist. `reasoning` is the residual, and it is capped: see the conditional below."
|
|
76
|
+
},
|
|
77
|
+
"evidence": {
|
|
78
|
+
"type": "string",
|
|
79
|
+
"minLength": 1,
|
|
80
|
+
"description": "The command that ran and what it returned, or the search that enumerated the sites. A verdict with nothing behind it is the operation this replaces."
|
|
81
|
+
},
|
|
82
|
+
"verifier": {
|
|
83
|
+
"type": "string",
|
|
84
|
+
"minLength": 1,
|
|
85
|
+
"description": "Who checked. Beyond the three properties AC7 names, and present because AC9 is otherwise untraceable after the fact: the merge refuses a claim whose verifier equals the finding's `reviewer`, but a record that does not say who verified cannot be audited for that rule — the same defect as `reviewer` being hardcoded to `review-orchestrator` on all 83 recorded findings."
|
|
86
|
+
}
|
|
87
|
+
},
|
|
88
|
+
"if": {
|
|
89
|
+
"required": ["method"],
|
|
90
|
+
"properties": { "method": { "const": "reasoning" } }
|
|
91
|
+
},
|
|
92
|
+
"then": {
|
|
93
|
+
"properties": { "verdict": { "const": "unverifiable" } },
|
|
94
|
+
"description": "Reasoning alone produces no new evidence, so it can never reach `confirmed` (AC7). It cannot reach `refuted` either, and that extension is deliberate: `refuted` is the only verdict with a destructive consequence, so granting it to the weakest method would reinstate review-strict with the sign flipped. Anything checkable is `site-check` or `execution`; `reasoning` is what is left when nothing was run."
|
|
95
|
+
}
|
|
96
|
+
},
|
|
20
97
|
"severity": { "type": "string", "enum": ["blocker", "major", "minor", "info"] },
|
|
21
98
|
"file": { "type": ["string", "null"] },
|
|
22
99
|
"line": { "type": ["integer", "null"], "minimum": 1 },
|
|
@@ -30,6 +107,36 @@
|
|
|
30
107
|
"blocking_merge": { "type": "boolean", "default": false },
|
|
31
108
|
"related_skill": { "type": ["string", "null"] },
|
|
32
109
|
"learning_candidate": { "type": "boolean", "default": false },
|
|
110
|
+
"source": {
|
|
111
|
+
"title": "Who raised this finding: one of our reviewers, or somebody on the pull request",
|
|
112
|
+
"type": "string",
|
|
113
|
+
"enum": ["internal", "external"],
|
|
114
|
+
"default": "internal",
|
|
115
|
+
"description": "Absent reads as `internal`, which is every reviewer-emitted finding and all 83 pre-contract records; it is written only for `external`, on the same rule as `disposition` — a property present on every record says nothing. The value is not descriptive: three mechanisms branch on it. An external finding cannot be removed by the verifier (a machine deciding a human's question was invalid is not an answer), is not truncated by the per-reviewer findings cap, and blocks `flow complete` while it has no reply. A finding that lost this property would quietly acquire all three of the behaviours the criteria forbid."
|
|
116
|
+
},
|
|
117
|
+
"external_ref": {
|
|
118
|
+
"title": "The GitHub comment this finding is the record of",
|
|
119
|
+
"type": "object",
|
|
120
|
+
"additionalProperties": false,
|
|
121
|
+
"required": ["id", "author", "url", "submitted_at"],
|
|
122
|
+
"description": "Every property here exists to make ONE operation possible after a session restart: replying in the right place, to the right person, exactly once. `id` is namespaced (`review-comment:12`) because the three GitHub endpoints number independently and a bare integer collides across them.",
|
|
123
|
+
"properties": {
|
|
124
|
+
"id": { "type": "string", "minLength": 1 },
|
|
125
|
+
"author": { "type": "string", "minLength": 1 },
|
|
126
|
+
"url": { "type": "string", "minLength": 1 },
|
|
127
|
+
"path": { "type": ["string", "null"] },
|
|
128
|
+
"line": { "type": ["integer", "null"], "minimum": 1 },
|
|
129
|
+
"thread_id": {
|
|
130
|
+
"type": ["string", "null"],
|
|
131
|
+
"description": "The root comment of the review thread. Null for a review submission body and for a PR-level comment, neither of which has a thread — which is why the reply for those is a new top-level comment and is recorded as such rather than pretending to be threaded."
|
|
132
|
+
},
|
|
133
|
+
"submitted_at": {
|
|
134
|
+
"type": "string",
|
|
135
|
+
"minLength": 1,
|
|
136
|
+
"description": "What decides whether a later reply from someone else makes an already-handled comment new again."
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
},
|
|
33
140
|
"class_scope": {
|
|
34
141
|
"title": "Every site of this shape, and how the set was derived",
|
|
35
142
|
"description": "A finding anchored to one file:line is a claim about one site. Repeatedly, a fix repaired that site and left its siblings — one writer of five, one operator instruction of four, six readers of eight — and the next review round found them. Required for blocker and major; optional below, because enumerating the class for every info finding is theatre.",
|
|
@@ -55,5 +162,16 @@
|
|
|
55
162
|
"required": ["severity"],
|
|
56
163
|
"properties": { "severity": { "enum": ["blocker", "major"] } }
|
|
57
164
|
},
|
|
58
|
-
"then": { "required": ["class_scope"] }
|
|
165
|
+
"then": { "required": ["class_scope"] },
|
|
166
|
+
"allOf": [
|
|
167
|
+
{
|
|
168
|
+
"title": "An external finding must say which comment it is the record of",
|
|
169
|
+
"description": "Kept in `allOf` rather than merged into the root `if`/`then` above, because that pair is pinned identical to the bundled reviewer-finding schema by review-finding-class-scope.test.ts — and `external_ref` deliberately does NOT belong there. A reviewer never emits an external finding; only the comment collector does. A `source: external` without a ref is a finding that can never be replied to, which is the one outcome AC13 calls unacceptable.",
|
|
170
|
+
"if": {
|
|
171
|
+
"required": ["source"],
|
|
172
|
+
"properties": { "source": { "const": "external" } }
|
|
173
|
+
},
|
|
174
|
+
"then": { "required": ["external_ref"] }
|
|
175
|
+
}
|
|
176
|
+
]
|
|
59
177
|
}
|
|
@@ -79,17 +79,73 @@
|
|
|
79
79
|
"max_prompt_tokens": { "type": ["integer", "null"], "minimum": 1000 },
|
|
80
80
|
"max_output_tokens": { "type": ["integer", "null"], "minimum": 500 },
|
|
81
81
|
"max_runtime_ms": { "type": ["integer", "null"], "minimum": 1000 },
|
|
82
|
-
"max_findings": { "type": ["integer", "null"], "minimum": 1 }
|
|
82
|
+
"max_findings": { "type": ["integer", "null"], "minimum": 1, "default": 10, "description": "Findings a reviewer may report. Per reviewer, not per review: a global cap on a fan-out is spent by whichever reviewer runs first. Blockers and merge-blocking findings are exempt and consume no budget. Applies to reported findings only, never to dismissal or refutation records - truncating those would rebuild the triage-survivor corpus that made precision unmeasurable." }
|
|
83
83
|
}
|
|
84
84
|
},
|
|
85
85
|
"model": {
|
|
86
|
+
"description": "Which model runs this dispatch. Skills declare a `tier`; at dispatch time src/gdskills/model-tier.ts DISCOVERS the models runtime provider detection reported for the active provider, ranks them by size markers in their names, and places the tier relative to the session's own model — `standard` is the session model, `deep` a discovered model ranked above it, `light` one ranked below. No table of model ids exists anywhere in the resolution. When the candidates cannot be ranked, every tier keeps the session's model. `provider`/`model` remain available for an explicit lateral selection, and an omitted block inherits the parent verbatim. Two things fill this block. The interactive `spawn_subagent` tool does it in code: it builds the tier map with `buildTierMap` from its host's detection result, hands it to `resolveChildModel`, and records the outcome on the dispatch's run trace. An orchestrator authoring a dispatch document RUNS `keryx review tier`, which takes the §4.4 signals as flags and prints exactly the fields below, ready to paste (`--json` prints this object alone). It is a command rather than a documented function call because an orchestrator is an agent following prose and cannot call a TypeScript function: told to `decideDispatchModel`, it computed the tier in its head from a table, which is the mechanical step that belongs in code. When nothing can be worked out the command emits `inherit: true` instead of `provider`/`model` — the dispatch runs on the caller's own model, never a downgrade. Neither path is enforced by a reader — nothing validates a recorded resolution against the model that actually ran — so these fields are a record, not a gate.",
|
|
86
87
|
"type": "object",
|
|
87
88
|
"additionalProperties": false,
|
|
88
89
|
"properties": {
|
|
89
90
|
"provider": { "type": "string", "minLength": 1 },
|
|
90
91
|
"model": { "type": "string", "minLength": 1 },
|
|
91
|
-
"tier": {
|
|
92
|
-
|
|
92
|
+
"tier": {
|
|
93
|
+
"description": "`light` is mechanical, verifiable work (pre-filter, class_scope existence checks, comment collection, formatting a reply); `standard` is ordinary reviewing and implementation and is the default; `deep` is genuinely hard reasoning (regression review across a blast radius, a strategy change after a failed loop, synthesis across many findings). An ENUM, not a free string, and that is the enforcement: a skill cannot write a model name into this field. The older `cheap` spelling frozen into docs/requirements/keryx-multi-agent-engine/schemas/child-model-selection.schema.json is still READ by parseModelTier, but is not accepted here — new dispatches use the three names above.",
|
|
94
|
+
"type": "string",
|
|
95
|
+
"enum": ["light", "standard", "deep"]
|
|
96
|
+
},
|
|
97
|
+
"tier_reasons": {
|
|
98
|
+
"description": "The ordered rule ids `assignTier` applied, e.g. [\"base:standard\", \"floor:blast-radius\"]. Produced by `keryx review tier` from the signals passed as flags, never written by hand: a reason list an author composes records the story they tell about the tier rather than the rules that produced it. Recorded so a tier can be explained after the run instead of re-derived from a state nobody kept. Ids rather than prose, because they are compared across runs.",
|
|
99
|
+
"type": "array",
|
|
100
|
+
"items": { "type": "string", "minLength": 1 }
|
|
101
|
+
},
|
|
102
|
+
"tier_resolution": {
|
|
103
|
+
"description": "Which of three outcomes produced the model, as `resolveTierModel` reports it on `TierResolution.source`. `discovered` — a model reported by runtime provider detection was assigned to this tier. `session-ranked` — ranking WORKED and placed the tier at the session's own model (always so for `standard`; for `light`/`deep` when nothing discovered ranks below/above the session). `session-fallback` — ranking was refused, or the ranking carried no rank for the session model, so the session's model is used because nothing could be worked out. Collapsing the last two would hide whether the environment was understood. Recorded, never read back: it explains a finished run and gates nothing.",
|
|
104
|
+
"type": "string",
|
|
105
|
+
"enum": ["discovered", "session-ranked", "session-fallback"]
|
|
106
|
+
},
|
|
107
|
+
"model_discovery": {
|
|
108
|
+
"description": "What was on the table when the tier was resolved: which models runtime detection reported for the active provider, how the size-marker hints ordered them, and — when ranking was refused — why. `tier_resolution` alone says a fallback happened and never says what it fell back FROM.",
|
|
109
|
+
"type": "object",
|
|
110
|
+
"additionalProperties": false,
|
|
111
|
+
"required": ["provider", "candidates", "ranked", "session_rank", "fallback_reason"],
|
|
112
|
+
"properties": {
|
|
113
|
+
"provider": {
|
|
114
|
+
"description": "The provider whose catalogue was consulted. Always the session's own: candidates are never taken from a provider the parent holds no grant for.",
|
|
115
|
+
"type": "string"
|
|
116
|
+
},
|
|
117
|
+
"candidates": {
|
|
118
|
+
"description": "Every model discovered for that provider, de-duplicated, in the order detection reported it. Empty when nothing was discovered.",
|
|
119
|
+
"type": "array",
|
|
120
|
+
"items": { "type": "string", "minLength": 1 }
|
|
121
|
+
},
|
|
122
|
+
"ranked": {
|
|
123
|
+
"description": "The subset the hints could place, best first. A candidate absent from this list carried no size marker and was therefore not orderable — recorded by its absence rather than by a guessed rank.",
|
|
124
|
+
"type": "array",
|
|
125
|
+
"items": {
|
|
126
|
+
"type": "object",
|
|
127
|
+
"additionalProperties": false,
|
|
128
|
+
"required": ["model", "rank"],
|
|
129
|
+
"properties": {
|
|
130
|
+
"model": { "type": "string", "minLength": 1 },
|
|
131
|
+
"rank": { "description": "Ordinal only. Absolute values mean nothing; only comparisons between candidates are read.", "type": "integer" }
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
},
|
|
135
|
+
"session_rank": {
|
|
136
|
+
"description": "Where the session's own model sits in the same ordering, or null when the hints could not place it — which is itself a refusal to rank, because nothing can be called above or below an unplaced anchor.",
|
|
137
|
+
"type": ["integer", "null"]
|
|
138
|
+
},
|
|
139
|
+
"fallback_reason": {
|
|
140
|
+
"description": "Why ranking was refused, or null when it was not. Prose, because it is read by a person reconstructing a run and never compared across runs.",
|
|
141
|
+
"type": ["string", "null"]
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
},
|
|
145
|
+
"inherit": {
|
|
146
|
+
"description": "Run this dispatch on the caller's own model. `keryx review tier` writes it — in place of `provider`/`model`, never alongside them — when the environment could not be worked out: no persisted session selection, an unrankable catalogue, or an unrankable session model. It is the recorded form of \"never a downgrade\": the alternative is an empty `provider`/`model` pair, which is schema-invalid and still reads like an answer. A `tier` and `tier_reasons` may accompany it, because the tier was assigned from signals even when no model could be named for it.",
|
|
147
|
+
"type": "boolean"
|
|
148
|
+
}
|
|
93
149
|
}
|
|
94
150
|
},
|
|
95
151
|
"runtime": {
|