@mrciphersmith/keryx 0.3.1 → 0.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +3435 -785
- package/dist/core.js +50 -0
- package/package.json +1 -1
- package/src/gdskills/bundled/skills/review/review-jev-comments/SKILL.md +184 -0
- package/src/gdskills/bundled/skills/review/review-jev-docs/SKILL.md +189 -0
- package/src/gdskills/bundled/skills/review/review-jev-risk/SKILL.md +190 -0
- package/src/gdskills/bundled/skills/review/review-jev-scenarios/SKILL.md +187 -0
- package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.detail.md +28 -15
- package/src/gdskills/bundled/skills/review/review-orchestrator/SKILL.md +1 -1
package/dist/core.js
CHANGED
|
@@ -8179,6 +8179,18 @@ var init_catalog = __esm(() => {
|
|
|
8179
8179
|
"Dispatch specialized review passes conceptually or as separate skill loads.",
|
|
8180
8180
|
"Report findings first, ordered by severity, with concrete file references."
|
|
8181
8181
|
]),
|
|
8182
|
+
skill("review-jev-docs", "review", ["recommended", "full"], "Find documentation sections that went stale because of a diff, scored by Jev.", [
|
|
8183
|
+
"Split every discovered documentation source into sections by heading, deterministically.",
|
|
8184
|
+
"Link a section to changed code by an explicit path/symbol/verb it mentions.",
|
|
8185
|
+
"Ask Jev one noul question per linked section; keryx writes every word of the finding.",
|
|
8186
|
+
"Flag a removed/renamed CLI flag still mentioned in docs with no Jev call at all."
|
|
8187
|
+
]),
|
|
8188
|
+
skill("review-jev-comments", "review", ["recommended", "full"], "Check whether open PR review comments were addressed, scored by Jev.", [
|
|
8189
|
+
"Read open comments from the existing ledger, never re-collect from GitHub.",
|
|
8190
|
+
"Compute commits-after-comment, thread-resolved, and reply facts deterministically.",
|
|
8191
|
+
"Ask Jev one choice question per open comment; keryx writes every word of the finding.",
|
|
8192
|
+
"Never post a reply or resolve a thread \u2014 the label is advisory only."
|
|
8193
|
+
]),
|
|
8182
8194
|
skill("review-logic", "review", ["recommended", "full"], "Review logic correctness, contracts, edge cases, nullability, and async behavior.", [
|
|
8183
8195
|
"Trace behavior through call sites and affected context.",
|
|
8184
8196
|
"Look for incorrect assumptions, missing branches, race conditions, and error paths.",
|
|
@@ -8263,6 +8275,16 @@ var init_catalog = __esm(() => {
|
|
|
8263
8275
|
"Fall back to confirming the class_scope sites exist; reasoning alone is capped at unverifiable.",
|
|
8264
8276
|
"Emit one verdict per finding checked; never add a finding, raise a severity, or edit a finding's text."
|
|
8265
8277
|
]),
|
|
8278
|
+
skill("review-jev-risk", "review", ["recommended", "full"], "Risk map of a diff's hunks \u2014 deterministic facts plus one Jev noul per risk dimension, ranked, with a routing hint for security/concurrency.", [
|
|
8279
|
+
"Score every retained hunk on security, data/migration, public-API, concurrency, and error-handling.",
|
|
8280
|
+
"Emit a finding only above threshold and with no nearby test; severity capped at info/minor.",
|
|
8281
|
+
"Feed a routing hint for review-security-code/review-highload when a hunk crosses threshold on that dimension."
|
|
8282
|
+
]),
|
|
8283
|
+
skill("review-jev-scenarios", "review", ["recommended", "full"], "Functional review \u2014 which user scenarios a diff likely changes, from gdwiki/PRD/README sources plus one Jev noul per touched scenario.", [
|
|
8284
|
+
"Discover scenarios from gdwiki user-scenario pages, PRD requirement sections, and README/docs how-to sections.",
|
|
8285
|
+
"Ask only scenarios whose linked code this diff touches.",
|
|
8286
|
+
"Emit the ranked manual-check list, and a minor finding for a likely-affected scenario with no covering test."
|
|
8287
|
+
]),
|
|
8266
8288
|
skill("review-frontend-conventions", "review", ["recommended", "full"], "Review frontend code against repository-local frontend conventions and agent entrypoints.", [
|
|
8267
8289
|
"Load local AGENTS.md/CLAUDE.md and matched frontend rules.",
|
|
8268
8290
|
"Check component, state, styling, i18n, error, and Storybook conventions.",
|
|
@@ -8632,7 +8654,11 @@ var init_skill_length_ceilings = __esm(() => {
|
|
|
8632
8654
|
"review/review-frontend": 634,
|
|
8633
8655
|
"review/review-frontend-conventions": 213,
|
|
8634
8656
|
"review/review-highload": 550,
|
|
8657
|
+
"review/review-jev-comments": 184,
|
|
8658
|
+
"review/review-jev-docs": 189,
|
|
8659
|
+
"review/review-jev-risk": 190,
|
|
8635
8660
|
"review/review-jev-rules": 267,
|
|
8661
|
+
"review/review-jev-scenarios": 187,
|
|
8636
8662
|
"review/review-layout": 238,
|
|
8637
8663
|
"review/review-logic": 377,
|
|
8638
8664
|
"review/review-orchestrator": 1723,
|
|
@@ -37127,12 +37153,36 @@ var HELP_GROUPS = [
|
|
|
37127
37153
|
group: "Managed work",
|
|
37128
37154
|
summary: "Check a PR, a review report, or a diff against a reference document's clauses, with Jev."
|
|
37129
37155
|
},
|
|
37156
|
+
{
|
|
37157
|
+
kind: "slash",
|
|
37158
|
+
name: "/risk",
|
|
37159
|
+
group: "Managed work",
|
|
37160
|
+
summary: "Risk map of the working diff's hunks \u2014 deterministic facts plus Jev per risk dimension, ranked, highest risk first."
|
|
37161
|
+
},
|
|
37162
|
+
{
|
|
37163
|
+
kind: "slash",
|
|
37164
|
+
name: "/scenarios",
|
|
37165
|
+
group: "Managed work",
|
|
37166
|
+
summary: "Which user scenarios the working diff likely changes \u2014 deterministic scenario/code links plus Jev, ranked."
|
|
37167
|
+
},
|
|
37130
37168
|
{
|
|
37131
37169
|
kind: "slash",
|
|
37132
37170
|
name: "/jevrules",
|
|
37133
37171
|
group: "Managed work",
|
|
37134
37172
|
summary: "Check the working diff's hunks against every applicable project rule clause, with Jev, grouped by rule."
|
|
37135
37173
|
},
|
|
37174
|
+
{
|
|
37175
|
+
kind: "slash",
|
|
37176
|
+
name: "/staledocs",
|
|
37177
|
+
group: "Managed work",
|
|
37178
|
+
summary: "List doc sections that likely went stale because of the working diff, with Jev."
|
|
37179
|
+
},
|
|
37180
|
+
{
|
|
37181
|
+
kind: "slash",
|
|
37182
|
+
name: "/opencomments",
|
|
37183
|
+
group: "Managed work",
|
|
37184
|
+
summary: "List open PR review comments with a Jev resolved/still-open/escalation label \u2014 /opencomments <owner/repo> <pr>."
|
|
37185
|
+
},
|
|
37136
37186
|
{
|
|
37137
37187
|
kind: "slash",
|
|
37138
37188
|
name: "/guard",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@mrciphersmith/keryx",
|
|
3
|
-
"version": "0.3.
|
|
3
|
+
"version": "0.3.2",
|
|
4
4
|
"description": "Version-controlled project context for AI coding agents: code graph, architecture wiki, project memory, relevant tests, quality signals, and task flows.",
|
|
5
5
|
"private": false,
|
|
6
6
|
"publishConfig": {
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: review-jev-comments
|
|
3
|
+
model_tier: light
|
|
4
|
+
description: |
|
|
5
|
+
Use when: an ADDITIONAL, machine-scored pass is wanted over open PR review comments
|
|
6
|
+
already in the existing ledger — never replacing any other reviewer, and never
|
|
7
|
+
auto-replying or auto-resolving. Dispatched by review-orchestrator in Wave B, via the
|
|
8
|
+
CLI (`keryx review jev-comments`), when `review.jev.comments: true` in
|
|
9
|
+
.metaproject/tasks.config.json and a Jev/OpenRouter credential is resolvable. It is not
|
|
10
|
+
an LLM sub-agent: there is nothing to dispatch through a platform-native agent
|
|
11
|
+
mechanism, and no prose is generated by a model — Jev only answers a `choice` among
|
|
12
|
+
resolved-by-fix/still-open/not-actionable/needs-escalation per open comment; keryx
|
|
13
|
+
writes every word of every finding.
|
|
14
|
+
NOT for: posting a reply, resolving a thread, or deciding what `keryx review comments
|
|
15
|
+
reply` sends — this reviewer only ever reads the ledger and labels.
|
|
16
|
+
triggers:
|
|
17
|
+
- "jev comments"
|
|
18
|
+
- "open comments"
|
|
19
|
+
- "review --jev-comments"
|
|
20
|
+
metadata:
|
|
21
|
+
author: "MrCipherSmith"
|
|
22
|
+
version: "1.0.0"
|
|
23
|
+
category: "review"
|
|
24
|
+
compatible_harnesses: "cursor,codex,zed,opencode,claude"
|
|
25
|
+
engine: "jev"
|
|
26
|
+
origin: "authored"
|
|
27
|
+
license: "MIT"
|
|
28
|
+
---
|
|
29
|
+
|
|
30
|
+
# Review — Jev Comments (open PR comment triage)
|
|
31
|
+
|
|
32
|
+
An ADDITIONAL orchestrator reviewer, flow 333. It is a **keryx program**, not an
|
|
33
|
+
LLM sub-agent: `review-orchestrator` never dispatches a platform-native agent
|
|
34
|
+
for it, it runs `keryx review jev-comments` and reads the `--json` output.
|
|
35
|
+
Every finding is composed by keryx; Jev ("System One" on OpenRouter) supplies
|
|
36
|
+
only a `choice` label per comment — it never writes prose, and nothing it
|
|
37
|
+
returns is quoted verbatim into a finding.
|
|
38
|
+
|
|
39
|
+
---
|
|
40
|
+
|
|
41
|
+
## What it does
|
|
42
|
+
|
|
43
|
+
1. Reads open comments from the EXISTING ledger
|
|
44
|
+
(`.metaproject/reviews/pr-comments/*.json`, written by `keryx review
|
|
45
|
+
comments collect`) — never re-collects from GitHub itself.
|
|
46
|
+
2. Computes deterministic facts per open comment: commits after the comment's
|
|
47
|
+
timestamp touching its file (local `git log`, capped), the review thread's
|
|
48
|
+
resolved flag (best-effort, via one read-only GraphQL query — degrades to
|
|
49
|
+
`"unknown"` on any failure, never a guess), and its replies already on the
|
|
50
|
+
ledger.
|
|
51
|
+
3. Asks Jev exactly ONE `choice` question per open comment (batched under the
|
|
52
|
+
vendor's 64k token budget): `resolved-by-fix` / `still-open` /
|
|
53
|
+
`not-actionable` / `needs-escalation`, given the comment, its replies, and
|
|
54
|
+
the code's current hunk(s) at that location.
|
|
55
|
+
4. Synthesizes findings for `still-open` (severity `minor`) and
|
|
56
|
+
`needs-escalation` (severity `major`) only — `resolved-by-fix` and
|
|
57
|
+
`not-actionable` produce no finding. It never posts a reply and never
|
|
58
|
+
resolves a thread; the choice for every open comment is also written to an
|
|
59
|
+
advisory-label cache
|
|
60
|
+
(`.metaproject/data/review-jev-comments/labels__<repo>__<n>.json`) that
|
|
61
|
+
`keryx review comments reply` reads back and prints, informationally, next
|
|
62
|
+
to its own output — it never changes what that command sends.
|
|
63
|
+
|
|
64
|
+
---
|
|
65
|
+
|
|
66
|
+
## Input Contract
|
|
67
|
+
|
|
68
|
+
Not dispatched with a prompt — invoked as a CLI command:
|
|
69
|
+
|
|
70
|
+
```text
|
|
71
|
+
keryx review jev-comments --pr <n> --repo <owner/repo>
|
|
72
|
+
[--model <jev-1.13|jev-latest>] [--fixtures <dir>] [--json]
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
Requires a comment ledger to already exist for `--repo`/`--pr` (`keryx review
|
|
76
|
+
comments collect` first) — this reviewer refuses, before any read, when it
|
|
77
|
+
does not.
|
|
78
|
+
|
|
79
|
+
---
|
|
80
|
+
|
|
81
|
+
## Opt-in and privacy
|
|
82
|
+
|
|
83
|
+
Refuses before any ledger read or network call unless
|
|
84
|
+
`.metaproject/tasks.config.json` declares:
|
|
85
|
+
|
|
86
|
+
```json
|
|
87
|
+
{ "review": { "jev": { "comments": true } } }
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Every comment body, reply, and hunk sent to Jev is redacted first through
|
|
91
|
+
`src/security/service.ts` — the same floor every other Jev-backed reviewer
|
|
92
|
+
applies. No cache or output file stores a credential.
|
|
93
|
+
|
|
94
|
+
---
|
|
95
|
+
|
|
96
|
+
## Output Contract
|
|
97
|
+
|
|
98
|
+
Emits a `REVIEW_RESULT`-shaped object matching
|
|
99
|
+
`.metaproject/skills/gdskills/review/review-orchestrator/reviewer-finding.schema.json`
|
|
100
|
+
— `status`, `reviewer: "review-jev-comments"`, `summary`, `findings`, `stats`
|
|
101
|
+
— under `--json`. Its findings merge into the consolidated array exactly like
|
|
102
|
+
any other reviewer's: same Quality Gate, same dedup, same Wave C
|
|
103
|
+
verification.
|
|
104
|
+
|
|
105
|
+
### Class scope — required for `blocker` and `major`
|
|
106
|
+
|
|
107
|
+
A `needs-escalation` finding is `major` and always carries `class_scope`:
|
|
108
|
+
`sites` is the single `file:line` the comment is anchored to, and
|
|
109
|
+
`enumeration_method` states it is exactly that one comment — there is no
|
|
110
|
+
"class" beyond it, one comment is one finding, never a set to enumerate
|
|
111
|
+
further. `still-open` is `minor` and carries none, matching the schema's own
|
|
112
|
+
rule (`class_scope` required only for `blocker`/`major`).
|
|
113
|
+
|
|
114
|
+
---
|
|
115
|
+
|
|
116
|
+
## Orchestrator integration
|
|
117
|
+
|
|
118
|
+
- `keryx review reviewers --json` lists it under `bundled` with
|
|
119
|
+
`"engine": "jev"`.
|
|
120
|
+
- `review-orchestrator` dispatches it in **Wave B**, alongside the domain/
|
|
121
|
+
convention reviewers, when the opt-in is on and a Jev/OpenRouter credential
|
|
122
|
+
resolves (`resolveJevApiKeyResolution`). When either is false it is
|
|
123
|
+
**skipped with the stated reason**, recorded in `Skipped reviewers` —
|
|
124
|
+
never silently absent.
|
|
125
|
+
- It is **ADDITIONAL** and strictly advisory over the comment-reply flow: its
|
|
126
|
+
label is printed by `keryx review comments reply`, never acted on
|
|
127
|
+
automatically.
|
|
128
|
+
|
|
129
|
+
---
|
|
130
|
+
|
|
131
|
+
## Red Flags
|
|
132
|
+
|
|
133
|
+
| Rationalization | Why it is wrong |
|
|
134
|
+
|---|---|
|
|
135
|
+
| "Jev said still-open, so post a reply now" | This reviewer never posts anything — a `still-open`/`needs-escalation` finding is a signal for a human or `keryx review comments reply`'s own logic, not an instruction to act |
|
|
136
|
+
| "The thread's resolved flag came back unknown, so treat it as resolved" | `unknown` is a distinct, honest state — it means the read failed, not that the thread is closed |
|
|
137
|
+
| "No findings means every comment was addressed" | Only `still-open`/`needs-escalation` produce findings; `resolved-by-fix` and `not-actionable` are correctly silent, but check `openComments` in `--json` to see how many were actually judged |
|
|
138
|
+
|
|
139
|
+
---
|
|
140
|
+
|
|
141
|
+
## Verification
|
|
142
|
+
|
|
143
|
+
Before trusting a run's findings:
|
|
144
|
+
|
|
145
|
+
1. Read `openComments` in the `--json` output against the ledger's own
|
|
146
|
+
`unanswered so far` count from the last `comments collect` — they should
|
|
147
|
+
agree.
|
|
148
|
+
2. Spot-check a `needs-escalation` finding against the comment text: does it
|
|
149
|
+
actually block progress, or is it a `still-open` case Jev over-called?
|
|
150
|
+
3. Confirm every finding carries `reviewer: "review-jev-comments"` and a
|
|
151
|
+
`dedupe_key` — both are required for the orchestrator's Quality Gate and
|
|
152
|
+
Wave C verification to route it correctly.
|
|
153
|
+
|
|
154
|
+
---
|
|
155
|
+
|
|
156
|
+
## Iron Laws
|
|
157
|
+
|
|
158
|
+
### Shared laws (every reviewer)
|
|
159
|
+
|
|
160
|
+
1. **A claim of runtime harm with no reproducible path is `info`.** If you cannot
|
|
161
|
+
name the input, call, or condition that reaches the code, you have an
|
|
162
|
+
observation, not a finding. Report it as `info` and say what would settle it.
|
|
163
|
+
2. **Never flag the theoretical.** The path you describe must exist in the code
|
|
164
|
+
under review. Do not report a safe API because it could be misused, or a
|
|
165
|
+
pattern because it is often wrong elsewhere.
|
|
166
|
+
3. **One finding per class, not one per occurrence.** When the same shape appears
|
|
167
|
+
at several sites, report it once and list every site. Ten findings that are one
|
|
168
|
+
finding hide the other nine problems.
|
|
169
|
+
|
|
170
|
+
Severity levels are defined once, in `review-orchestrator/SKILL.md` →
|
|
171
|
+
**Severity (canonical)**. This reviewer does not restate them: it emits only
|
|
172
|
+
`minor` (`still-open`) or `major` (`needs-escalation`, always with
|
|
173
|
+
`class_scope`) — `blocker` never applies to an unanswered comment.
|
|
174
|
+
|
|
175
|
+
---
|
|
176
|
+
|
|
177
|
+
## Scope Boundaries
|
|
178
|
+
|
|
179
|
+
| Concern | This skill | Use instead |
|
|
180
|
+
|---------|------------|-------------|
|
|
181
|
+
| An open PR comment looks unaddressed or blocking | YES | — |
|
|
182
|
+
| Posting a reply, resolving a thread | NO | `keryx review comments reply`, a human |
|
|
183
|
+
| Judging whether a reported finding is real | NO | `review-verifier` |
|
|
184
|
+
| Collecting comments from GitHub | NO | `keryx review comments collect` |
|
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: review-jev-docs
|
|
3
|
+
model_tier: light
|
|
4
|
+
description: |
|
|
5
|
+
Use when: an ADDITIONAL, machine-scored pass is wanted over documentation sections that
|
|
6
|
+
may have gone STALE because of a diff — never replacing any other reviewer. Dispatched
|
|
7
|
+
by review-orchestrator in Wave B, via the CLI (`keryx review jev-docs`), when
|
|
8
|
+
`review.jev.docs: true` in .metaproject/tasks.config.json and a Jev/OpenRouter
|
|
9
|
+
credential is resolvable. It is not an LLM sub-agent: there is nothing to dispatch
|
|
10
|
+
through a platform-native agent mechanism, and no prose is generated by a model —
|
|
11
|
+
Jev only answers a `noul` staleness probability per doc section; keryx writes every
|
|
12
|
+
word of every finding, and one class of finding (a removed/renamed CLI flag still
|
|
13
|
+
documented) is found with no Jev call at all.
|
|
14
|
+
NOT for: judging whether documentation is well-written, complete, or accurate about
|
|
15
|
+
something the diff never touched — this reviewer only ever looks at sections
|
|
16
|
+
deterministically linked to code the diff changed.
|
|
17
|
+
triggers:
|
|
18
|
+
- "jev docs"
|
|
19
|
+
- "stale docs"
|
|
20
|
+
- "review --jev-docs"
|
|
21
|
+
metadata:
|
|
22
|
+
author: "MrCipherSmith"
|
|
23
|
+
version: "1.0.0"
|
|
24
|
+
category: "review"
|
|
25
|
+
compatible_harnesses: "cursor,codex,zed,opencode,claude"
|
|
26
|
+
engine: "jev"
|
|
27
|
+
origin: "authored"
|
|
28
|
+
license: "MIT"
|
|
29
|
+
---
|
|
30
|
+
|
|
31
|
+
# Review — Jev Docs (stale-documentation detection)
|
|
32
|
+
|
|
33
|
+
An ADDITIONAL orchestrator reviewer, flow 333. It is a **keryx program**, not an
|
|
34
|
+
LLM sub-agent: `review-orchestrator` never dispatches a platform-native agent
|
|
35
|
+
for it, it runs `keryx review jev-docs` and reads the `--json` output. Every
|
|
36
|
+
finding is composed by keryx; Jev ("System One" on OpenRouter) supplies only a
|
|
37
|
+
`noul` staleness probability per doc section — it never writes prose, and
|
|
38
|
+
nothing it returns is quoted verbatim into a finding.
|
|
39
|
+
|
|
40
|
+
---
|
|
41
|
+
|
|
42
|
+
## What it does
|
|
43
|
+
|
|
44
|
+
1. Splits every discovered documentation file — the default corpus is
|
|
45
|
+
USER-FACING docs only: `docs/**`, the root `README*` (never `CHANGELOG*`),
|
|
46
|
+
and gdwiki pages (`.metaproject/wiki/**`) — into sections by heading,
|
|
47
|
+
deterministically (`src/review/jev-docs.ts`'s `extractDocSections` — no
|
|
48
|
+
model call). `--include <glob>` (repeatable) widens the corpus back out
|
|
49
|
+
(skills, rules, anything project-specific); see Input Contract.
|
|
50
|
+
2. Links each section to code deterministically — an explicit repo-relative
|
|
51
|
+
path it quotes, a backtick-quoted symbol that appears verbatim in one of
|
|
52
|
+
the diff's own hunks, or a `keryx <verb>` invocation whose command file the
|
|
53
|
+
diff changed. Only sections linked to code the diff CHANGES, and that the
|
|
54
|
+
diff does NOT itself edit, are candidates.
|
|
55
|
+
3. Ranks linked sections by link strength — path mention > symbol mention >
|
|
56
|
+
verb mention; more distinct links to changed code within the same kind
|
|
57
|
+
rank higher — and asks Jev exactly ONE `noul` question per selected
|
|
58
|
+
section, up to `--max-calls` (default 30) and 8 per doc file, batched
|
|
59
|
+
under the vendor's 64k token budget, with the ranking basis and every drop
|
|
60
|
+
reported (`selection`), never silent, never alphabetical.
|
|
61
|
+
4. Separately, and with NO Jev call: a CLI flag that disappears from a
|
|
62
|
+
changed file's diff (present on a removed line, absent from every added
|
|
63
|
+
line of the same file) and is still mentioned by an untouched doc section
|
|
64
|
+
is flagged deterministically.
|
|
65
|
+
5. Synthesizes findings deterministically: `problem` names the section
|
|
66
|
+
(file + heading + line) and the code change that likely outdated it,
|
|
67
|
+
`suggested_fix` names the update target, `evidence` carries Jev's
|
|
68
|
+
probability (or, for the flag check, the deterministic reasoning).
|
|
69
|
+
Severity is always `minor`.
|
|
70
|
+
|
|
71
|
+
---
|
|
72
|
+
|
|
73
|
+
## Input Contract
|
|
74
|
+
|
|
75
|
+
Not dispatched with a prompt — invoked as a CLI command:
|
|
76
|
+
|
|
77
|
+
```text
|
|
78
|
+
keryx review jev-docs (--diff <ref> | --pr <n>) [--max-calls <n>] [--threshold <0..1>]
|
|
79
|
+
[--repo <owner/repo>] [--model <jev-1.13|jev-latest>]
|
|
80
|
+
[--fixtures <dir>] [--include <glob>]... [--json]
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
`review-orchestrator` passes `--diff <ref>` (or `--pr <n>`) matching the same
|
|
84
|
+
target every other reviewer's dispatch checks.
|
|
85
|
+
|
|
86
|
+
---
|
|
87
|
+
|
|
88
|
+
## Opt-in and privacy
|
|
89
|
+
|
|
90
|
+
Refuses before any doc read or network call unless
|
|
91
|
+
`.metaproject/tasks.config.json` declares:
|
|
92
|
+
|
|
93
|
+
```json
|
|
94
|
+
{ "review": { "jev": { "docs": true } } }
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Every doc-section excerpt and hunk sent to Jev is redacted first through
|
|
98
|
+
`src/security/service.ts` — the same floor `review conform`/`review jev-rules`
|
|
99
|
+
already apply. No cache or output file stores a credential.
|
|
100
|
+
|
|
101
|
+
---
|
|
102
|
+
|
|
103
|
+
## Output Contract
|
|
104
|
+
|
|
105
|
+
Emits a `REVIEW_RESULT`-shaped object matching
|
|
106
|
+
`.metaproject/skills/gdskills/review/review-orchestrator/reviewer-finding.schema.json`
|
|
107
|
+
— `status`, `reviewer: "review-jev-docs"`, `summary`, `findings`, `stats` —
|
|
108
|
+
under `--json`. Its findings merge into the consolidated array exactly like
|
|
109
|
+
any other reviewer's: same Quality Gate, same dedup, same Wave C
|
|
110
|
+
verification.
|
|
111
|
+
|
|
112
|
+
### Class scope — required for `blocker` and `major`
|
|
113
|
+
|
|
114
|
+
Every finding this reviewer emits is `minor` (`docsFindingStats` caps it by
|
|
115
|
+
construction), so `class_scope` never applies here — `blocker`/`major`
|
|
116
|
+
findings, and the `class_scope` contract that comes with them, belong to the
|
|
117
|
+
reviewers named in `review-orchestrator/SKILL.md`'s own Finding Format
|
|
118
|
+
section.
|
|
119
|
+
|
|
120
|
+
---
|
|
121
|
+
|
|
122
|
+
## Orchestrator integration
|
|
123
|
+
|
|
124
|
+
- `keryx review reviewers --json` lists it under `bundled` with
|
|
125
|
+
`"engine": "jev"`.
|
|
126
|
+
- `review-orchestrator` dispatches it in **Wave B**, alongside the domain/
|
|
127
|
+
convention reviewers, when the opt-in is on and a Jev/OpenRouter credential
|
|
128
|
+
resolves (`resolveJevApiKeyResolution`). When either is false it is
|
|
129
|
+
**skipped with the stated reason**, recorded in `Skipped reviewers` —
|
|
130
|
+
never silently absent.
|
|
131
|
+
- It is **ADDITIONAL**: it never replaces any documentation-adjacent finding
|
|
132
|
+
another reviewer already raises; the dedup pass merges rather than
|
|
133
|
+
double-counts.
|
|
134
|
+
|
|
135
|
+
---
|
|
136
|
+
|
|
137
|
+
## Red Flags
|
|
138
|
+
|
|
139
|
+
| Rationalization | Why it is wrong |
|
|
140
|
+
|---|---|
|
|
141
|
+
| "Jev said 0.9, so the doc is definitely wrong now" | A `noul` score is a probability, not a verdict — read the linked hunk yourself before editing the doc |
|
|
142
|
+
| "No findings means the docs are current" | The default corpus is `docs/**`/`README*`/gdwiki only — a project's `.metaproject/skills/**`/`rules/**` need `--include` to be scored at all; `--max-calls`/8-per-file also bound what gets scored (`selection.droppedSections`) |
|
|
143
|
+
| "This section wasn't linked, so it's fine" | Linking is deliberately narrow (explicit paths/symbols/verbs) — a section describing behaviour with no explicit code reference is out of this reviewer's reach by design, not proven current |
|
|
144
|
+
|
|
145
|
+
---
|
|
146
|
+
|
|
147
|
+
## Verification
|
|
148
|
+
|
|
149
|
+
Before trusting a run's findings:
|
|
150
|
+
|
|
151
|
+
1. Read `selection.droppedSections` in the `--json` output — a capped run
|
|
152
|
+
covered less than every linked section.
|
|
153
|
+
2. Spot-check a finding against the cited hunk: does the section's own text
|
|
154
|
+
actually conflict with what the diff changed?
|
|
155
|
+
3. Confirm every finding carries `reviewer: "review-jev-docs"` and a
|
|
156
|
+
`dedupe_key` — both are required for the orchestrator's Quality Gate and
|
|
157
|
+
Wave C verification to route it correctly.
|
|
158
|
+
|
|
159
|
+
---
|
|
160
|
+
|
|
161
|
+
## Iron Laws
|
|
162
|
+
|
|
163
|
+
### Shared laws (every reviewer)
|
|
164
|
+
|
|
165
|
+
1. **A claim of runtime harm with no reproducible path is `info`.** If you cannot
|
|
166
|
+
name the input, call, or condition that reaches the code, you have an
|
|
167
|
+
observation, not a finding. Report it as `info` and say what would settle it.
|
|
168
|
+
2. **Never flag the theoretical.** The path you describe must exist in the code
|
|
169
|
+
under review. Do not report a safe API because it could be misused, or a
|
|
170
|
+
pattern because it is often wrong elsewhere.
|
|
171
|
+
3. **One finding per class, not one per occurrence.** When the same shape appears
|
|
172
|
+
at several sites, report it once and list every site. Ten findings that are one
|
|
173
|
+
finding hide the other nine problems.
|
|
174
|
+
|
|
175
|
+
Severity levels are defined once, in `review-orchestrator/SKILL.md` →
|
|
176
|
+
**Severity (canonical)**. This reviewer does not restate them — it only ever
|
|
177
|
+
emits `minor`, capped by construction (`docsFindingStats`), so the
|
|
178
|
+
`major`/`blocker` shapes never apply here.
|
|
179
|
+
|
|
180
|
+
---
|
|
181
|
+
|
|
182
|
+
## Scope Boundaries
|
|
183
|
+
|
|
184
|
+
| Concern | This skill | Use instead |
|
|
185
|
+
|---------|------------|-------------|
|
|
186
|
+
| A doc section linked to changed code likely went stale | YES | — |
|
|
187
|
+
| Documentation quality/completeness unrelated to this diff | NO | a human, or a dedicated docs review |
|
|
188
|
+
| Judging whether a reported finding is real | NO | `review-verifier` |
|
|
189
|
+
| Editing the documentation itself | NO | a human, following `suggested_fix` |
|
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: review-jev-risk
|
|
3
|
+
model_tier: light
|
|
4
|
+
description: |
|
|
5
|
+
Use when: an ADDITIONAL, machine-scored RISK MAP is wanted over every changed hunk —
|
|
6
|
+
never replacing any other reviewer. Dispatched by review-orchestrator in Wave B, via
|
|
7
|
+
the CLI (`keryx review jev-risk`), when `review.jev.risk: true` in
|
|
8
|
+
.metaproject/tasks.config.json and a Jev/OpenRouter credential is resolvable. It is not
|
|
9
|
+
an LLM sub-agent: there is nothing to dispatch through a platform-native agent
|
|
10
|
+
mechanism, and no prose is generated by a model — Jev only answers one `noul`
|
|
11
|
+
probability per (hunk, risk dimension) pair; keryx writes every word of every finding.
|
|
12
|
+
NOT for: judgement calls a risk dimension does not state (style, architecture beyond
|
|
13
|
+
risk-flagging — those stay with the reviewers that already cover them), and not a
|
|
14
|
+
substitute for a human reviewer reading the ranked map it produces.
|
|
15
|
+
triggers:
|
|
16
|
+
- "jev risk"
|
|
17
|
+
- "risk map"
|
|
18
|
+
- "review --jev-risk"
|
|
19
|
+
metadata:
|
|
20
|
+
author: "MrCipherSmith"
|
|
21
|
+
version: "1.0.0"
|
|
22
|
+
category: "review"
|
|
23
|
+
compatible_harnesses: "cursor,codex,zed,opencode,claude"
|
|
24
|
+
engine: "jev"
|
|
25
|
+
origin: "authored"
|
|
26
|
+
license: "MIT"
|
|
27
|
+
---
|
|
28
|
+
|
|
29
|
+
# Review — Jev Risk (deterministic risk map of the diff)
|
|
30
|
+
|
|
31
|
+
An ADDITIONAL orchestrator reviewer, flow 332. It is a **keryx program**, not
|
|
32
|
+
an LLM sub-agent: `review-orchestrator` never dispatches a platform-native
|
|
33
|
+
agent for it, it runs `keryx review jev-risk` and reads the `--json` output.
|
|
34
|
+
Every finding is composed by keryx; Jev ("System One" on OpenRouter) supplies
|
|
35
|
+
only five `noul` probabilities per hunk — one per risk dimension — never
|
|
36
|
+
prose.
|
|
37
|
+
|
|
38
|
+
---
|
|
39
|
+
|
|
40
|
+
## What it does
|
|
41
|
+
|
|
42
|
+
1. Takes every changed hunk from `keryx review scope`/`buildReviewScope`
|
|
43
|
+
(mechanical bulk already dropped).
|
|
44
|
+
2. Computes deterministic facts FIRST, per hunk: a path class (auth/
|
|
45
|
+
permissions, crypto, migrations, schema, public API, config, concurrency
|
|
46
|
+
primitives, IO), the exported symbols it touches, its changed-line count,
|
|
47
|
+
and whether a test file elsewhere in the diff touches the same module.
|
|
48
|
+
3. Asks Jev **one `noul` per risk dimension** per hunk — security-sensitive,
|
|
49
|
+
data/migration, public-API/contract change, concurrency, error-handling —
|
|
50
|
+
batched under the vendor's 64k token budget, capped at `--max-calls`
|
|
51
|
+
(default 150, counted as hunk x dimension pairs), with every drop
|
|
52
|
+
reported.
|
|
53
|
+
4. Ranks hunks by combined risk (the MAX across its five dimensions — one
|
|
54
|
+
high-risk dimension is enough to draw attention) so a human reviewer knows
|
|
55
|
+
where to look first.
|
|
56
|
+
5. Emits a finding ONLY for a hunk above threshold (default 0.7) AND with no
|
|
57
|
+
test touched nearby in the same diff (a fact) — severity capped at
|
|
58
|
+
`info`/`minor`, never higher: this flags attention, it does not assert a
|
|
59
|
+
defect.
|
|
60
|
+
6. Additionally emits a **routing hint**: hunks above threshold with a
|
|
61
|
+
security or concurrency dimension are listed with a suggested reviewer
|
|
62
|
+
(`review-security-code`/`review-highload`) for the orchestrator to
|
|
63
|
+
consider dispatching — see "Orchestrator integration" below.
|
|
64
|
+
|
|
65
|
+
---
|
|
66
|
+
|
|
67
|
+
## Input Contract
|
|
68
|
+
|
|
69
|
+
Not dispatched with a prompt — invoked as a CLI command:
|
|
70
|
+
|
|
71
|
+
```text
|
|
72
|
+
keryx review jev-risk (--diff <ref> | --pr <n> | --scope <scope.json>)
|
|
73
|
+
[--max-calls <n>] [--threshold <0..1>]
|
|
74
|
+
[--repo <owner/repo>] [--model <jev-1.13|jev-latest>]
|
|
75
|
+
[--fixtures <dir>] [--json]
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
`review-orchestrator` passes `--scope <scope.json>` — the same `keryx review
|
|
79
|
+
scope --json` file every other reviewer's dispatch already reads — so this
|
|
80
|
+
reviewer checks exactly the same hunks as everyone else in the round.
|
|
81
|
+
|
|
82
|
+
---
|
|
83
|
+
|
|
84
|
+
## Opt-in and privacy
|
|
85
|
+
|
|
86
|
+
Refuses before any read or network call unless `.metaproject/tasks.config.json`
|
|
87
|
+
declares:
|
|
88
|
+
|
|
89
|
+
```json
|
|
90
|
+
{ "review": { "jev": { "risk": true } } }
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
Every hunk sent to Jev is redacted first through `src/security/service.ts` —
|
|
94
|
+
the same floor `review conform`/`review ci-triage`/`review jev-rules` already
|
|
95
|
+
apply. No cache or output file stores a credential.
|
|
96
|
+
|
|
97
|
+
---
|
|
98
|
+
|
|
99
|
+
## Output Contract
|
|
100
|
+
|
|
101
|
+
Emits a `REVIEW_RESULT`-shaped object matching
|
|
102
|
+
`.metaproject/skills/gdskills/review/review-orchestrator/reviewer-finding.schema.json`
|
|
103
|
+
— `status`, `reviewer: "review-jev-risk"`, `summary`, `findings`, `stats` —
|
|
104
|
+
under `--json`, plus two orchestrator-facing extras: `ranked` (every scored
|
|
105
|
+
hunk, highest risk first) and `routingHints` (see above). Its findings merge
|
|
106
|
+
into the consolidated array exactly like any other reviewer's: same Quality
|
|
107
|
+
Gate, same dedup, same Wave C verification.
|
|
108
|
+
|
|
109
|
+
---
|
|
110
|
+
|
|
111
|
+
## Orchestrator integration
|
|
112
|
+
|
|
113
|
+
- `keryx review reviewers --json` lists it under `bundled` with
|
|
114
|
+
`"engine": "jev"`.
|
|
115
|
+
- `review-orchestrator` dispatches it in **Wave B**, alongside the domain/
|
|
116
|
+
convention reviewers and flow 330's `review-jev-rules`, when the opt-in is
|
|
117
|
+
on and a Jev/OpenRouter credential resolves. When either is false it is
|
|
118
|
+
**skipped with the stated reason**, recorded in `Skipped reviewers` — never
|
|
119
|
+
silently absent.
|
|
120
|
+
- It is **ADDITIONAL**: it never replaces `review-security-code`,
|
|
121
|
+
`review-highload`, or any other pass.
|
|
122
|
+
- **Routing hint**: read `routingHints` from its `--json` output. Each entry
|
|
123
|
+
names a hunk location, the dimension that crossed threshold
|
|
124
|
+
(`security`/`concurrency`) and a `suggestedReviewer`
|
|
125
|
+
(`review-security-code`/`review-highload`). When that reviewer was not
|
|
126
|
+
already selected for the round, dispatch it too — the hint is advisory,
|
|
127
|
+
not a hard requirement, and the orchestrator's own path-based selection
|
|
128
|
+
always takes precedence when it already covers the same hunk.
|
|
129
|
+
|
|
130
|
+
---
|
|
131
|
+
|
|
132
|
+
### Shared laws (every reviewer)
|
|
133
|
+
|
|
134
|
+
1. **A claim of runtime harm with no reproducible path is `info`.** If you cannot
|
|
135
|
+
name the input, call, or condition that reaches the code, you have an
|
|
136
|
+
observation, not a finding. Report it as `info` and say what would settle it.
|
|
137
|
+
2. **Never flag the theoretical.** The path you describe must exist in the code
|
|
138
|
+
under review. Do not report a safe API because it could be misused, or a
|
|
139
|
+
pattern because it is often wrong elsewhere.
|
|
140
|
+
3. **One finding per class, not one per occurrence.** When the same shape appears
|
|
141
|
+
at several sites, report it once and list every site. Ten findings that are one
|
|
142
|
+
finding hide the other nine problems.
|
|
143
|
+
|
|
144
|
+
This reviewer satisfies all three by construction: `evidence` always names the
|
|
145
|
+
hunk location, quotes the changed line, and states every dimension's
|
|
146
|
+
probability (never an unreproducible claim); a finding is synthesized only
|
|
147
|
+
against the hunk actually scored, never a hypothetical pattern; and each
|
|
148
|
+
hunk is a distinct site with its own probabilities, so there is no repeated
|
|
149
|
+
class across sites for this reviewer to collapse — the routing hint follows
|
|
150
|
+
the same discipline, naming the exact hunk and dimension that crossed
|
|
151
|
+
threshold rather than a general warning.
|
|
152
|
+
|
|
153
|
+
---
|
|
154
|
+
|
|
155
|
+
## Red Flags
|
|
156
|
+
|
|
157
|
+
| Rationalization | Why it is wrong |
|
|
158
|
+
|---|---|
|
|
159
|
+
| "Jev said 0.9, so this hunk is definitely dangerous" | A `noul` score is a probability across one dimension, not a verdict — severity stays capped at `info`/`minor` precisely because Jev alone is not authoritative |
|
|
160
|
+
| "No nearby test means nobody tested this at all" | `hasNearbyTest` requires the nearby test's OWN diff text to mention a touched symbol or import the hunk's module (`testHunkEvidence`), tightened after a live-check false negative — still a regex heuristic, not proof of coverage through an indirection it cannot see |
|
|
161
|
+
| "This hunk touches a .md file, so 'no nearby test' is a meaningful finding" | Docs (`.md`/`.txt`) hunks are excluded before scoring (`isNonCodeHunk`) after a live check where prose scored `public-api` — this cannot happen; `selection.hunksNotCode` reports the count |
|
|
162
|
+
| "No findings means the diff is low risk" | `--max-calls` bounds how many hunks are scored; a capped run reports `selection.hunksSkipped` for exactly this reason — read the selection stats before treating silence as clean |
|
|
163
|
+
|
|
164
|
+
---
|
|
165
|
+
|
|
166
|
+
## Verification
|
|
167
|
+
|
|
168
|
+
Before trusting a run's findings:
|
|
169
|
+
|
|
170
|
+
1. Read `selection.hunksSkipped` in the `--json` output — a capped run
|
|
171
|
+
covered fewer than "every retained hunk" and the report should say so.
|
|
172
|
+
2. Spot-check a handful of `ranked` entries against the actual hunk: does the
|
|
173
|
+
file/line range and the top dimension make sense for what actually
|
|
174
|
+
changed there? Docs (`.md`/`.txt`) hunks never reach `ranked` at all —
|
|
175
|
+
`selection.hunksNotCode` should account for every one of them in the
|
|
176
|
+
diff.
|
|
177
|
+
3. Confirm every finding carries `reviewer: "review-jev-risk"` and a
|
|
178
|
+
`dedupe_key` — both are required for the orchestrator's Quality Gate and
|
|
179
|
+
Wave C verification to route it correctly.
|
|
180
|
+
|
|
181
|
+
---
|
|
182
|
+
|
|
183
|
+
## Scope Boundaries
|
|
184
|
+
|
|
185
|
+
| Concern | This skill | Use instead |
|
|
186
|
+
|---------|------------|-------------|
|
|
187
|
+
| A ranked risk map of the diff's hunks | YES | — |
|
|
188
|
+
| An actual security/concurrency finding | NO (routing hint only) | `review-security-code` / `review-highload` |
|
|
189
|
+
| Judging whether a reported finding is real | NO | `review-verifier` |
|
|
190
|
+
| Which user scenarios changed | NO | `review-jev-scenarios` |
|