strikethroo 3.17.2 → 3.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist-web/assets/{arc-C3L1jkFY.js → arc-BlSNa2QY.js} +1 -1
- package/dist-web/assets/{architectureDiagram-3BPJPVTR-Dc5wdWPF.js → architectureDiagram-3BPJPVTR-CPcCF2a4.js} +1 -1
- package/dist-web/assets/{blockDiagram-GPEHLZMM-BCqaESV7.js → blockDiagram-GPEHLZMM-ECDR0edZ.js} +1 -1
- package/dist-web/assets/{c4Diagram-AAUBKEIU-DNecSAwM.js → c4Diagram-AAUBKEIU-oM3VnEAQ.js} +1 -1
- package/dist-web/assets/channel-CmU4mBgT.js +1 -0
- package/dist-web/assets/{chunk-2J33WTMH-3QS28848.js → chunk-2J33WTMH-DHk5SJLz.js} +1 -1
- package/dist-web/assets/{chunk-4BX2VUAB-B6R-x-SN.js → chunk-4BX2VUAB-CjBwgkMu.js} +1 -1
- package/dist-web/assets/{chunk-55IACEB6-BYY5vsbu.js → chunk-55IACEB6-D95zqyiV.js} +1 -1
- package/dist-web/assets/{chunk-727SXJPM-BZxIJIkN.js → chunk-727SXJPM-DCCPRUfv.js} +1 -1
- package/dist-web/assets/{chunk-AQP2D5EJ-Dwpt8x10.js → chunk-AQP2D5EJ-qur7KQyD.js} +1 -1
- package/dist-web/assets/{chunk-FMBD7UC4-BkIlqLqc.js → chunk-FMBD7UC4-D46Su7Fv.js} +1 -1
- package/dist-web/assets/{chunk-ND2GUHAM-BZAfIvWk.js → chunk-ND2GUHAM-BioGsqxt.js} +1 -1
- package/dist-web/assets/{chunk-QZHKN3VN-wAp9ruCQ.js → chunk-QZHKN3VN-C8LdDiqn.js} +1 -1
- package/dist-web/assets/classDiagram-4FO5ZUOK-D_FMKQsB.js +1 -0
- package/dist-web/assets/classDiagram-v2-Q7XG4LA2-D_FMKQsB.js +1 -0
- package/dist-web/assets/{cose-bilkent-S5V4N54A-BWhls-Li.js → cose-bilkent-S5V4N54A-ccY03BRh.js} +1 -1
- package/dist-web/assets/{dagre-BM42HDAG-D_rm5H8p.js → dagre-BM42HDAG-BYLmCViB.js} +1 -1
- package/dist-web/assets/{diagram-2AECGRRQ-CgrRz62Y.js → diagram-2AECGRRQ-DTvenvXR.js} +1 -1
- package/dist-web/assets/{diagram-5GNKFQAL-BRdPFx5y.js → diagram-5GNKFQAL-CCzsYDmr.js} +1 -1
- package/dist-web/assets/{diagram-KO2AKTUF-CRkShN2W.js → diagram-KO2AKTUF-QSff7J_V.js} +1 -1
- package/dist-web/assets/{diagram-LMA3HP47-Bxah1b5X.js → diagram-LMA3HP47-CVqlEmu2.js} +1 -1
- package/dist-web/assets/{diagram-OG6HWLK6-D9gsqeVI.js → diagram-OG6HWLK6-dBVz-I46.js} +1 -1
- package/dist-web/assets/{erDiagram-TEJ5UH35-BxuqpCGH.js → erDiagram-TEJ5UH35-B_y4ftNu.js} +1 -1
- package/dist-web/assets/{flowDiagram-I6XJVG4X-CVc346ab.js → flowDiagram-I6XJVG4X-M410hT_n.js} +1 -1
- package/dist-web/assets/{ganttDiagram-6RSMTGT7-CoV2q8mu.js → ganttDiagram-6RSMTGT7-qiri2OJa.js} +1 -1
- package/dist-web/assets/{gitGraphDiagram-PVQCEYII-DpBtbvUQ.js → gitGraphDiagram-PVQCEYII-FqGGBobG.js} +1 -1
- package/dist-web/assets/{index-qiGGVGhU.js → index-1SV04K_c.js} +1 -1
- package/dist-web/assets/{index-DXAUzXLU.js → index-BaFSLlgl.js} +1 -1
- package/dist-web/assets/{index-B5AkN942.js → index-DrUIu0u6.js} +4 -4
- package/dist-web/assets/{infoDiagram-5YYISTIA-Cs1LaltR.js → infoDiagram-5YYISTIA-CczHqRr1.js} +1 -1
- package/dist-web/assets/{ishikawaDiagram-YF4QCWOH-2dwQITAJ.js → ishikawaDiagram-YF4QCWOH-ty1DRy6n.js} +1 -1
- package/dist-web/assets/{journeyDiagram-JHISSGLW-C5NXwRcO.js → journeyDiagram-JHISSGLW-D_JYvt6i.js} +1 -1
- package/dist-web/assets/{kanban-definition-UN3LZRKU-D8L4UiFp.js → kanban-definition-UN3LZRKU-CT67nUTK.js} +1 -1
- package/dist-web/assets/{linear-BmwJXk2t.js → linear-B1HaE_t1.js} +1 -1
- package/dist-web/assets/{mermaid.core-CxfmFJso.js → mermaid.core-Dumx_iKV.js} +4 -4
- package/dist-web/assets/{mindmap-definition-RKZ34NQL-CCk10yMr.js → mindmap-definition-RKZ34NQL-CVulJo63.js} +1 -1
- package/dist-web/assets/{pieDiagram-4H26LBE5-CNaROrHh.js → pieDiagram-4H26LBE5-C1yD5ntZ.js} +1 -1
- package/dist-web/assets/{quadrantDiagram-W4KKPZXB-C9eBwT_4.js → quadrantDiagram-W4KKPZXB-B7PoiM8c.js} +1 -1
- package/dist-web/assets/{requirementDiagram-4Y6WPE33-CuWFUe-Z.js → requirementDiagram-4Y6WPE33-COvHb3Mv.js} +1 -1
- package/dist-web/assets/{sankeyDiagram-5OEKKPKP-xGYYGBDj.js → sankeyDiagram-5OEKKPKP-CAnwm5-J.js} +1 -1
- package/dist-web/assets/{sequenceDiagram-3UESZ5HK-Corn4mP3.js → sequenceDiagram-3UESZ5HK-C2KFLIUq.js} +1 -1
- package/dist-web/assets/{stateDiagram-AJRCARHV-DP1EwgDv.js → stateDiagram-AJRCARHV-CqmhS5P6.js} +1 -1
- package/dist-web/assets/stateDiagram-v2-BHNVJYJU-C3Kw8ZDR.js +1 -0
- package/dist-web/assets/{timeline-definition-PNZ67QCA-DgVY9v5q.js → timeline-definition-PNZ67QCA-DvPre9M_.js} +1 -1
- package/dist-web/assets/{vennDiagram-CIIHVFJN-B47n0Ab2.js → vennDiagram-CIIHVFJN-Top6i7KH.js} +1 -1
- package/dist-web/assets/{wardley-L42UT6IY-ubK4o7Fl.js → wardley-L42UT6IY-_jVLsWqf.js} +1 -1
- package/dist-web/assets/{wardleyDiagram-YWT4CUSO-ChirN_61.js → wardleyDiagram-YWT4CUSO-BV4e5mP6.js} +1 -1
- package/dist-web/assets/{xychartDiagram-2RQKCTM6-Cf8q2_P7.js → xychartDiagram-2RQKCTM6-Bn1hw4Jk.js} +1 -1
- package/dist-web/index.html +1 -1
- package/package.json +1 -1
- package/templates/harness/skills/st-code-review/SKILL.md +47 -281
- package/templates/harness/skills/st-code-review/scripts/code-review.cjs +70 -300
- package/templates/harness/skills/st-execute-blueprint/SKILL.md +15 -18
- package/templates/harness/skills/st-full-workflow/SKILL.md +15 -18
- package/templates/strikethroo/config/hooks/CODE_REVIEW.md +19 -29
- package/dist-web/assets/channel-DmY2Krrw.js +0 -1
- package/dist-web/assets/classDiagram-4FO5ZUOK-C91aiCZF.js +0 -1
- package/dist-web/assets/classDiagram-v2-Q7XG4LA2-C91aiCZF.js +0 -1
- package/dist-web/assets/stateDiagram-v2-BHNVJYJU-DQIao4rH.js +0 -1
|
@@ -1,324 +1,90 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: st-code-review
|
|
3
|
-
description: Use when the blueprint execution gate asks for an independent second-harness review of a Strikethroo plan's cumulative diff in this repository — triggers include code review gate, review the plan diff, second-model review, CODE_REVIEW hook, review
|
|
3
|
+
description: Use when the blueprint execution gate asks for an independent second-harness review of a Strikethroo plan's cumulative diff in this repository — triggers include code review gate, review the plan diff, second-model review, CODE_REVIEW hook, review the cumulative diff. Do not use to review a single task, to review code outside a Strikethroo plan, or to give general code-quality or style opinions.
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
<!--
|
|
7
|
-
Review
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
scope filters out most of its deliberately broad rubric.
|
|
7
|
+
Review categories and false-positive heuristics draw on PR-Agent
|
|
8
|
+
(https://github.com/The-PR-Agent/pr-agent), used under its permissive
|
|
9
|
+
licence. No PR-Agent code is vendored.
|
|
11
10
|
-->
|
|
12
11
|
|
|
13
12
|
# st-code-review
|
|
14
13
|
|
|
15
|
-
|
|
16
|
-
on a different harness than the one that wrote the code, and emit the findings as
|
|
17
|
-
a schema-validated `review.xml`.
|
|
14
|
+
Review a Strikethroo plan's cumulative diff as an independent reviewer, running on a different harness than the one that wrote the code. `<root>` is the workspace root the dispatch supplies.
|
|
18
15
|
|
|
19
|
-
|
|
20
|
-
with this dispatch.
|
|
16
|
+
**You detect. You never fix.** Do not edit files, run formatters, or commit. Your output is one findings document plus a count report. Raising nothing is a correct and common result. Report it and stop.
|
|
21
17
|
|
|
22
|
-
|
|
18
|
+
Your findings are recorded, not applied. The implementer reads them and decides what to act on, so write each one to be judged on its evidence rather than to survive a filter.
|
|
23
19
|
|
|
24
|
-
|
|
20
|
+
## Grading
|
|
25
21
|
|
|
26
|
-
|
|
27
|
-
commit.
|
|
28
|
-
- Fixes are dispatched separately, on the implementer route. The implementer sees
|
|
29
|
-
a finding, never your reasoning about it.
|
|
30
|
-
- Your entire output is one findings document, printed as described below, plus
|
|
31
|
-
a short report.
|
|
22
|
+
Severity is impact **if the finding is real**. Confidence is **how sure you are that it is real**. Both are advisory labels that help whoever reads the review sort it. Nothing is filtered on them and nothing is applied automatically, so there is no floor to clear and no reason to inflate either one. Grade honestly. A `low` confidence finding that is marked `low` is useful; the same finding marked `high` is a trap.
|
|
32
23
|
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
1. **Conformance and defects only.** Check the diff against the plan's stated
|
|
36
|
-
requirements and for demonstrable defects. Nothing else.
|
|
37
|
-
2. **Every finding cites evidence.** A file, a line range, and the concrete input
|
|
38
|
-
or state that produces the failure.
|
|
39
|
-
3. **Every finding traces.** To an explicit requirement written in the plan, or
|
|
40
|
-
to a defect demonstrable in the code as written.
|
|
41
|
-
4. **Emit `severity` and `confidence` on every comment.** Both attributes,
|
|
42
|
-
always, judged by the tests below.
|
|
43
|
-
5. **The hook is authoritative.** `<root>/config/hooks/CODE_REVIEW.md` carries the
|
|
44
|
-
mandate, the floors, and the categories in scope. Where it and this prompt
|
|
45
|
-
disagree, the hook wins.
|
|
46
|
-
6. **Findings do not accumulate power by volume.** Zero findings above the floor
|
|
47
|
-
is a correct and common outcome. Report it and stop.
|
|
48
|
-
|
|
49
|
-
## Mandate: Conformance and Defects Only
|
|
50
|
-
|
|
51
|
-
### In scope — exactly two categories
|
|
52
|
-
|
|
53
|
-
| `<category>` value | What it means |
|
|
24
|
+
| `severity` | Impact if real. Never confidence, never fix cost. |
|
|
54
25
|
| --- | --- |
|
|
55
|
-
| `
|
|
56
|
-
| `
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
### Out of mandate — record nothing
|
|
61
|
-
|
|
62
|
-
- Style, naming, formatting, import order. The linter owns these.
|
|
63
|
-
- Design and abstraction opinions. "This works, but I would have structured it
|
|
64
|
-
differently" is not a finding.
|
|
65
|
-
- Requirements the plan does not state. The YAGNI posture that binds a planner to
|
|
66
|
-
explicit requirements binds your demands equally.
|
|
67
|
-
- Speculative hardening for inputs no stated requirement admits.
|
|
68
|
-
- Missing tests the plan did not ask for.
|
|
26
|
+
| `critical` | Data loss, a security hole, or a crash on a path real usage reaches. |
|
|
27
|
+
| `major` | Wrong behaviour or a broken declared contract on a reached path, or a stated requirement left unmet. |
|
|
28
|
+
| `minor` | Real but bounded. Behaviour is correct today. |
|
|
29
|
+
| `info` | No defect. A recorded observation. |
|
|
69
30
|
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
A conformance-only reviewer will not catch *"this works and matches the plan but
|
|
73
|
-
the abstraction is wrong"* — a real category a second model is unusually good at
|
|
74
|
-
spotting. That is a deliberate trade against false-positive rate, and it is the
|
|
75
|
-
right trade only because this gate is unattended and auto-applies fixes. It is
|
|
76
|
-
not an oversight. Do not widen scope to cover it.
|
|
77
|
-
|
|
78
|
-
## Anti-rationalization
|
|
79
|
-
|
|
80
|
-
Read `<root>/config/shared/anti-rationalization.md` and apply it to this table.
|
|
81
|
-
Every row points at you, the reviewer.
|
|
82
|
-
|
|
83
|
-
| You catch yourself thinking… | The binding rule |
|
|
31
|
+
| `confidence` | An evidentiary test, never a feel. |
|
|
84
32
|
| --- | --- |
|
|
85
|
-
|
|
|
86
|
-
|
|
|
87
|
-
|
|
|
88
|
-
| "I am not certain, so I will explain at length to be safe." | Length is not evidence. Prompts demanding detailed justification show measurably higher misjudgment rates. Lower the `confidence` attribute instead. |
|
|
89
|
-
| "The caller probably handles this." / "The caller probably does not handle this." | An unread caller is an assumption, not evidence. Read it, or mark the finding `confidence="medium"` and state the assumption in the body. |
|
|
90
|
-
| "This is only `minor`, but it is real, so I will call it `major` so it gets fixed." | Severity is impact if real, never a lever for clearing the floor. Inflating it is the failure this gate is built to resist. Grade it honestly and let it be recorded. |
|
|
91
|
-
| "I have found nothing above the floor; I should look harder so this round produces something." | A clean diff is a valid result. Manufacturing a finding to justify the round is the exact defect — an over-rejecting reviewer injecting speculative changes into working code. Report zero and stop. |
|
|
92
|
-
|
|
93
|
-
## Confidence and Severity
|
|
94
|
-
|
|
95
|
-
These are independent. Severity says how bad it is **if the finding is real**.
|
|
96
|
-
Confidence says **how sure you are that it is real**. An automated consumer
|
|
97
|
-
thresholds on both and relies on you lowering `confidence` honestly. LLM
|
|
98
|
-
reviewers systematically overstate certainty here.
|
|
33
|
+
| `high` | Traceable from the diff and the files you read. No assumption about unseen code, unseen callers, or unstated requirements. |
|
|
34
|
+
| `medium` | Exactly one unverified assumption, written out explicitly in the body. |
|
|
35
|
+
| `low` | You imagined the failure rather than traced it, or it rests on two or more unverified assumptions. |
|
|
99
36
|
|
|
100
|
-
|
|
37
|
+
Wording changes nothing. A confident sentence resting on an unread caller is `medium`.
|
|
101
38
|
|
|
102
|
-
|
|
103
|
-
| --- | --- |
|
|
104
|
-
| `high` | Traceable entirely from the diff and the files you read. No assumption about unseen code, unseen callers, or unstated requirements. |
|
|
105
|
-
| `medium` | Exactly one unverified assumption, and that assumption is written out explicitly in the finding body. |
|
|
106
|
-
| `low` | The failure was imagined rather than traced, or the finding rests on more than one unverified assumption, or on a constraint you invented. |
|
|
107
|
-
|
|
108
|
-
How strongly the prose is worded changes nothing. A confident sentence resting on
|
|
109
|
-
an unread caller is `medium`.
|
|
39
|
+
## Anti-rationalization
|
|
110
40
|
|
|
111
|
-
|
|
41
|
+
Read `<root>/config/shared/anti-rationalization.md`. Every row below is an excuse you will be tempted to make.
|
|
112
42
|
|
|
113
|
-
|
|
|
43
|
+
| You catch yourself thinking... | The binding rule |
|
|
114
44
|
| --- | --- |
|
|
115
|
-
|
|
|
116
|
-
|
|
|
117
|
-
|
|
|
118
|
-
|
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
Default floors: severity `major`, confidence `high`. The hook is authoritative if
|
|
123
|
-
it names different ones.
|
|
124
|
-
|
|
125
|
-
- Findings below either floor are **recorded in `review.xml` and never
|
|
126
|
-
auto-applied**. Recording them is useful; that is where they belong.
|
|
127
|
-
- A comment that **omits** `severity` or `confidence` falls below every floor.
|
|
128
|
-
Omission is never a route to getting a finding applied.
|
|
129
|
-
- A finding that lacks concrete evidence, or lacks a trace to an explicit plan
|
|
130
|
-
requirement or a demonstrable defect, is not emitted above the floor. Grade it
|
|
131
|
-
`info`/`low` or leave it out.
|
|
132
|
-
|
|
133
|
-
## Scope: Cumulative Diff and Blast Radius
|
|
134
|
-
|
|
135
|
-
- The dispatch supplies the recorded **base commit** for this plan. Review
|
|
136
|
-
everything that changed between that commit and the **current working tree**.
|
|
137
|
-
- **Uncommitted changes are in scope.** Post-execution cleanup and any fix
|
|
138
|
-
applied by an earlier round are not committed by the time you run.
|
|
139
|
-
- Review the **cumulative** diff every round. Never the incremental fix diff. The
|
|
140
|
-
scope never narrows between rounds.
|
|
141
|
-
- Prior findings arrive marked **adjudicated**. Do not re-litigate them. Their
|
|
142
|
-
presence does not narrow what you look at.
|
|
143
|
-
- The default round budget is 3. Rounds are counted and terminated in code, not
|
|
144
|
-
by you. Review this round, emit, report, and stop. Do not schedule, request, or
|
|
145
|
-
simulate another round.
|
|
146
|
-
- **Blast radius.** For each symbol the diff changed — renamed, resignatured,
|
|
147
|
-
deleted, or given different behaviour — locate references **outside** the diff
|
|
148
|
-
and read those callsites. This is targeted expansion, not whole-codebase
|
|
149
|
-
review, and it is a partial mitigation rather than a complete one.
|
|
150
|
-
|
|
151
|
-
<details>
|
|
152
|
-
<summary>Obtaining the diff and running the blast-radius pass</summary>
|
|
153
|
-
|
|
154
|
-
When the dispatch hands you the diff, review what it hands you. When it hands you
|
|
155
|
-
only the base commit id, produce the cumulative diff yourself — `git diff <base>`
|
|
156
|
-
compares the base against the working tree and therefore includes uncommitted
|
|
157
|
-
changes, which `git diff <base>..HEAD` would omit.
|
|
158
|
-
|
|
159
|
-
For the blast-radius pass, list the symbols whose declaration or behaviour the
|
|
160
|
-
diff changed, then search the repository for each one and read every hit that is
|
|
161
|
-
not in a file the diff touched. A hit that still type-checks and still receives
|
|
162
|
-
what it expects needs no finding. A hit that now receives a different shape, a
|
|
163
|
-
different arity, or a different error contract is a `defect` finding with the
|
|
164
|
-
callsite's file and line as its evidence.
|
|
165
|
-
|
|
166
|
-
</details>
|
|
167
|
-
|
|
168
|
-
## Output: `review.xml`
|
|
45
|
+
| "This could theoretically fail if..." | Name the concrete input or state, or it is not a finding. |
|
|
46
|
+
| "The plan does not say this, but it should have." | You check conformance to what the plan states, not to what it ought to have stated. |
|
|
47
|
+
| "This works, but the abstraction is wrong." | Design opinion. Record nothing. |
|
|
48
|
+
| "I am unsure, so I will justify at length to be safe." | Length is not evidence. Lower `confidence` instead. |
|
|
49
|
+
| "The caller probably handles this." / "...probably does not." | An unread caller is an assumption. Read it, or mark `confidence="medium"` and state the assumption in the body. |
|
|
50
|
+
| "Only `minor`, but real, so I will call it `major` so someone acts on it." | Severity is impact. Inflating it corrupts the one signal the reader sorts by. |
|
|
51
|
+
| "I found nothing, so I should look harder until this produces something." | A clean diff is a valid result. Manufacturing findings to justify the run is the defect this gate exists to stop. |
|
|
169
52
|
|
|
170
|
-
|
|
171
|
-
validate against the vendored schema at
|
|
172
|
-
`<root>/config/schemas/self-review-v2.xsd`.
|
|
53
|
+
## Output
|
|
173
54
|
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
- `<review>` — root. Requires `timestamp` (ISO 8601 with timezone). Set
|
|
177
|
-
`git-diff-args` to the base commit id and `repository` to the absolute
|
|
178
|
-
repository root.
|
|
179
|
-
- `<file>` — one per file in the diff, **including files you reviewed and had no
|
|
180
|
-
comment on**. Requires `path` (relative to the repository root),
|
|
181
|
-
`change-type` (`added`, `modified`, `deleted`, `renamed`), and `viewed`
|
|
182
|
-
(`true` when you read it).
|
|
183
|
-
- `<comment>` — zero or more per file, in child order `<body>`, `<category>`,
|
|
184
|
-
then an optional `<suggestion>`. Carry `severity` and `confidence` on every
|
|
185
|
-
one. For a line-level comment give **exactly one** pair: `new-line-start` and
|
|
186
|
-
`new-line-end` for added or context lines, or `old-line-start` and
|
|
187
|
-
`old-line-end` for deleted lines. Both pairs absent means a file-level
|
|
188
|
-
comment. Single-line comments set start equal to end.
|
|
189
|
-
- `<suggestion>` — optional, at most one per comment, containing
|
|
190
|
-
`<original-code>` then `<proposed-code>`.
|
|
55
|
+
Emit one document in the `urn:self-review:v2` namespace. It must validate against `<root>/config/schemas/self-review-v2.xsd`. Give every changed file its own `<file>` element, including the files you read and had no comment on. `path` is repository-relative. `change-type` is `added`, `modified`, `deleted`, or `renamed`. Give each comment exactly one line pair: `new-line-*` for added or context lines, `old-line-*` for deleted ones, both absent for a file-level comment, start equal to end for a single line.
|
|
191
56
|
|
|
192
57
|
```xml
|
|
193
58
|
<?xml version="1.0" encoding="UTF-8"?>
|
|
194
|
-
<review xmlns="urn:self-review:v2"
|
|
195
|
-
|
|
196
|
-
git-diff-args="a1b2c3d4"
|
|
197
|
-
repository="/abs/path/to/repo">
|
|
59
|
+
<review xmlns="urn:self-review:v2" timestamp="2026-07-27T09:41:00Z"
|
|
60
|
+
git-diff-args="a1b2c3d4" repository="/abs/path/to/repo">
|
|
198
61
|
<file path="src/parse.ts" change-type="modified" viewed="true">
|
|
199
|
-
<comment new-line-start="42" new-line-end="
|
|
200
|
-
severity="major" confidence="high">
|
|
62
|
+
<comment new-line-start="42" new-line-end="42" severity="major" confidence="high">
|
|
201
63
|
<body>Plan requirement "reject an empty id" is unmet: parseId("") returns
|
|
202
|
-
`{ ok: true }` because the length guard runs after the early return
|
|
203
|
-
line 42.</body>
|
|
64
|
+
`{ ok: true }` because the length guard runs after the early return.</body>
|
|
204
65
|
<category>requirement-conformance</category>
|
|
205
|
-
<suggestion>
|
|
206
|
-
<original-code> if (!raw) return { ok: true };</original-code>
|
|
207
|
-
<proposed-code> if (!raw) return { ok: false };</proposed-code>
|
|
208
|
-
</suggestion>
|
|
209
66
|
</comment>
|
|
210
67
|
</file>
|
|
211
68
|
<file path="src/index.ts" change-type="added" viewed="true" />
|
|
212
69
|
</review>
|
|
213
70
|
```
|
|
214
71
|
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
A `<suggestion>` carries `original-code` copied **verbatim** from the file,
|
|
218
|
-
because it is applied by exact text matching. A finding whose fix cannot be
|
|
219
|
-
expressed as a local text replacement carries **no** suggestion — record the
|
|
220
|
-
finding and stop. Do not restructure a fix to fit the element.
|
|
221
|
-
|
|
222
|
-
Why, once: this is what blocks broad speculative refactors structurally rather
|
|
223
|
-
than by request. It must not be designed away.
|
|
224
|
-
|
|
225
|
-
### Delivering the findings document
|
|
72
|
+
**Never emit a `<suggestion>`.** The element exists so that a human reviewer can hand the implementer exact replacement text, and whatever it contains gets applied verbatim, without anyone reading it first. You are not a human reviewer. Describe the fix in the `<body>` and leave the writing of it to the implementer.
|
|
226
73
|
|
|
227
|
-
|
|
228
|
-
Print the complete document between those exact lines as the final thing you
|
|
229
|
-
print. This is the only channel that is read — do not write the document to a
|
|
230
|
-
file.
|
|
231
|
-
|
|
232
|
-
- Copy the delimiter lines from the dispatch; never invent a token.
|
|
233
|
-
- Print nothing after the closing delimiter line.
|
|
234
|
-
- The same schema validates the document. An incomplete or invented document
|
|
235
|
-
fails the round.
|
|
236
|
-
- Being unable to read the repository is **not** a reason to emit the block.
|
|
237
|
-
A review you could not perform is a failed round. Report it as one.
|
|
74
|
+
**Delivery.** Print the document between the dispatch's BEGIN/END delimiters, copied exactly, as the last thing you print, with nothing after the closing line. The orchestrator reads that block and nothing else. Never write the document to a file. Never invent a token. Being unable to read the repository is not a reason to emit the block. A review you could not perform is a failed review, so report it as one.
|
|
238
75
|
|
|
239
76
|
## Operating Procedure
|
|
240
77
|
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
Read `<root>/config/hooks/CODE_REVIEW.md`.
|
|
244
|
-
|
|
245
|
-
**Exit criterion:** you can state the severity floor, the confidence floor, and
|
|
246
|
-
the finding categories in scope from that file. Those values govern this round.
|
|
247
|
-
|
|
248
|
-
### 2. Read the plan
|
|
249
|
-
|
|
250
|
-
Read the plan document named in the dispatch, in full.
|
|
251
|
-
|
|
252
|
-
**Exit criterion:** you have written down the list of explicit requirements this
|
|
253
|
-
diff is answerable to. A requirement not on that list cannot produce a
|
|
254
|
-
`requirement-conformance` finding.
|
|
255
|
-
|
|
256
|
-
### 3. Read the cumulative diff
|
|
257
|
-
|
|
258
|
-
Take the base commit from the dispatch and review base against the working tree.
|
|
259
|
-
|
|
260
|
-
**Exit criterion:** every changed file is enumerated, and each one will receive a
|
|
261
|
-
`<file>` element — including the ones you have no comment on.
|
|
262
|
-
|
|
263
|
-
### 4. Run the blast-radius pass
|
|
264
|
-
|
|
265
|
-
For each symbol the diff changed, read the references outside the diff.
|
|
266
|
-
|
|
267
|
-
**Exit criterion:** each changed symbol has been searched, and each out-of-diff
|
|
268
|
-
callsite has been read or explicitly noted as unread in the body of any finding
|
|
269
|
-
that depends on it.
|
|
270
|
-
|
|
271
|
-
### 5. Critique under the mandate
|
|
272
|
-
|
|
273
|
-
<details>
|
|
274
|
-
<summary>Per-candidate procedure</summary>
|
|
275
|
-
|
|
276
|
-
For each candidate finding, in order:
|
|
277
|
-
|
|
278
|
-
1. Name the category — `requirement-conformance` or `defect`. No third option;
|
|
279
|
-
if neither fits, discard the candidate.
|
|
280
|
-
2. Cite the evidence — file, line range, and the concrete input or state that
|
|
281
|
-
produces the failure.
|
|
282
|
-
3. Trace it — to a requirement on your step 2 list, or to a defect demonstrable
|
|
283
|
-
in the code as written.
|
|
284
|
-
4. Grade `severity` by impact if real.
|
|
285
|
-
5. Grade `confidence` by the evidentiary test. Count your unverified
|
|
286
|
-
assumptions: zero is `high`, exactly one is `medium` and must be written into
|
|
287
|
-
the body, more than one is `low`.
|
|
288
|
-
6. Check the anti-rationalization table against the sentence you just wrote.
|
|
289
|
-
7. Attach a `<suggestion>` only when the fix is a local text replacement, with
|
|
290
|
-
`original-code` copied verbatim.
|
|
291
|
-
|
|
292
|
-
</details>
|
|
293
|
-
|
|
294
|
-
**Exit criterion:** every finding carries a category, evidence, a trace, and both
|
|
295
|
-
attributes. Findings that fail any of those are dropped or graded `info`/`low`.
|
|
296
|
-
|
|
297
|
-
### 6. Emit the findings document
|
|
298
|
-
|
|
299
|
-
Print the document between the dispatch's delimiters, in the shape above.
|
|
300
|
-
|
|
301
|
-
**Exit criterion:** the document you printed declares `urn:self-review:v2`, has
|
|
302
|
-
one `<file>` per changed file, and every `<comment>` carries `severity` and
|
|
303
|
-
`confidence`.
|
|
304
|
-
|
|
305
|
-
### 7. Report
|
|
306
|
-
|
|
307
|
-
State the counts: total findings, findings at or above both floors, and findings
|
|
308
|
-
recorded below a floor. Name the floors you applied.
|
|
78
|
+
Each step ends only when its exit criterion holds.
|
|
309
79
|
|
|
310
|
-
**Exit
|
|
311
|
-
|
|
312
|
-
|
|
80
|
+
1. **Read `<root>/config/hooks/CODE_REVIEW.md`.** It is authoritative and beats this prompt wherever they disagree. Exit: you can state the finding categories in scope from that file.
|
|
81
|
+
2. **Read the dispatched plan in full.** Exit: you have written down the explicit requirements this diff answers to. Nothing off that list can produce a `requirement-conformance` finding.
|
|
82
|
+
3. **Read the cumulative diff.** Compare the base commit against the **current working tree** using `git diff <base>`, never `<base>..HEAD`, which drops the uncommitted post-execution cleanup that belongs in scope. Read the whole diff. You run once, so never schedule, request, or simulate a second pass. Exit: every changed file is enumerated and will get a `<file>` element.
|
|
83
|
+
4. **Run the blast-radius pass.** Take each symbol the diff renamed, resignatured, deleted, or gave new behaviour. Search the repository and read every hit outside the diff. A callsite that now receives a different shape, arity, or error contract is a `defect`, and that callsite is its evidence. This is targeted expansion, not whole-codebase review. Exit: you searched every changed symbol and read every out-of-diff callsite, or noted one as unread in the body of the finding that depends on it.
|
|
84
|
+
5. **Critique each candidate.** Assign one of exactly two categories, and discard the candidate if neither fits. `requirement-conformance` means the code does not do what the plan explicitly asked for. `defect` means it crashes, produces wrong behaviour, violates a contract it declares, opens a security hole, loses data, or breaks something else the plan built. Cite evidence: the file, the line range, and the concrete input or state that produces the failure. Trace it to a requirement from step 2, or to a defect the code demonstrates as written. Grade both attributes. Check the sentence you just wrote against the table above. Attach no `<suggestion>`. Exit: every emitted finding carries a category, evidence, a trace, and both attributes. Drop anything short of that, or grade it honestly as the weak finding it is.
|
|
85
|
+
6. **Emit the document.** Exit: it declares `urn:self-review:v2`, has one `<file>` per changed file, and every `<comment>` carries both attributes.
|
|
86
|
+
7. **Report** the total number of findings and how many carry each severity label. Exit: the counts match the document. Never claim a clean review without having emitted it.
|
|
313
87
|
|
|
314
|
-
|
|
88
|
+
**Out of mandate.** Record nothing for style, naming, or formatting, which the linter owns. Record nothing for design and abstraction opinions, for requirements the plan does not state, for speculative hardening, or for tests the plan never asked for. Missing "this works, matches the plan, and has the wrong abstraction" is deliberate. Do not widen scope to recover it.
|
|
315
89
|
|
|
316
|
-
|
|
317
|
-
reconstructed idea of the requirements.
|
|
318
|
-
- **The diff is empty.** Emit a `<review>` with no `<file>` children and report
|
|
319
|
-
zero findings. This is not an error.
|
|
320
|
-
- **A finding will not fit the schema.** Fix the finding's shape, never the
|
|
321
|
-
schema. A fix that cannot be a local text replacement is recorded without a
|
|
322
|
-
suggestion.
|
|
323
|
-
- **You are tempted to apply a fix yourself.** Stop. Detection and remediation
|
|
324
|
-
run on different routes on purpose — nobody marks their own homework.
|
|
90
|
+
**Failure modes.** Cannot read the plan: stop and report, and never review against requirements you reconstructed. Empty diff: emit `<review>` with no children and report zero findings, which counts as success. A finding will not fit the schema: fix the shape of the finding, never the schema. Tempted to apply a fix yourself: stop, because detection and remediation run on separate routes by design.
|