@nathapp/nax 0.80.1 → 0.81.0-canary.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/nax.js +44737 -42009
- package/package.json +6 -6
- package/flows/nax-finish/commit-message.ts +0 -239
- package/flows/nax-finish/errors.ts +0 -23
- package/flows/nax-finish/exec.ts +0 -135
- package/flows/nax-finish/findings-parse.ts +0 -150
- package/flows/nax-finish/flow-ctx.ts +0 -150
- package/flows/nax-finish/narrative.ts +0 -215
- package/flows/nax-finish/nax-finish.flow.ts +0 -566
- package/flows/nax-finish/pr-template-merge.ts +0 -253
- package/flows/nax-finish/pr-template.ts +0 -56
- package/flows/nax-finish/pr-title.ts +0 -140
- package/flows/nax-finish/review-prompts.ts +0 -468
- package/flows/nax-finish/steps/acceptance.ts +0 -76
- package/flows/nax-finish/steps/commit-round.ts +0 -67
- package/flows/nax-finish/steps/context.ts +0 -158
- package/flows/nax-finish/steps/escalate.ts +0 -93
- package/flows/nax-finish/steps/forge.ts +0 -93
- package/flows/nax-finish/steps/gates.ts +0 -183
- package/flows/nax-finish/steps/git.ts +0 -130
- package/flows/nax-finish/steps/index.ts +0 -13
- package/flows/nax-finish/steps/pr-body.ts +0 -462
- package/flows/nax-finish/steps/pr-narrative.ts +0 -46
- package/flows/nax-finish/steps/pr.ts +0 -116
- package/flows/nax-finish/steps/quality.ts +0 -102
- package/flows/nax-finish/steps/result.ts +0 -112
- package/flows/nax-finish/steps/review-audit.ts +0 -91
- package/flows/nax-finish/steps/review-round.ts +0 -109
- package/flows/nax-finish/types.ts +0 -260
- package/flows/nax-finish/verdict.ts +0 -210
|
@@ -1,468 +0,0 @@
|
|
|
1
|
-
import type { Finding } from "./types";
|
|
2
|
-
|
|
3
|
-
export const SPEC_REVIEW_DIMENSIONS = `# Spec-relative review dimensions
|
|
4
|
-
|
|
5
|
-
Reference for the post-impl-review **spec-relative** pass: Compliance, Drift,
|
|
6
|
-
Integration, and Convention Compliance. Apply every dimension below against the
|
|
7
|
-
spec, the filtered diff, the collaborator code you read, and the loaded project
|
|
8
|
-
rules.
|
|
9
|
-
|
|
10
|
-
## Map external touchpoints first (read the unchanged collaborators)
|
|
11
|
-
|
|
12
|
-
**Do this before judging anything.** Most real defects in a focused diff live on
|
|
13
|
-
the boundary between the changed code and the *unchanged* code it calls into —
|
|
14
|
-
and that unchanged code is, by definition, not in the diff. A diff-only read
|
|
15
|
-
cannot see them.
|
|
16
|
-
|
|
17
|
-
From the filtered diff, build a list of **external touchpoints** — every symbol
|
|
18
|
-
the changed code *uses* but does not *define* in the diff:
|
|
19
|
-
|
|
20
|
-
- **Callees:** functions/methods the new lines call whose body lives in an
|
|
21
|
-
unchanged file (e.g. \`strategy_instance.get_references(...)\`,
|
|
22
|
-
\`provider.cache.get_ohlcv(...)\`).
|
|
23
|
-
- **Polymorphic / interface calls:** any call dispatched through a base class,
|
|
24
|
-
protocol, or registry. The diff sees one signature; the real behaviour is in
|
|
25
|
-
*every concrete implementation*. Enumerate them.
|
|
26
|
-
- **New or changed arguments to existing APIs:** a value the diff now passes that
|
|
27
|
-
the callee didn't receive before — especially empty/sentinel/\`None\`/\`{}\`/\`[]\`
|
|
28
|
-
values, or a newly-shaped object. Verify the callee tolerates it.
|
|
29
|
-
- **Consumers of changed outputs:** unchanged code that reads a field, sentinel,
|
|
30
|
-
or return value whose meaning the diff altered.
|
|
31
|
-
- **Collaborators named in the spec:** if the spec asserts a cross-cutting goal
|
|
32
|
-
("every built-in strategy works", "all callers", "each adapter"), that goal is
|
|
33
|
-
a claim *about unchanged code*. You must open those files to verify it — the
|
|
34
|
-
diff alone can never prove it.
|
|
35
|
-
|
|
36
|
-
For each touchpoint, **read the actual definition(s)** with Read/Grep and check
|
|
37
|
-
the changed code's assumption holds for *all* of them, not just the convenient
|
|
38
|
-
case. Use Grep to find every implementation of an overridden method before
|
|
39
|
-
concluding it's safe.
|
|
40
|
-
|
|
41
|
-
Examples of assumptions that only break in unchanged code:
|
|
42
|
-
- The diff calls \`iface.method(emptyValue)\`; one concrete implementation
|
|
43
|
-
immediately indexes a required key → runtime \`KeyError\`/\`NullPointerException\`
|
|
44
|
-
for that case.
|
|
45
|
-
- The diff sets a sentinel to \`{}\` instead of \`None\`; a downstream guard checks
|
|
46
|
-
\`is None\`, so \`{}\` slips through and produces a misleading error or silent NaN.
|
|
47
|
-
- The spec says "works for every strategy"; the diff only added tests for
|
|
48
|
-
static/test-double strategies, leaving the dynamic real ones unverified.
|
|
49
|
-
|
|
50
|
-
Treat an untested cross-cutting claim as **unverified, not satisfied** — surface
|
|
51
|
-
it as a finding rather than assuming coverage.
|
|
52
|
-
|
|
53
|
-
## Compliance — per AC/story/requirement
|
|
54
|
-
|
|
55
|
-
For each numbered or named AC, story, or requirement in the spec, determine:
|
|
56
|
-
- **Covered** — the diff clearly addresses it
|
|
57
|
-
- **Partial** — the diff touches it but leaves something incomplete
|
|
58
|
-
- **Missing** — nothing in the diff implements it
|
|
59
|
-
|
|
60
|
-
**Coverage ≠ correctness:** when an AC's coverage is a test, do not stop at "a
|
|
61
|
-
test exists." Open the test body and verify it (a) restores any global /
|
|
62
|
-
\`os.environ\` / filesystem / singleton state it mutates (teardown or fixture),
|
|
63
|
-
(b) is deterministic and order-independent, and (c) asserts the AC's actual
|
|
64
|
-
behaviour rather than a tautology. A test that passes only by accident of
|
|
65
|
-
ordering, or that asserts nothing meaningful, is **Partial**, not Covered.
|
|
66
|
-
|
|
67
|
-
**If the spec has no numbered or named ACs** (it's written as prose): derive
|
|
68
|
-
implicit requirements from the prose — treat each described behaviour, endpoint,
|
|
69
|
-
or constraint as a requirement. Note in the findings header: \`(Spec has no
|
|
70
|
-
structured ACs — requirements inferred from prose)\`.
|
|
71
|
-
|
|
72
|
-
**Renames and deletions:** treat them as intentional changes when evaluating
|
|
73
|
-
compliance. A diff showing \`rename from A to B\` or a deleted file counts as
|
|
74
|
-
coverage for an AC that required moving or removing that module.
|
|
75
|
-
|
|
76
|
-
## Drift — holistic across the diff
|
|
77
|
-
|
|
78
|
-
Check whether the implementation matches the spec's described intent:
|
|
79
|
-
- API shape: do endpoints, request fields, response fields, and status codes
|
|
80
|
-
match?
|
|
81
|
-
- Approach: is the architectural pattern (module structure, design pattern, data
|
|
82
|
-
flow) what the spec called for?
|
|
83
|
-
- Constraints: are hard requirements respected (e.g. "must use HMAC-SHA256",
|
|
84
|
-
"must be idempotent", "must validate at startup")?
|
|
85
|
-
- Naming: do key identifiers (routes, types, env vars, functions) match the
|
|
86
|
-
spec's terminology?
|
|
87
|
-
|
|
88
|
-
## Integration — does the changed code actually work against the unchanged collaborators?
|
|
89
|
-
|
|
90
|
-
For each external touchpoint mapped above, check whether the changed code's
|
|
91
|
-
assumptions hold for *every* real implementation/consumer:
|
|
92
|
-
- Will any concrete callee raise (KeyError, NPE, ValueError, panic) for an input
|
|
93
|
-
the diff now passes — especially empty/sentinel/\`None\`/\`[]\`/\`{}\` values?
|
|
94
|
-
- Does any sentinel or output the diff changed reach a downstream guard that
|
|
95
|
-
interprets it the wrong way (\`{}\` slipping past an \`is None\` check; \`[]\`
|
|
96
|
-
treated as "provided")?
|
|
97
|
-
- Does the spec's cross-cutting claim ("every X works") actually hold for the
|
|
98
|
-
real, non-test-double implementations — and is each one exercised by a test?
|
|
99
|
-
An untested real path is a finding, not a pass.
|
|
100
|
-
- Are there edge inputs the new tool/endpoint schema now permits (e.g. an
|
|
101
|
-
explicitly empty array) that route into a broken branch?
|
|
102
|
-
|
|
103
|
-
## Convention Compliance — does the diff obey the project's own rules?
|
|
104
|
-
|
|
105
|
-
**Load the rules first.** Find the repo's own rule files and read every one that
|
|
106
|
-
exists (they may all be absent — then skip this dimension entirely and note
|
|
107
|
-
\`(No project rule files found — Convention Compliance skipped)\` in your findings):
|
|
108
|
-
|
|
109
|
-
\`\`\`bash
|
|
110
|
-
ls CLAUDE.md AGENTS.md 2>/dev/null
|
|
111
|
-
find .nax/rules .claude/rules -name "*.md" 2>/dev/null
|
|
112
|
-
\`\`\`
|
|
113
|
-
|
|
114
|
-
\`.nax/rules/\` takes priority over \`.claude/rules/\` when they conflict (nax-native
|
|
115
|
-
is canonical). Honour each file's \`paths:\` / \`appliesTo:\` frontmatter if present —
|
|
116
|
-
a rule scoped to \`src/agents/**\` does not apply to a diff under \`apps/web/\`.
|
|
117
|
-
Extract the concrete, checkable directives (forbidden APIs, required patterns,
|
|
118
|
-
naming, logging fields, file-size limits) and hold them for the checks below.
|
|
119
|
-
|
|
120
|
-
For each concrete directive extracted from the loaded rule files, check whether
|
|
121
|
-
the changed lines violate it. Only flag rules that actually apply to the changed
|
|
122
|
-
files (respect \`paths:\` / \`appliesTo:\` scoping). Examples of the *kind* of
|
|
123
|
-
directive to check — the real list comes from the loaded files, not this list:
|
|
124
|
-
- Forbidden APIs / patterns (e.g. a banned import, \`console.log\` in source, a
|
|
125
|
-
Node API in a Bun-native repo, hardcoded patterns the project routes through a
|
|
126
|
-
resolver).
|
|
127
|
-
- Required structure (barrel imports vs internal paths, file-size limits,
|
|
128
|
-
mandated error/base classes, dependency-injection patterns).
|
|
129
|
-
- Required fields / format (e.g. a mandated structured-log field,
|
|
130
|
-
conventional-commit style, naming conventions for routes/types/env vars).
|
|
131
|
-
|
|
132
|
-
Cite the specific rule file and directive in the finding
|
|
133
|
-
(\`forbidden-patterns.md: no console.log in src/\`). A violation of an explicit,
|
|
134
|
-
in-scope project rule is a real finding; a generic style opinion **not** backed
|
|
135
|
-
by a loaded rule is not — do not invent rules. If no rule files were found, skip
|
|
136
|
-
this dimension entirely.
|
|
137
|
-
|
|
138
|
-
## Confidence threshold (spec-relative)
|
|
139
|
-
|
|
140
|
-
**Report only findings you are ≥80% confident are real**, not pre-existing ones
|
|
141
|
-
the diff didn't introduce. A false "AC missing" or a phantom integration crash
|
|
142
|
-
is expensive and erodes trust in the whole report. A missing AC, a runtime crash
|
|
143
|
-
you traced through the collaborator, and an in-scope project-rule violation clear
|
|
144
|
-
this bar easily; a "this might be slow" hunch you haven't reasoned through does
|
|
145
|
-
not — drop it.
|
|
146
|
-
`;
|
|
147
|
-
|
|
148
|
-
export const QUALITY_REVIEW_DIMENSIONS = `# Code quality & test integrity dimension
|
|
149
|
-
|
|
150
|
-
Reference for the post-impl-review **code-quality** pass: spec-independent
|
|
151
|
-
defects and design/maintainability concerns in the changed lines. These findings
|
|
152
|
-
are real regardless of what the spec says. Scan the diff — production **and**
|
|
153
|
-
test code — and judge each changed function on its own merits.
|
|
154
|
-
|
|
155
|
-
## Forcing function — enumerate before you conclude
|
|
156
|
-
|
|
157
|
-
Before reporting, walk **every function/method the diff adds or changes** and
|
|
158
|
-
write yourself a one-line verdict for each: *earns its place* or *concern: …*.
|
|
159
|
-
This enumeration is a thinking tool — it does not go in the final report, only
|
|
160
|
-
the resulting findings do. Skipping it is how real maintainability issues get
|
|
161
|
-
missed: an agent that pattern-matches a few obvious smells and stops will always
|
|
162
|
-
under-report. Look at each changed function deliberately.
|
|
163
|
-
|
|
164
|
-
## What to look for
|
|
165
|
-
|
|
166
|
-
Report only concrete, objective issues, not style preferences:
|
|
167
|
-
|
|
168
|
-
- **Test isolation:** mutating \`os.environ\` / globals / singletons / filesystem
|
|
169
|
-
without teardown; cross-test ordering dependence; shared mutable fixtures. A
|
|
170
|
-
test that only passes because another test happens to clean up after it is a
|
|
171
|
-
defect even when the suite is currently green.
|
|
172
|
-
- **Dead / redundant code:** assignments with no effect, unreachable branches,
|
|
173
|
-
set-up the constructor already performed, unused locals introduced by the diff;
|
|
174
|
-
logic duplicated from an existing helper the diff could have reused.
|
|
175
|
-
- **Resource leaks:** opened files / sockets / handles / subprocesses not closed;
|
|
176
|
-
timers / listeners not cleared.
|
|
177
|
-
- **Error handling:** swallowed exceptions, bare catches that hide failures,
|
|
178
|
-
missing validation on a newly-introduced input path.
|
|
179
|
-
- **Concurrency:** shared state mutated without synchronisation; \`await\` inside a
|
|
180
|
-
loop that should be batched; a race between a check and the action it guards.
|
|
181
|
-
- **Performance:** N+1 queries or network calls in a loop; blocking I/O on a hot
|
|
182
|
-
path; an obviously quadratic scan over a large collection the diff introduces.
|
|
183
|
-
- **Accessibility (UI diffs only):** interactive elements without an accessible
|
|
184
|
-
name/label, missing \`alt\`, non-keyboard-reachable controls, form inputs with
|
|
185
|
-
no associated label.
|
|
186
|
-
- **Security (only when the diff touches it):** hardcoded secrets, unvalidated
|
|
187
|
-
user input reaching a sink, injection vectors.
|
|
188
|
-
- **Type safety:** unsafe casts, \`any\`/\`unknown\` escapes, weakened or widened
|
|
189
|
-
types the diff introduces, or a narrowing the diff drops — anywhere the change
|
|
190
|
-
trades a compile-time guarantee for a runtime risk.
|
|
191
|
-
- **Design & maintainability (open-ended — not a closed checklist):** a changed
|
|
192
|
-
function that conflates multiple responsibilities (poor separation of
|
|
193
|
-
concerns); an abstraction the diff introduces that is premature (single caller,
|
|
194
|
-
speculative generality) or leaky (callers must understand its internals to use
|
|
195
|
-
it safely); logic the diff writes inline that an existing helper already
|
|
196
|
-
provides (reinvention, not just literal duplication); control flow so nested or
|
|
197
|
-
convoluted the next reader will misread it; an identifier whose name actively
|
|
198
|
-
misleads about what it holds or does; a comment or docstring the diff now leaves
|
|
199
|
-
stale or contradicting the code it sits on, or a changed public API left without
|
|
200
|
-
the docs a caller needs; an edge case the changed code's *own* logic implies but
|
|
201
|
-
doesn't handle. These are the qualitative "this code isn't good yet" findings —
|
|
202
|
-
judge them, don't skip them because they aren't on the defect list above. Anchor
|
|
203
|
-
each to the changed line and state the concrete cost (what breaks, or who is
|
|
204
|
-
misled, later).
|
|
205
|
-
|
|
206
|
-
Every finding here must point at a specific changed line and name a concrete cost
|
|
207
|
-
— a bug, a future break, or a reader who will be misled. Skip pure formatting and
|
|
208
|
-
personal taste that carry no such cost, and skip hypotheticals about code outside
|
|
209
|
-
the diff. But a design or maintainability concern grounded in a changed line and
|
|
210
|
-
its cost **is** in scope even though it's not on the defect checklist above —
|
|
211
|
-
that is exactly the signal this dimension exists to surface.
|
|
212
|
-
|
|
213
|
-
## Confidence threshold (code quality)
|
|
214
|
-
|
|
215
|
-
**Report findings you are ≥60% confident are real**, *provided* each is anchored
|
|
216
|
-
to a specific changed line and names a concrete maintenance or correctness cost.
|
|
217
|
-
Design and maintainability problems are inherently probabilistic — a muddy
|
|
218
|
-
abstraction, a misleading name, or a fragile edge case rarely clears 80%, and a
|
|
219
|
-
blanket 80% gate is precisely what makes a review miss the quality issues it
|
|
220
|
-
exists to catch. Let these land as MEDIUM or LOW per the severity table rather
|
|
221
|
-
than dropping them; the implementer can waive them, but they should see them.
|
|
222
|
-
Still exclude pure formatting and personal taste that carry no stated cost. Do
|
|
223
|
-
**not** over-suppress this tier to hit an arbitrary count — a real
|
|
224
|
-
maintainability concern stated with its cost is worth surfacing even at moderate
|
|
225
|
-
confidence.
|
|
226
|
-
`;
|
|
227
|
-
|
|
228
|
-
export const WORKER_PROTOCOL = `# Worker protocol (shared mechanics)
|
|
229
|
-
|
|
230
|
-
Self-contained mechanics for a post-impl-review **worker** (SPEC or QUALITY).
|
|
231
|
-
Read this plus your dimension reference (\`spec-review.md\` or \`code-quality.md\`) —
|
|
232
|
-
you do **not** need to read the dispatcher's \`SKILL.md\`. The dispatcher already
|
|
233
|
-
resolved the spec, detected the base branch, and ran the empty-diff/size guards;
|
|
234
|
-
your job is to review the diff and return findings.
|
|
235
|
-
|
|
236
|
-
## Get the diff
|
|
237
|
-
|
|
238
|
-
The dispatcher gave you the base branch. Fetch the diff content:
|
|
239
|
-
|
|
240
|
-
\`\`\`bash
|
|
241
|
-
git diff origin/<branch>...HEAD --name-only # changed file list
|
|
242
|
-
git diff origin/<branch>...HEAD # full diff content
|
|
243
|
-
\`\`\`
|
|
244
|
-
|
|
245
|
-
## Filter noise
|
|
246
|
-
|
|
247
|
-
Do not treat churn in these as a reviewable change (you may still *read* them as
|
|
248
|
-
context):
|
|
249
|
-
|
|
250
|
-
- Lockfiles: \`bun.lock\`, \`package-lock.json\`, \`yarn.lock\`, \`pnpm-lock.yaml\`,
|
|
251
|
-
\`Cargo.lock\`, \`poetry.lock\`, or anything ending in \`.lock\`.
|
|
252
|
-
- Generated output: files in \`dist/\`, \`build/\`, \`.next/\`, \`.turbo/\`,
|
|
253
|
-
\`__pycache__/\`, or matching \`*.generated.*\`.
|
|
254
|
-
- nax artifacts: anything under a \`.nax/\` directory at **any depth**
|
|
255
|
-
(\`**/.nax/**\` — root or nested per-package in a monorepo): specs, PRDs,
|
|
256
|
-
acceptance result JSON, config, and the generated acceptance tests.
|
|
257
|
-
- Binary files: git marks these \`Binary files a/... and b/... differ\` — skip them.
|
|
258
|
-
|
|
259
|
-
## Read the unchanged collaborators (before judging)
|
|
260
|
-
|
|
261
|
-
Most real defects in a focused diff live on the boundary between the changed code
|
|
262
|
-
and the *unchanged* code it calls into — which is, by definition, not in the
|
|
263
|
-
diff. Build the list of external touchpoints (every symbol the changed code
|
|
264
|
-
*uses* but does not *define*: callees, polymorphic/interface calls, new arguments
|
|
265
|
-
to existing APIs, consumers of changed outputs, collaborators named in the spec)
|
|
266
|
-
and read each definition with Read/Grep before concluding. Treat an untested
|
|
267
|
-
cross-cutting claim as **unverified, not satisfied**. (The SPEC dimension file
|
|
268
|
-
has the full procedure with worked examples; the QUALITY worker needs the same
|
|
269
|
-
reads to judge an integration-shaped defect.)
|
|
270
|
-
|
|
271
|
-
## Severity table
|
|
272
|
-
|
|
273
|
-
| Severity | Meaning |
|
|
274
|
-
|:---------|:--------|
|
|
275
|
-
| CRITICAL | AC entirely missing; implementation directly contradicts a hard spec requirement; the changed code raises/crashes at runtime for a case the spec requires to work; or a security defect the diff introduces (hardcoded secret, injection sink) |
|
|
276
|
-
| HIGH | Significant drift (wrong API shape, missing constraint, wrong architectural approach); an integration defect that breaks a real collaborator the spec depends on; or a violation of a project rule explicitly marked as required/forbidden (a banned API, a hard-blocked pattern) |
|
|
277
|
-
| MEDIUM | Partial coverage — AC present but incomplete; minor drift affecting correctness; an integration gap reachable through a now-permitted input; a test-isolation defect that can cause false positives or flakiness under reordering/parallelism; a resource leak; a swallowed error on a real path; a concurrency/race or performance regression the diff introduces; or an accessibility defect on a new interactive UI element |
|
|
278
|
-
| LOW | Minor naming deviation, style mismatch, dead/redundant/duplicated code, unused locals, a soft convention deviation, or other non-blocking gap |
|
|
279
|
-
|
|
280
|
-
## Output format — return ONLY this
|
|
281
|
-
|
|
282
|
-
Return **only your findings**, nothing else: no \`Spec:\`/\`Base:\` header, no
|
|
283
|
-
\`FINDINGS\` divider, no \`VERDICT\` line (the dispatcher adds those). Emit each
|
|
284
|
-
finding as a block:
|
|
285
|
-
|
|
286
|
-
\`\`\`
|
|
287
|
-
[SEVERITY] <short title>
|
|
288
|
-
Problem: <what's wrong, with file/line and the concrete cost>
|
|
289
|
-
Fix: <the concrete change, or "note intentional deviation">
|
|
290
|
-
\`\`\`
|
|
291
|
-
|
|
292
|
-
If you found nothing in your group, return the literal line \`No findings.\` as
|
|
293
|
-
your entire final message. That message is the only thing that travels back to
|
|
294
|
-
the dispatcher.
|
|
295
|
-
`;
|
|
296
|
-
|
|
297
|
-
const CLASSIFIER = [
|
|
298
|
-
"Mark a finding for human judgment — and only such a finding — by adding a",
|
|
299
|
-
"`Judgment: yes — <why>` line to its block. Use it when the finding is a spec",
|
|
300
|
-
"conflict, or a design/judgment call with no safe mechanical fix. Everything",
|
|
301
|
-
"else will be fixed and re-verified automatically, so a finding with a clear,",
|
|
302
|
-
"low-risk fix must NOT carry the marker.",
|
|
303
|
-
].join("\n");
|
|
304
|
-
|
|
305
|
-
/**
|
|
306
|
-
* The reply contract.
|
|
307
|
-
*
|
|
308
|
-
* This supersedes the "Output format — return ONLY this" section of
|
|
309
|
-
* `WORKER_PROTOCOL` above, which is kept verbatim so it stays diffable against
|
|
310
|
-
* the skill's `references/worker-protocol.md`. Its `[SEVERITY]` blocks are
|
|
311
|
-
* exactly this contract's `## FINDINGS` section; the two sections before it are
|
|
312
|
-
* what the flow adds, because the flow — unlike the skill's dispatcher — has no
|
|
313
|
-
* human reading the reply and so must be able to check the obligations itself.
|
|
314
|
-
*
|
|
315
|
-
* There is no JSON. A reply constrained to one JSON object has nowhere to put
|
|
316
|
-
* the per-AC and per-function enumerations both dimension references depend on,
|
|
317
|
-
* and an unreadable object used to discard the entire review (#1614).
|
|
318
|
-
*/
|
|
319
|
-
function outputContract(phase: "spec" | "quality"): string {
|
|
320
|
-
const walk =
|
|
321
|
-
phase === "spec"
|
|
322
|
-
? "one line per AC in the spec: `AC-3 Covered|Partial|Missing — <one clause>`"
|
|
323
|
-
: "one line per function or method the diff adds or changes: `path.ts:name — earns its place|concern: <one clause>`";
|
|
324
|
-
return `# Reply contract — your reply must be these three sections, in this order
|
|
325
|
-
|
|
326
|
-
## TOUCHPOINTS
|
|
327
|
-
One line per external touchpoint whose definition you actually opened:
|
|
328
|
-
\`- path/to/file.ts:symbol — why you opened it\`. If the diff genuinely has none,
|
|
329
|
-
the single line \`- none — <justification>\`. The paths are checked against the
|
|
330
|
-
repo: a list whose paths do not exist is treated as an incomplete review, not a
|
|
331
|
-
clean one, and you will be asked again.
|
|
332
|
-
|
|
333
|
-
## WALK
|
|
334
|
-
${walk}. This is the enumeration your dimension reference requires. One line each,
|
|
335
|
-
no prose. A missing or empty WALK section is an incomplete review.
|
|
336
|
-
|
|
337
|
-
## FINDINGS
|
|
338
|
-
The \`[SEVERITY] …\` blocks defined in the worker protocol above — or the literal
|
|
339
|
-
line \`No findings.\` if you found none. This section is the only thing that
|
|
340
|
-
becomes a finding; the two above are the evidence that you were in a position to
|
|
341
|
-
write it.`;
|
|
342
|
-
}
|
|
343
|
-
|
|
344
|
-
/**
|
|
345
|
-
* Build the reviewer prompt.
|
|
346
|
-
*
|
|
347
|
-
* With `since` set this is a **re-review**: the same reviewer already read the
|
|
348
|
-
* whole branch and raised `priorFindings`, a fix was applied and committed, and
|
|
349
|
-
* the only new material is `since..HEAD`. Re-reading the full branch diff every
|
|
350
|
-
* round made reviews 58% of the flow's wall clock, most of it re-reading code an
|
|
351
|
-
* earlier round had already cleared. The narrowed round still has the full repo
|
|
352
|
-
* available — it is told to open whatever the fix touches — it just is not asked
|
|
353
|
-
* to re-derive a verdict on unchanged code.
|
|
354
|
-
*
|
|
355
|
-
* `since` is the parent of the *first* commit that landed after the previous
|
|
356
|
-
* verdict, not of the latest one (see `incrementalSince`) — the acceptance loop
|
|
357
|
-
* can commit between a spec fix and its re-review, and the window has to span
|
|
358
|
-
* both. So `since..HEAD` provably contains every change made since that
|
|
359
|
-
* verdict, however many commits that took, and it is never supplied at all when
|
|
360
|
-
* no commit landed.
|
|
361
|
-
*/
|
|
362
|
-
export function buildReviewPrompt(
|
|
363
|
-
phase: "spec" | "quality",
|
|
364
|
-
args: {
|
|
365
|
-
base: string;
|
|
366
|
-
specPath: string;
|
|
367
|
-
since?: string | null;
|
|
368
|
-
priorFindings?: Finding[];
|
|
369
|
-
gaps?: string[];
|
|
370
|
-
},
|
|
371
|
-
): string {
|
|
372
|
-
const dims = phase === "spec" ? SPEC_REVIEW_DIMENSIONS : QUALITY_REVIEW_DIMENSIONS;
|
|
373
|
-
const gapNotice =
|
|
374
|
-
args.gaps && args.gaps.length > 0
|
|
375
|
-
? [
|
|
376
|
-
[
|
|
377
|
-
"IMPORTANT — your previous review was not accepted, because it skipped a required section:",
|
|
378
|
-
...args.gaps.map((g) => `- ${g}`),
|
|
379
|
-
"Do the reading this time and emit all three sections. A verdict without them is not a review.",
|
|
380
|
-
].join("\n"),
|
|
381
|
-
]
|
|
382
|
-
: [];
|
|
383
|
-
if (!args.since) {
|
|
384
|
-
return [
|
|
385
|
-
...gapNotice,
|
|
386
|
-
`You are the ${phase.toUpperCase()} reviewer for a completed feature.`,
|
|
387
|
-
`The spec/requirements source is: ${args.specPath}. Read it in full.`,
|
|
388
|
-
`Fetch and review the diff: \`git diff ${args.base}...HEAD\` (also \`--name-only\` for the file list).`,
|
|
389
|
-
WORKER_PROTOCOL,
|
|
390
|
-
dims,
|
|
391
|
-
CLASSIFIER,
|
|
392
|
-
outputContract(phase),
|
|
393
|
-
].join("\n\n");
|
|
394
|
-
}
|
|
395
|
-
return [
|
|
396
|
-
...gapNotice,
|
|
397
|
-
`You are the ${phase.toUpperCase()} reviewer for a completed feature, continuing a review you already started.`,
|
|
398
|
-
`On your previous pass over \`git diff ${args.base}...HEAD\` you raised the findings below, and they have since been fixed and committed. Everything else in that diff you already judged acceptable — do not re-derive a verdict on it.`,
|
|
399
|
-
`Your findings from the previous pass:\n${JSON.stringify(args.priorFindings ?? [], null, 2)}`,
|
|
400
|
-
`The fix is \`git diff ${args.since}..HEAD\` — this is the only code that has changed since your last verdict. Review it, and only it, for two questions:`,
|
|
401
|
-
[
|
|
402
|
-
"1. **Resolved?** Does the fix actually resolve each finding above? A finding that was papered over (assertion weakened, test deleted, check disabled) is NOT resolved — re-raise it.",
|
|
403
|
-
"2. **Broken?** Did the fix introduce a new problem, in the changed lines or in the unchanged code they now call into?",
|
|
404
|
-
"",
|
|
405
|
-
`Read whatever files you need — the spec is at ${args.specPath} and the whole repo is available. Scope means *what you judge*, not *what you may read*.`,
|
|
406
|
-
].join("\n"),
|
|
407
|
-
WORKER_PROTOCOL,
|
|
408
|
-
dims,
|
|
409
|
-
CLASSIFIER,
|
|
410
|
-
outputContract(phase),
|
|
411
|
-
].join("\n\n");
|
|
412
|
-
}
|
|
413
|
-
|
|
414
|
-
/**
|
|
415
|
-
* The fix node's contract.
|
|
416
|
-
*
|
|
417
|
-
* For the two review phases it is no longer "apply these". Every finding used to
|
|
418
|
-
* be implemented unconditionally, so a false positive was always built — and on
|
|
419
|
-
* the diff behind #1614 the comparison review raised one finding a human
|
|
420
|
-
* withdrew, because the change it proposed contradicted a test deliberately
|
|
421
|
-
* pinning the current behaviour. `quality_gates` would then have proved the
|
|
422
|
-
* resulting suite green. Reporting more findings without a way to reject one
|
|
423
|
-
* scales the false-positive exposure with the true-positive one, so the two ship
|
|
424
|
-
* together.
|
|
425
|
-
*
|
|
426
|
-
* A rejection must cite the file:line that pins the behaviour: an unevidenced
|
|
427
|
-
* "I disagree" is how a real finding gets waived, and `commit_<phase>` checks
|
|
428
|
-
* that the cited path exists.
|
|
429
|
-
*/
|
|
430
|
-
export function fixPrompt(
|
|
431
|
-
phase: "acceptance" | "spec" | "quality" | "gate",
|
|
432
|
-
ctx: { outputs: Record<string, unknown> },
|
|
433
|
-
): string {
|
|
434
|
-
const outs = ctx.outputs as Record<string, { findings?: Finding[]; output?: string }>;
|
|
435
|
-
if (phase === "gate" || phase === "acceptance") {
|
|
436
|
-
const detail = phase === "gate" ? (outs.quality_gates?.output ?? "") : (outs.acceptance?.output ?? "");
|
|
437
|
-
return [
|
|
438
|
-
`Apply the recommended fixes for the ${phase} phase, directly in the repo.`,
|
|
439
|
-
"Do not commit, push, or open PRs — nax-finish commits and pushes your edits itself.",
|
|
440
|
-
`Context:\n${detail}`,
|
|
441
|
-
"After fixing, re-run the feature's acceptance tests and the relevant checks; only proceed when they pass.",
|
|
442
|
-
'Return exactly {"route":"proceed"} when done and green.',
|
|
443
|
-
].join("\n\n");
|
|
444
|
-
}
|
|
445
|
-
const findings = outs[`review_${phase}`]?.findings ?? [];
|
|
446
|
-
const numbered = findings
|
|
447
|
-
.map((f, i) => `[${i + 1}] [${f.severity}] ${f.title}\n Problem: ${f.problem}\n Fix: ${f.fix}`)
|
|
448
|
-
.join("\n");
|
|
449
|
-
return [
|
|
450
|
-
`Resolve the ${phase} review findings below, directly in the repo.`,
|
|
451
|
-
"Do not commit, push, or open PRs — nax-finish commits and pushes your edits itself.",
|
|
452
|
-
numbered,
|
|
453
|
-
[
|
|
454
|
-
"A finding may be REJECTED rather than fixed — but only on evidence, not on preference.",
|
|
455
|
-
"Reject when the change it asks for would contradict an existing test that deliberately",
|
|
456
|
-
"pins the current behaviour, or a spec statement that requires it. Cite the `file:line`.",
|
|
457
|
-
"Anything else you fix.",
|
|
458
|
-
].join("\n"),
|
|
459
|
-
[
|
|
460
|
-
"After fixing, re-run the feature's acceptance tests and the relevant checks; only proceed when they pass.",
|
|
461
|
-
"Then end your reply with one line per finding, in this exact shape:",
|
|
462
|
-
"",
|
|
463
|
-
"## DISPOSITIONS",
|
|
464
|
-
"[1] fixed",
|
|
465
|
-
"[2] rejected — evidence: test/unit/foo.test.ts:42",
|
|
466
|
-
].join("\n"),
|
|
467
|
-
].join("\n\n");
|
|
468
|
-
}
|
|
@@ -1,76 +0,0 @@
|
|
|
1
|
-
import { DEFAULT_ACCEPTANCE_TIMEOUT_MS, runShell } from "../exec";
|
|
2
|
-
import type { AcceptanceGroup, ShellRunFn } from "../types";
|
|
3
|
-
|
|
4
|
-
export const _acceptanceDeps: { runShell: ShellRunFn } = { runShell };
|
|
5
|
-
|
|
6
|
-
/** Single-quote a path for `/bin/sh -c`, so spaces in a repo path can't split the command. */
|
|
7
|
-
function shellQuote(value: string): string {
|
|
8
|
-
return `'${value.replaceAll("'", `'\\''`)}'`;
|
|
9
|
-
}
|
|
10
|
-
|
|
11
|
-
/**
|
|
12
|
-
* Build the acceptance command for one group, mirroring nax's own execution:
|
|
13
|
-
* the configured `command` template (any `{{FILE}}` / `{{file}}` / `{{files}}`
|
|
14
|
-
* placeholder replaced by the **absolute** test path) run from the group's
|
|
15
|
-
* package dir. The command is a string, run through `/bin/sh -c`, because
|
|
16
|
-
* configured commands legitimately contain `&&`, quotes and flags.
|
|
17
|
-
*/
|
|
18
|
-
export function buildAcceptanceCommand(repoRoot: string, group: AcceptanceGroup): string {
|
|
19
|
-
const absFile = shellQuote(`${repoRoot}/${group.testPath}`);
|
|
20
|
-
const template = group.command ?? `${languageRunner(group.language)} {{FILE}}`;
|
|
21
|
-
return template.replace(/\{\{FILE\}\}|\{\{file\}\}|\{\{files\}\}/g, absFile);
|
|
22
|
-
}
|
|
23
|
-
|
|
24
|
-
export interface AcceptanceGateOutcome {
|
|
25
|
-
/** Every group that ran exited 0. Says nothing about groups that could not run. */
|
|
26
|
-
passed: boolean;
|
|
27
|
-
ran: number;
|
|
28
|
-
/**
|
|
29
|
-
* Package names whose acceptance test the resolver expected at its canonical
|
|
30
|
-
* path but which is absent on disk — never generated, or generation failed.
|
|
31
|
-
* The caller must treat a non-empty list as a coverage hole rather than a
|
|
32
|
-
* pass: skipping these silently is how a feature reached a "ready" PR with
|
|
33
|
-
* nothing verified.
|
|
34
|
-
*/
|
|
35
|
-
missing: string[];
|
|
36
|
-
output: string;
|
|
37
|
-
}
|
|
38
|
-
|
|
39
|
-
export async function runAcceptanceGate(
|
|
40
|
-
repoRoot: string,
|
|
41
|
-
groups: AcceptanceGroup[],
|
|
42
|
-
opts: { timeoutMs?: number } = {},
|
|
43
|
-
): Promise<AcceptanceGateOutcome> {
|
|
44
|
-
const chunks: string[] = [];
|
|
45
|
-
const timeoutMs = opts.timeoutMs ?? DEFAULT_ACCEPTANCE_TIMEOUT_MS;
|
|
46
|
-
const missing: string[] = [];
|
|
47
|
-
let ran = 0;
|
|
48
|
-
for (const g of groups) {
|
|
49
|
-
const name = g.packageDir || "root";
|
|
50
|
-
if (!g.exists) {
|
|
51
|
-
missing.push(name);
|
|
52
|
-
continue;
|
|
53
|
-
}
|
|
54
|
-
const cwd = g.packageDir ? `${repoRoot}/${g.packageDir}` : repoRoot;
|
|
55
|
-
ran += 1;
|
|
56
|
-
const res = await _acceptanceDeps.runShell(buildAcceptanceCommand(repoRoot, g), { cwd, timeoutMs });
|
|
57
|
-
chunks.push(`[${name}] exit=${res.exitCode}\n${res.stdout}\n${res.stderr}`);
|
|
58
|
-
if (res.exitCode !== 0) return { passed: false, ran, missing, output: chunks.join("\n\n") };
|
|
59
|
-
}
|
|
60
|
-
if (missing.length > 0) {
|
|
61
|
-
chunks.push(`[acceptance] no acceptance test file on disk for: ${missing.join(", ")}`);
|
|
62
|
-
}
|
|
63
|
-
if (ran === 0) chunks.push("[acceptance] no acceptance test files present — nothing to run");
|
|
64
|
-
return { passed: true, ran, missing, output: chunks.join("\n\n") };
|
|
65
|
-
}
|
|
66
|
-
|
|
67
|
-
function languageRunner(language: string): string {
|
|
68
|
-
switch (language) {
|
|
69
|
-
case "python":
|
|
70
|
-
return "uv run pytest";
|
|
71
|
-
case "go":
|
|
72
|
-
return "go test";
|
|
73
|
-
default:
|
|
74
|
-
return "bun test";
|
|
75
|
-
}
|
|
76
|
-
}
|
|
@@ -1,67 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Assembling the audit round a `commit_<phase>` node records.
|
|
3
|
-
*
|
|
4
|
-
* Split out of `nax-finish.flow.ts` for room — that file sits against the
|
|
5
|
-
* 600-line source cap — but the split earns its keep independently: the round's
|
|
6
|
-
* shape is a data decision with four conditional fields, and it is now unit
|
|
7
|
-
* testable without driving a flow node through a git mock.
|
|
8
|
-
*
|
|
9
|
-
* Pairs with `./review-round`, which records the rounds that produce no commit.
|
|
10
|
-
*/
|
|
11
|
-
import type { Finding, FindingDisposition, FinishPhase, FinishRound, FinishRoundOutcome } from "../types";
|
|
12
|
-
|
|
13
|
-
/** Phases that own a reviewer node; every other phase's round has nobody behind it. */
|
|
14
|
-
const REVIEWED_PHASES: FinishPhase[] = ["spec", "quality"];
|
|
15
|
-
|
|
16
|
-
/**
|
|
17
|
-
* What produced this round, given the successor the commit routed to.
|
|
18
|
-
*
|
|
19
|
-
* `route` no longer changes the answer, and that is the point: since #1510
|
|
20
|
-
* every committed gate fix re-enters `review_quality`, so no route can skip an
|
|
21
|
-
* owed re-review. The parameter stays because the round is still keyed on the
|
|
22
|
-
* successor conceptually, and a future route that *does* bypass a reviewer
|
|
23
|
-
* would need to be reflected here rather than silently inheriting
|
|
24
|
-
* `no-reviewer`.
|
|
25
|
-
*/
|
|
26
|
-
export function commitRoundOutcome(phase: FinishPhase, _route: string): FinishRoundOutcome {
|
|
27
|
-
if (REVIEWED_PHASES.includes(phase)) return "fixed";
|
|
28
|
-
return "no-reviewer";
|
|
29
|
-
}
|
|
30
|
-
|
|
31
|
-
export interface CommitRoundInput {
|
|
32
|
-
phase: FinishPhase;
|
|
33
|
-
attempt: number;
|
|
34
|
-
committed: boolean;
|
|
35
|
-
/** The successor this commit routed to — see `commitRoundOutcome`. */
|
|
36
|
-
route: string;
|
|
37
|
-
findings: Finding[];
|
|
38
|
-
/** Gate commands that were red this round; omitted for non-gate phases. */
|
|
39
|
-
failing?: string[];
|
|
40
|
-
/** Post-commit HEAD, when there was a commit. */
|
|
41
|
-
shaAfter?: string | null;
|
|
42
|
-
now: string;
|
|
43
|
-
/** What the fixer did with each finding it was handed (spec/quality phases). */
|
|
44
|
-
dispositions?: FindingDisposition[];
|
|
45
|
-
}
|
|
46
|
-
|
|
47
|
-
/**
|
|
48
|
-
* Build the round record for a commit checkpoint.
|
|
49
|
-
*
|
|
50
|
-
* `sha` and `failing` are omitted rather than set to null/undefined: a reader of
|
|
51
|
-
* the JSONL distinguishes "no commit" from "record lost" by the key's absence,
|
|
52
|
-
* and that only works if absence is never used to mean anything else.
|
|
53
|
-
*/
|
|
54
|
-
export function buildCommitRound(i: CommitRoundInput): FinishRound {
|
|
55
|
-
return {
|
|
56
|
-
ts: i.now,
|
|
57
|
-
phase: i.phase,
|
|
58
|
-
attempt: i.attempt,
|
|
59
|
-
committed: i.committed,
|
|
60
|
-
outcome: commitRoundOutcome(i.phase, i.route),
|
|
61
|
-
findings: i.findings,
|
|
62
|
-
route: i.route,
|
|
63
|
-
...(i.failing ? { failing: i.failing } : {}),
|
|
64
|
-
...(i.committed && i.shaAfter ? { sha: i.shaAfter } : {}),
|
|
65
|
-
...(i.dispositions && i.dispositions.length > 0 ? { dispositions: i.dispositions } : {}),
|
|
66
|
-
};
|
|
67
|
-
}
|