@nathapp/nax 0.80.1 → 0.81.0-canary.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/nax.js +44737 -42009
- package/package.json +6 -6
- package/flows/nax-finish/commit-message.ts +0 -239
- package/flows/nax-finish/errors.ts +0 -23
- package/flows/nax-finish/exec.ts +0 -135
- package/flows/nax-finish/findings-parse.ts +0 -150
- package/flows/nax-finish/flow-ctx.ts +0 -150
- package/flows/nax-finish/narrative.ts +0 -215
- package/flows/nax-finish/nax-finish.flow.ts +0 -566
- package/flows/nax-finish/pr-template-merge.ts +0 -253
- package/flows/nax-finish/pr-template.ts +0 -56
- package/flows/nax-finish/pr-title.ts +0 -140
- package/flows/nax-finish/review-prompts.ts +0 -468
- package/flows/nax-finish/steps/acceptance.ts +0 -76
- package/flows/nax-finish/steps/commit-round.ts +0 -67
- package/flows/nax-finish/steps/context.ts +0 -158
- package/flows/nax-finish/steps/escalate.ts +0 -93
- package/flows/nax-finish/steps/forge.ts +0 -93
- package/flows/nax-finish/steps/gates.ts +0 -183
- package/flows/nax-finish/steps/git.ts +0 -130
- package/flows/nax-finish/steps/index.ts +0 -13
- package/flows/nax-finish/steps/pr-body.ts +0 -462
- package/flows/nax-finish/steps/pr-narrative.ts +0 -46
- package/flows/nax-finish/steps/pr.ts +0 -116
- package/flows/nax-finish/steps/quality.ts +0 -102
- package/flows/nax-finish/steps/result.ts +0 -112
- package/flows/nax-finish/steps/review-audit.ts +0 -91
- package/flows/nax-finish/steps/review-round.ts +0 -109
- package/flows/nax-finish/types.ts +0 -260
- package/flows/nax-finish/verdict.ts +0 -210
|
@@ -1,102 +0,0 @@
|
|
|
1
|
-
import { readFile } from "node:fs/promises";
|
|
2
|
-
import { FinishError } from "../errors";
|
|
3
|
-
import { DEFAULT_GATE_TIMEOUT_MS, runShell } from "../exec";
|
|
4
|
-
import type { ShellRunFn } from "../types";
|
|
5
|
-
|
|
6
|
-
export interface QualityCommands {
|
|
7
|
-
build?: string;
|
|
8
|
-
typecheck?: string;
|
|
9
|
-
lint?: string;
|
|
10
|
-
test?: string;
|
|
11
|
-
format?: string;
|
|
12
|
-
}
|
|
13
|
-
|
|
14
|
-
export const _qualityDeps: { runShell: ShellRunFn; readText: (path: string) => Promise<string | null> } = {
|
|
15
|
-
runShell,
|
|
16
|
-
// node:fs, not Bun.file — this module runs inside acpx's Node process, where
|
|
17
|
-
// the `Bun` global does not exist (see the header of `../exec.ts`). A single
|
|
18
|
-
// read that treats ENOENT as "absent" also avoids the exists()-then-read race
|
|
19
|
-
// the Bun version had.
|
|
20
|
-
readText: async (path) => {
|
|
21
|
-
try {
|
|
22
|
-
return await readFile(path, "utf8");
|
|
23
|
-
} catch (err) {
|
|
24
|
-
if ((err as NodeJS.ErrnoException).code === "ENOENT") return null;
|
|
25
|
-
throw err;
|
|
26
|
-
}
|
|
27
|
-
},
|
|
28
|
-
};
|
|
29
|
-
|
|
30
|
-
const GATE_ORDER: (keyof QualityCommands)[] = ["build", "typecheck", "lint", "test", "format"];
|
|
31
|
-
|
|
32
|
-
export interface QualityGateOutcome {
|
|
33
|
-
passed: boolean;
|
|
34
|
-
/** Gate names that actually ran — empty means nothing was configured. */
|
|
35
|
-
ran: string[];
|
|
36
|
-
failing: string[];
|
|
37
|
-
output: string;
|
|
38
|
-
}
|
|
39
|
-
|
|
40
|
-
/**
|
|
41
|
-
* Run the repo's configured quality commands in order, each through
|
|
42
|
-
* `/bin/sh -c` (matching `src/quality/runner.ts`) so `&&`, quoting and globs
|
|
43
|
-
* survive, and each under a wall-clock cap so a hung gate can't stall the flow.
|
|
44
|
-
*
|
|
45
|
-
* `passed` is only true when at least one gate ran. A repo with no configured
|
|
46
|
-
* commands must not report a green gate — that previously let the flow open a
|
|
47
|
-
* "ready" PR having verified nothing.
|
|
48
|
-
*/
|
|
49
|
-
export async function runQualityGates(
|
|
50
|
-
repoRoot: string,
|
|
51
|
-
commands: QualityCommands,
|
|
52
|
-
opts: { timeoutMs?: number } = {},
|
|
53
|
-
): Promise<QualityGateOutcome> {
|
|
54
|
-
const failing: string[] = [];
|
|
55
|
-
const ran: string[] = [];
|
|
56
|
-
const chunks: string[] = [];
|
|
57
|
-
const timeoutMs = opts.timeoutMs ?? DEFAULT_GATE_TIMEOUT_MS;
|
|
58
|
-
for (const gate of GATE_ORDER) {
|
|
59
|
-
const command = commands[gate];
|
|
60
|
-
if (!command) continue;
|
|
61
|
-
ran.push(gate);
|
|
62
|
-
const res = await _qualityDeps.runShell(command, { cwd: repoRoot, timeoutMs });
|
|
63
|
-
chunks.push(`[${gate}] exit=${res.exitCode}\n${res.stdout}\n${res.stderr}`);
|
|
64
|
-
if (res.exitCode !== 0) failing.push(gate);
|
|
65
|
-
}
|
|
66
|
-
if (ran.length === 0) {
|
|
67
|
-
return {
|
|
68
|
-
passed: false,
|
|
69
|
-
ran,
|
|
70
|
-
failing: [],
|
|
71
|
-
output: "[quality] no quality.commands configured in .nax/config.json — nothing was verified",
|
|
72
|
-
};
|
|
73
|
-
}
|
|
74
|
-
return { passed: failing.length === 0, ran, failing, output: chunks.join("\n\n") };
|
|
75
|
-
}
|
|
76
|
-
|
|
77
|
-
/**
|
|
78
|
-
* Read `quality.commands` from the repo-root `.nax/config.json` only.
|
|
79
|
-
*
|
|
80
|
-
* Deliberately root-scoped: per-package `.nax/mono/<pkg>/config.json` overrides
|
|
81
|
-
* are unreliable as a repo-root gate (a package's own `test` command does not
|
|
82
|
-
* verify the repo), and this gate runs once at the root by design.
|
|
83
|
-
*/
|
|
84
|
-
export async function loadQualityCommands(workdir: string): Promise<QualityCommands> {
|
|
85
|
-
const text = await _qualityDeps.readText(`${workdir}/.nax/config.json`);
|
|
86
|
-
if (!text) return {};
|
|
87
|
-
let cfg: { quality?: { commands?: QualityCommands } };
|
|
88
|
-
try {
|
|
89
|
-
cfg = JSON.parse(text);
|
|
90
|
-
} catch (cause) {
|
|
91
|
-
throw new FinishError(
|
|
92
|
-
"Failed to parse .nax/config.json while loading quality commands",
|
|
93
|
-
"FINISH_CONFIG_UNPARSEABLE",
|
|
94
|
-
{
|
|
95
|
-
stage: "finish-quality",
|
|
96
|
-
path: `${workdir}/.nax/config.json`,
|
|
97
|
-
cause,
|
|
98
|
-
},
|
|
99
|
-
);
|
|
100
|
-
}
|
|
101
|
-
return cfg.quality?.commands ?? {};
|
|
102
|
-
}
|
|
@@ -1,112 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Finish-audit artifacts.
|
|
3
|
-
*
|
|
4
|
-
* These live under nax's global per-project output directory —
|
|
5
|
-
* `~/.nax/<project>/finish-audit/<feature>/` — alongside `prompt-audit/` and
|
|
6
|
-
* `review-audit/`, not in the user's repo. Two reasons the repo was the wrong
|
|
7
|
-
* home: the artifact describes a *run*, not the source tree, so committing it
|
|
8
|
-
* and gitignoring it are both wrong answers; and a per-feature, per-run path
|
|
9
|
-
* makes the history queryable across runs, which a single overwritten
|
|
10
|
-
* `.nax/nax-finish-result.json` never was.
|
|
11
|
-
*
|
|
12
|
-
* The plugin supplies `auditDir` because it owns nax's path SSOT
|
|
13
|
-
* (`src/runtime/paths.ts`), which this module may not import — `flows/` is
|
|
14
|
-
* loaded by acpx, outside nax's own process. Absent, we fall back to a
|
|
15
|
-
* repo-local directory so a hand-run `acpx flow run` still records something.
|
|
16
|
-
*
|
|
17
|
-
* Two files per run:
|
|
18
|
-
* - `<runId>.jsonl` — one line per fix round, appended as it happens
|
|
19
|
-
* - `<runId>.result.json` — the terminal result the plugin reads back
|
|
20
|
-
*/
|
|
21
|
-
import { mkdir, readFile, writeFile } from "node:fs/promises";
|
|
22
|
-
import { dirname, join } from "node:path";
|
|
23
|
-
import type { FinishInput, FinishResult, FinishRound } from "../types";
|
|
24
|
-
|
|
25
|
-
/** Used when the plugin supplied no run id (e.g. a hand-run `acpx flow run`). */
|
|
26
|
-
const FALLBACK_RUN_ID = "run";
|
|
27
|
-
|
|
28
|
-
type AuditTarget = Pick<FinishInput, "auditDir" | "workdir" | "feature" | "runId">;
|
|
29
|
-
|
|
30
|
-
export function resolveAuditDir(input: AuditTarget): string {
|
|
31
|
-
return input.auditDir ?? join(input.workdir, ".nax", "finish-audit", input.feature);
|
|
32
|
-
}
|
|
33
|
-
|
|
34
|
-
export function resultPath(input: AuditTarget): string {
|
|
35
|
-
return join(resolveAuditDir(input), `${input.runId || FALLBACK_RUN_ID}.result.json`);
|
|
36
|
-
}
|
|
37
|
-
|
|
38
|
-
export function roundsPath(input: AuditTarget): string {
|
|
39
|
-
return join(resolveAuditDir(input), `${input.runId || FALLBACK_RUN_ID}.jsonl`);
|
|
40
|
-
}
|
|
41
|
-
|
|
42
|
-
export const _resultDeps: {
|
|
43
|
-
writeText: (p: string, s: string) => Promise<void>;
|
|
44
|
-
appendText: (p: string, s: string) => Promise<void>;
|
|
45
|
-
readText: (p: string) => Promise<string | null>;
|
|
46
|
-
} = {
|
|
47
|
-
// node:fs, not Bun.write — this module runs inside acpx's Node process, where
|
|
48
|
-
// the `Bun` global does not exist (see the header of `../exec.ts`). The mkdir
|
|
49
|
-
// is not redundant: Bun.write creates missing parent directories implicitly,
|
|
50
|
-
// writeFile does not, and the audit directory now lives under `~/.nax/`,
|
|
51
|
-
// where for a project's first run nothing on the path exists yet.
|
|
52
|
-
writeText: async (p, s) => {
|
|
53
|
-
await mkdir(dirname(p), { recursive: true });
|
|
54
|
-
await writeFile(p, s, "utf8");
|
|
55
|
-
},
|
|
56
|
-
appendText: async (p, s) => {
|
|
57
|
-
await mkdir(dirname(p), { recursive: true });
|
|
58
|
-
await writeFile(p, s, { encoding: "utf8", flag: "a" });
|
|
59
|
-
},
|
|
60
|
-
readText: async (p) => {
|
|
61
|
-
try {
|
|
62
|
-
return await readFile(p, "utf8");
|
|
63
|
-
} catch {
|
|
64
|
-
return null;
|
|
65
|
-
}
|
|
66
|
-
},
|
|
67
|
-
};
|
|
68
|
-
|
|
69
|
-
/**
|
|
70
|
-
* Append one fix round to the run's audit trail.
|
|
71
|
-
*
|
|
72
|
-
* Best-effort: an unwritable audit directory must not take the flow down
|
|
73
|
-
* mid-loop. The round is a record of work already done — losing the record is
|
|
74
|
-
* bad, losing the run that did the work is worse.
|
|
75
|
-
*/
|
|
76
|
-
export async function appendRound(input: AuditTarget, round: FinishRound): Promise<void> {
|
|
77
|
-
try {
|
|
78
|
-
await _resultDeps.appendText(roundsPath(input), `${JSON.stringify(round)}\n`);
|
|
79
|
-
} catch {
|
|
80
|
-
// Intentionally swallowed — see the doc comment above.
|
|
81
|
-
}
|
|
82
|
-
}
|
|
83
|
-
|
|
84
|
-
/** Read back every round recorded for this run, so a terminal result can embed them. */
|
|
85
|
-
export async function readRounds(input: AuditTarget): Promise<FinishRound[]> {
|
|
86
|
-
const raw = await _resultDeps.readText(roundsPath(input));
|
|
87
|
-
if (!raw) return [];
|
|
88
|
-
const rounds: FinishRound[] = [];
|
|
89
|
-
for (const line of raw.split("\n")) {
|
|
90
|
-
if (!line.trim()) continue;
|
|
91
|
-
try {
|
|
92
|
-
rounds.push(JSON.parse(line) as FinishRound);
|
|
93
|
-
} catch {
|
|
94
|
-
// A torn final line (killed mid-write) must not lose the rounds before it.
|
|
95
|
-
}
|
|
96
|
-
}
|
|
97
|
-
return rounds;
|
|
98
|
-
}
|
|
99
|
-
|
|
100
|
-
/**
|
|
101
|
-
* Write the terminal result, embedding every round this run recorded.
|
|
102
|
-
*
|
|
103
|
-
* Rounds are attached on *every* status, not just `escalated`: a finish that
|
|
104
|
-
* succeeded after four rounds is precisely the case worth auditing — it says
|
|
105
|
-
* the run's own review gates missed four defects — and it was the one case
|
|
106
|
-
* that previously recorded nothing at all.
|
|
107
|
-
*/
|
|
108
|
-
export async function writeResult(input: AuditTarget, result: FinishResult): Promise<void> {
|
|
109
|
-
const rounds = await readRounds(input);
|
|
110
|
-
const withRounds: FinishResult = rounds.length > 0 ? { ...result, rounds } : result;
|
|
111
|
-
await _resultDeps.writeText(resultPath(input), `${JSON.stringify(withRounds, null, 2)}\n`);
|
|
112
|
-
}
|
|
@@ -1,91 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Checking that a review discharged its obligations before its verdict counts.
|
|
3
|
-
*
|
|
4
|
-
* `WORKER_PROTOCOL` has always told the reviewer to enumerate the external
|
|
5
|
-
* touchpoints and open their definitions, and both dimension references have
|
|
6
|
-
* always required a per-item enumeration. Neither was checkable: `routeReview`
|
|
7
|
-
* saw only a route and a finding list, so a reviewer that read the diff and
|
|
8
|
-
* nothing else was indistinguishable from one that did the work — and on the run
|
|
9
|
-
* behind #1614 that is exactly what happened, at 86 seconds for 3,716 changed
|
|
10
|
-
* lines.
|
|
11
|
-
*
|
|
12
|
-
* The disk check is what makes this a gate rather than a ritual. It proves the
|
|
13
|
-
* paths are real, not that they were read: a reviewer can still list files it
|
|
14
|
-
* only globbed. That raises the cost of faking the list without eliminating it,
|
|
15
|
-
* which is the honest ceiling for a check that costs one `stat` per line.
|
|
16
|
-
*
|
|
17
|
-
* A verdict from the legacy JSON parsing tier carries no `saw*` fields (they are
|
|
18
|
-
* optional), so it always reports both gaps and is sent back for one re-review
|
|
19
|
-
* under the new prompt. That is intentional, not a bug: the safe direction is
|
|
20
|
-
* an extra review, never a false approval, and the retry self-corrects because
|
|
21
|
-
* the new prompt contract produces a verdict the gate can actually check.
|
|
22
|
-
*
|
|
23
|
-
* `node:fs` — not `Bun.file` — because `flows/` runs inside acpx's Node process.
|
|
24
|
-
*
|
|
25
|
-
* Both `touchpoint.path` and a disposition's `evidence` come from the
|
|
26
|
-
* reviewer/fixer's reply text — untrusted the same way any parsed LLM output
|
|
27
|
-
* is. `exists()` confines its resolved path under `workdir` before stat-ing
|
|
28
|
-
* it, so a `../`-laden path can never be used to probe existence outside the
|
|
29
|
-
* repo; a path that escapes reads as "does not exist," which is the correct
|
|
30
|
-
* verdict anyway since a legitimate touchpoint is always inside it.
|
|
31
|
-
*/
|
|
32
|
-
import { stat } from "node:fs/promises";
|
|
33
|
-
import * as path from "node:path";
|
|
34
|
-
import type { FindingDisposition, ReviewVerdict } from "../types";
|
|
35
|
-
|
|
36
|
-
/** Paths stat-ed per review. A reviewer listing more than this is not the failure mode. */
|
|
37
|
-
const MAX_CHECKED = 20;
|
|
38
|
-
|
|
39
|
-
async function exists(workdir: string, rel: string): Promise<boolean> {
|
|
40
|
-
const root = path.resolve(workdir);
|
|
41
|
-
const resolved = path.resolve(root, rel);
|
|
42
|
-
if (resolved !== root && !resolved.startsWith(root + path.sep)) return false;
|
|
43
|
-
try {
|
|
44
|
-
await stat(resolved);
|
|
45
|
-
return true;
|
|
46
|
-
} catch {
|
|
47
|
-
return false;
|
|
48
|
-
}
|
|
49
|
-
}
|
|
50
|
-
|
|
51
|
-
/**
|
|
52
|
-
* What this review failed to do. Empty means it may be routed on.
|
|
53
|
-
*
|
|
54
|
-
* Only ever called for a verdict that already parsed; an unreadable reply is the
|
|
55
|
-
* `reprompt` path's business and is handled before this runs.
|
|
56
|
-
*/
|
|
57
|
-
export async function auditGaps(verdict: ReviewVerdict, workdir: string): Promise<string[]> {
|
|
58
|
-
const gaps: string[] = [];
|
|
59
|
-
const touchpoints = verdict.touchpoints ?? [];
|
|
60
|
-
if (!verdict.sawTouchpointsSection || touchpoints.length === 0) {
|
|
61
|
-
gaps.push("no `## TOUCHPOINTS` section: list every external definition you opened, or `- none — <justification>`");
|
|
62
|
-
} else if (!touchpoints.some((t) => t.path === "none")) {
|
|
63
|
-
const checked = touchpoints.slice(0, MAX_CHECKED);
|
|
64
|
-
const found = await Promise.all(checked.map((t) => exists(workdir, t.path)));
|
|
65
|
-
if (!found.some(Boolean)) {
|
|
66
|
-
gaps.push(
|
|
67
|
-
`touchpoint path does not exist in the repo (checked: ${checked
|
|
68
|
-
.map((t) => t.path)
|
|
69
|
-
.join(", ")}) — list files you actually opened`,
|
|
70
|
-
);
|
|
71
|
-
}
|
|
72
|
-
}
|
|
73
|
-
if (!verdict.sawWalkSection || (verdict.walk ?? []).length === 0) {
|
|
74
|
-
gaps.push("no `## WALK` section: the per-AC (spec) or per-function (quality) enumeration is required");
|
|
75
|
-
}
|
|
76
|
-
return gaps;
|
|
77
|
-
}
|
|
78
|
-
|
|
79
|
-
/** Mark any rejection whose cited `file:line` does not resolve in the repo. */
|
|
80
|
-
export async function validateDispositions(
|
|
81
|
-
workdir: string,
|
|
82
|
-
dispositions: FindingDisposition[],
|
|
83
|
-
): Promise<FindingDisposition[]> {
|
|
84
|
-
return Promise.all(
|
|
85
|
-
dispositions.map(async (d) => {
|
|
86
|
-
if (d.disposition !== "rejected" || !d.evidence) return d;
|
|
87
|
-
const file = d.evidence.split(":")[0];
|
|
88
|
-
return (await exists(workdir, file)) ? d : { ...d, evidenceMissing: true };
|
|
89
|
-
}),
|
|
90
|
-
);
|
|
91
|
-
}
|
|
@@ -1,109 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Recording the review rounds that produce no commit.
|
|
3
|
-
*
|
|
4
|
-
* `commit_<phase>` is the audit seam for rounds that *fix* something — it is the
|
|
5
|
-
* only point where a round's findings and its resulting commit are both known.
|
|
6
|
-
* But that made a commit the sole evidence a reviewer ever ran: a review that
|
|
7
|
-
* passed produced no `fix_*`, therefore no `commit_*`, therefore no round, and
|
|
8
|
-
* "this phase passed" became indistinguishable from "this phase never ran"
|
|
9
|
-
* (#1507). Worse, it made the owed re-review in #1506 unprovable after the fact.
|
|
10
|
-
*
|
|
11
|
-
* So the two seams split by what they know:
|
|
12
|
-
* - `commit_<phase>` records rounds that changed the tree (`outcome: "fixed"`).
|
|
13
|
-
* - here records rounds that did not (`passed` / `unparseable` / `escalated`).
|
|
14
|
-
*
|
|
15
|
-
* Wrapping `routeReview` rather than living inside it keeps that function pure
|
|
16
|
-
* and synchronous — it is the flow's routing SSOT and is exercised by a large
|
|
17
|
-
* table of unit tests that would all have to become async otherwise.
|
|
18
|
-
*/
|
|
19
|
-
import { inputOf } from "../flow-ctx";
|
|
20
|
-
import type { OutputsCtx, StepsCtx } from "../flow-ctx";
|
|
21
|
-
import type { Finding, FinishRoundOutcome, ReviewVerdict } from "../types";
|
|
22
|
-
import { MAX_INCOMPLETE_ATTEMPTS, routeReview } from "../verdict";
|
|
23
|
-
import { appendRound } from "./result";
|
|
24
|
-
import { auditGaps } from "./review-audit";
|
|
25
|
-
|
|
26
|
-
/** Route → what to call the round. `fix` is absent by construction — see below. */
|
|
27
|
-
const OUTCOME_BY_ROUTE: Record<string, FinishRoundOutcome> = {
|
|
28
|
-
clean: "passed",
|
|
29
|
-
reprompt: "unparseable",
|
|
30
|
-
escalate: "escalated",
|
|
31
|
-
incomplete: "incomplete",
|
|
32
|
-
};
|
|
33
|
-
|
|
34
|
-
/**
|
|
35
|
-
* The Nth time this phase's *review* node has run.
|
|
36
|
-
*
|
|
37
|
-
* Deliberately not `fixAttemptCount`: that counts `fix_<phase>` steps, which is
|
|
38
|
-
* the right number for a round that fixed something and the wrong one here — a
|
|
39
|
-
* review that passes on the first look runs zero fix nodes, and every clean
|
|
40
|
-
* round would be numbered 0. Self-inclusive, for the same reason `repromptCount`
|
|
41
|
-
* is: acpx records the `review_<phase>` step before `route_<phase>` executes.
|
|
42
|
-
*/
|
|
43
|
-
function reviewAttemptCount(ctx: StepsCtx, phase: "spec" | "quality"): number {
|
|
44
|
-
return (ctx.state.steps ?? []).filter((s) => s.nodeId === `review_${phase}`).length;
|
|
45
|
-
}
|
|
46
|
-
|
|
47
|
-
/**
|
|
48
|
-
* How many previous rounds of this phase were sent back as incomplete.
|
|
49
|
-
*
|
|
50
|
-
* NOT self-inclusive, unlike `repromptCount` — that one counts `review_<phase>`
|
|
51
|
-
* steps, which acpx has already recorded by the time `route_<phase>` runs, while
|
|
52
|
-
* this counts `route_<phase>` steps and we are *inside* the current one. So the
|
|
53
|
-
* comparison below is `<`, where `routeReview`'s reprompt comparison is `<=`.
|
|
54
|
-
*/
|
|
55
|
-
function incompleteCount(ctx: StepsCtx, phase: "spec" | "quality"): number {
|
|
56
|
-
return (ctx.state.steps ?? []).filter(
|
|
57
|
-
(s) => s.nodeId === `route_${phase}` && (s.output as { route?: string } | undefined)?.route === "incomplete",
|
|
58
|
-
).length;
|
|
59
|
-
}
|
|
60
|
-
|
|
61
|
-
/**
|
|
62
|
-
* Route this phase's review verdict, and record the round when it produced no
|
|
63
|
-
* commit.
|
|
64
|
-
*
|
|
65
|
-
* `fix` is the one route that records nothing here: it leads to `fix_<phase>` →
|
|
66
|
-
* `commit_<phase>`, which appends the round with the commit attached. Recording
|
|
67
|
-
* at both seams would double-count every fixed round in the PR body.
|
|
68
|
-
*
|
|
69
|
-
* Best-effort, exactly like `appendRound` itself: the route is returned whether
|
|
70
|
-
* or not the write lands. Losing the record is bad; failing the run that has
|
|
71
|
-
* already done the work is worse.
|
|
72
|
-
*/
|
|
73
|
-
export async function routeReviewAndRecord(
|
|
74
|
-
ctx: { input: unknown } & OutputsCtx & StepsCtx,
|
|
75
|
-
phase: "spec" | "quality",
|
|
76
|
-
): Promise<{ route: string; escalationReason?: string; findings: Finding[]; gaps?: string[] }> {
|
|
77
|
-
const routed = routeReview(ctx, phase);
|
|
78
|
-
const input = inputOf(ctx);
|
|
79
|
-
// The gate runs only on a verdict the flow would otherwise act on. `reprompt`
|
|
80
|
-
// and `escalate` already end the round, and re-checking a verdict with no
|
|
81
|
-
// content would report the same two gaps as a second failure mode.
|
|
82
|
-
let result: { route: string; escalationReason?: string; findings: Finding[]; gaps?: string[] } = routed;
|
|
83
|
-
if (routed.route === "clean" || routed.route === "fix") {
|
|
84
|
-
const verdict = (ctx.outputs as Record<string, ReviewVerdict | undefined>)[`review_${phase}`];
|
|
85
|
-
const gaps = verdict ? await auditGaps(verdict, input.workdir) : [];
|
|
86
|
-
if (gaps.length > 0) {
|
|
87
|
-
result =
|
|
88
|
-
incompleteCount(ctx, phase) < MAX_INCOMPLETE_ATTEMPTS
|
|
89
|
-
? { ...routed, route: "incomplete", gaps }
|
|
90
|
-
: {
|
|
91
|
-
...routed,
|
|
92
|
-
route: "escalate",
|
|
93
|
-
escalationReason: `${phase} review never discharged its reading obligations: ${gaps.join("; ")}`,
|
|
94
|
-
};
|
|
95
|
-
}
|
|
96
|
-
}
|
|
97
|
-
const outcome = OUTCOME_BY_ROUTE[result.route];
|
|
98
|
-
if (outcome) {
|
|
99
|
-
await appendRound(input, {
|
|
100
|
-
ts: new Date().toISOString(),
|
|
101
|
-
phase,
|
|
102
|
-
attempt: reviewAttemptCount(ctx, phase),
|
|
103
|
-
committed: false,
|
|
104
|
-
outcome,
|
|
105
|
-
findings: result.findings,
|
|
106
|
-
});
|
|
107
|
-
}
|
|
108
|
-
return result;
|
|
109
|
-
}
|
|
@@ -1,260 +0,0 @@
|
|
|
1
|
-
import type { TemplateMode } from "./pr-template-merge";
|
|
2
|
-
|
|
3
|
-
/** One acceptance-test group as reported by `nax features resolve --json`. */
|
|
4
|
-
export interface AcceptanceGroup {
|
|
5
|
-
packageDir: string;
|
|
6
|
-
testPath: string;
|
|
7
|
-
exists: boolean;
|
|
8
|
-
command?: string;
|
|
9
|
-
language: string;
|
|
10
|
-
}
|
|
11
|
-
|
|
12
|
-
export type Severity = "CRITICAL" | "HIGH" | "MEDIUM" | "LOW";
|
|
13
|
-
export interface Finding {
|
|
14
|
-
severity: Severity;
|
|
15
|
-
title: string;
|
|
16
|
-
problem: string;
|
|
17
|
-
fix: string;
|
|
18
|
-
/**
|
|
19
|
-
* Set when the reviewer marked this finding as needing a human — a spec
|
|
20
|
-
* conflict or a design call with no safe mechanical fix. This replaces the
|
|
21
|
-
* whole-reply `escalate` route the reviewer used to choose: escalation is a
|
|
22
|
-
* property of one finding, not of the phase, so reporting a design concern no
|
|
23
|
-
* longer halts the pipeline by itself.
|
|
24
|
-
*/
|
|
25
|
-
judgment?: boolean;
|
|
26
|
-
/** Why this finding needs a human; the escalation reason when it escalates. */
|
|
27
|
-
judgmentReason?: string;
|
|
28
|
-
}
|
|
29
|
-
|
|
30
|
-
/** One external definition the reviewer says it opened before judging. */
|
|
31
|
-
export interface Touchpoint {
|
|
32
|
-
/** Repo-relative path, or the literal `none` sentinel. */
|
|
33
|
-
path: string;
|
|
34
|
-
/** Symbol or line after the final `:`, when the reviewer gave one. */
|
|
35
|
-
symbol?: string;
|
|
36
|
-
/** The reviewer's stated reason for opening it (or for there being none). */
|
|
37
|
-
note: string;
|
|
38
|
-
}
|
|
39
|
-
|
|
40
|
-
/** A reviewer reply, parsed. Sections are reported separately from their content
|
|
41
|
-
* so an *absent* section is distinguishable from an *empty* one — only the first
|
|
42
|
-
* is a reviewer that skipped the obligation. */
|
|
43
|
-
export interface ReviewReport {
|
|
44
|
-
findings: Finding[];
|
|
45
|
-
touchpoints: Touchpoint[];
|
|
46
|
-
walk: string[];
|
|
47
|
-
sawNoFindings: boolean;
|
|
48
|
-
sawTouchpointsSection: boolean;
|
|
49
|
-
sawWalkSection: boolean;
|
|
50
|
-
}
|
|
51
|
-
|
|
52
|
-
/** What the fix node did with one finding it was handed. */
|
|
53
|
-
export interface FindingDisposition {
|
|
54
|
-
/** 1-based index into the findings list the fix prompt numbered. */
|
|
55
|
-
index: number;
|
|
56
|
-
disposition: "fixed" | "rejected";
|
|
57
|
-
/** `file:line` pinning the current behaviour; required for a rejection. */
|
|
58
|
-
evidence?: string;
|
|
59
|
-
/** Set by `commit_<phase>` when the cited evidence path does not exist. */
|
|
60
|
-
evidenceMissing?: boolean;
|
|
61
|
-
}
|
|
62
|
-
export interface ReviewVerdict {
|
|
63
|
-
/**
|
|
64
|
-
* Neither `clean` nor `reprompt` is a model-produced route.
|
|
65
|
-
*
|
|
66
|
-
* `clean` — `parse` rewrites `proceed` with zero findings, so the graph can
|
|
67
|
-
* skip the fix node instead of prompting an agent to "apply fixes" for nothing.
|
|
68
|
-
*
|
|
69
|
-
* `reprompt` — `parse` could not read JSON out of the reply at all. Returning
|
|
70
|
-
* this rather than throwing is deliberate: a throw fails the acp node and kills
|
|
71
|
-
* the whole flow with no result file, bypassing the `escalate` sink that exists
|
|
72
|
-
* to report exactly this kind of dead end.
|
|
73
|
-
*/
|
|
74
|
-
route: "proceed" | "escalate" | "clean" | "reprompt";
|
|
75
|
-
findings: Finding[];
|
|
76
|
-
escalationReason?: string;
|
|
77
|
-
/** Bounded tail of an unparseable reply; set only when `route` is `reprompt`. */
|
|
78
|
-
raw?: string;
|
|
79
|
-
/** Touchpoints the reviewer listed; read by the audit gate in `routeReviewAndRecord`. */
|
|
80
|
-
touchpoints?: Touchpoint[];
|
|
81
|
-
/** The per-AC or per-function walk lines the reviewer emitted. */
|
|
82
|
-
walk?: string[];
|
|
83
|
-
/** Whether the section was present at all — absent and empty are different failures. */
|
|
84
|
-
sawTouchpointsSection?: boolean;
|
|
85
|
-
sawWalkSection?: boolean;
|
|
86
|
-
/** Set on a `fix_<phase>` output: what the fixer did with each finding it was handed. */
|
|
87
|
-
dispositions?: FindingDisposition[];
|
|
88
|
-
}
|
|
89
|
-
/** Wall-clock budgets, forwarded from `finish.autoFlow.timeouts` by the plugin. */
|
|
90
|
-
export interface FinishTimeouts {
|
|
91
|
-
acceptanceMs?: number;
|
|
92
|
-
gateMs?: number;
|
|
93
|
-
}
|
|
94
|
-
/** The four fix-and-reverify loops, in graph order. */
|
|
95
|
-
export type FinishPhase = "acceptance" | "spec" | "quality" | "gate";
|
|
96
|
-
|
|
97
|
-
/**
|
|
98
|
-
* One completed fix round, appended to the audit trail as it happens.
|
|
99
|
-
*
|
|
100
|
-
* Rounds are appended at `commit_<phase>` as they happen rather than
|
|
101
|
-
* reconstructed by a terminal node from `ctx.state.steps` (which does retain
|
|
102
|
-
* every step's output). Appending live is what makes the trail survive a flow
|
|
103
|
-
* that is killed or times out: no terminal node runs on those paths, and a
|
|
104
|
-
* finish that died mid-loop is exactly when the record of what it already
|
|
105
|
-
* changed on the branch matters most.
|
|
106
|
-
*/
|
|
107
|
-
/**
|
|
108
|
-
* What produced a round — the difference between "a reviewer read this and
|
|
109
|
-
* approved it" and "nothing read this".
|
|
110
|
-
*
|
|
111
|
-
* Rounds used to be appended only where a fix produced a commit, so a review
|
|
112
|
-
* that passed left no record at all and was indistinguishable from a review
|
|
113
|
-
* that never ran (#1507). Every phase that executes now records a round, and
|
|
114
|
-
* this field says which of the five things happened.
|
|
115
|
-
*
|
|
116
|
-
* Optional because rounds recorded by earlier versions have no `outcome`, and
|
|
117
|
-
* the PR body still has to render those without claiming more than it knows.
|
|
118
|
-
*/
|
|
119
|
-
export type FinishRoundOutcome =
|
|
120
|
-
/** A reviewer reported findings and this phase's fix node ran. */
|
|
121
|
-
| "fixed"
|
|
122
|
-
/** A reviewer ran and reported nothing. The only value that means "approved". */
|
|
123
|
-
| "passed"
|
|
124
|
-
/** The reviewer replied, but no verdict could be read out of it. */
|
|
125
|
-
| "unparseable"
|
|
126
|
-
/** Handed off to a human — an explicit escalate, a cap, or a node that emitted nothing. */
|
|
127
|
-
| "escalated"
|
|
128
|
-
/**
|
|
129
|
-
* This phase has no reviewer at all (`gate`, `acceptance`). Distinct from
|
|
130
|
-
* `passed`: an empty finding list here means "nobody looked", and rendering it
|
|
131
|
-
* as "no findings" manufactures evidence of a review that does not exist.
|
|
132
|
-
*/
|
|
133
|
-
| "no-reviewer"
|
|
134
|
-
/**
|
|
135
|
-
* A re-review was owed and deliberately skipped.
|
|
136
|
-
*
|
|
137
|
-
* **No longer emitted.** It described the `gate` → `tests-only` route, which
|
|
138
|
-
* skipped `review_quality` as a cost tradeoff; #1510 closed that hole, so
|
|
139
|
-
* every committed gate fix is now re-reviewed and nothing writes this.
|
|
140
|
-
*
|
|
141
|
-
* Retained because the audit trail is read, not just written: a project that
|
|
142
|
-
* ran an earlier nax can hold rounds carrying this outcome, and dropping it
|
|
143
|
-
* from the union would make those unrenderable. Do not reuse the name for a
|
|
144
|
-
* new meaning — a reader hitting it in an old artifact must still be told
|
|
145
|
-
* what it meant when it was written.
|
|
146
|
-
*/
|
|
147
|
-
| "review-skipped"
|
|
148
|
-
/** The reviewer replied with findings but skipped a required audit section, so
|
|
149
|
-
* the verdict was not acted on. Distinct from `unparseable`: there was a
|
|
150
|
-
* readable verdict, it just had no evidence behind it. */
|
|
151
|
-
| "incomplete";
|
|
152
|
-
|
|
153
|
-
export interface FinishRound {
|
|
154
|
-
ts: string;
|
|
155
|
-
phase: FinishPhase;
|
|
156
|
-
/** 1-based; the Nth time this phase's fix node has run. */
|
|
157
|
-
attempt: number;
|
|
158
|
-
/** True when the fix produced a commit; false when it changed nothing. */
|
|
159
|
-
committed: boolean;
|
|
160
|
-
/** What produced this round; absent on rounds written before it existed. */
|
|
161
|
-
outcome?: FinishRoundOutcome;
|
|
162
|
-
/** Reviewer findings this round set out to fix (spec/quality phases). */
|
|
163
|
-
findings: Finding[];
|
|
164
|
-
/** Gate commands that were red this round (gate phase). */
|
|
165
|
-
failing?: string[];
|
|
166
|
-
/**
|
|
167
|
-
* The successor this round's commit routed to — `changed` / `tests-only` /
|
|
168
|
-
* `unchanged` for `gate`, `changed` / `unchanged` elsewhere.
|
|
169
|
-
*
|
|
170
|
-
* Recorded because `outcome` stopped carrying it. Until #1510 a tests-only
|
|
171
|
-
* gate fix was the only round writing `review-skipped`, so the outcome
|
|
172
|
-
* doubled as the classification; now every committed gate fix is reviewed
|
|
173
|
-
* and writes `no-reviewer`, which would leave "what did this fix touch?"
|
|
174
|
-
* unanswerable from the trail. That question is the input to deciding
|
|
175
|
-
* whether the re-review ever needs a cheaper, test-scoped form, so it has to
|
|
176
|
-
* survive the round it was computed in.
|
|
177
|
-
*/
|
|
178
|
-
route?: string;
|
|
179
|
-
/**
|
|
180
|
-
* `HEAD` SHA after this round's commit (set only when `committed`); absent
|
|
181
|
-
* on no-op rounds so a reader can distinguish "no commit" from "record lost".
|
|
182
|
-
* Lets "Fixed in `<sha>`" be reconstructed from the audit trail alone, rather
|
|
183
|
-
* than by matching round timestamps against `git log`.
|
|
184
|
-
*/
|
|
185
|
-
sha?: string;
|
|
186
|
-
/** What the fixer did with each finding it was handed (spec/quality phases). */
|
|
187
|
-
dispositions?: FindingDisposition[];
|
|
188
|
-
}
|
|
189
|
-
|
|
190
|
-
export interface FinishInput {
|
|
191
|
-
feature: string;
|
|
192
|
-
workdir: string;
|
|
193
|
-
branch: string;
|
|
194
|
-
prdPath: string;
|
|
195
|
-
/**
|
|
196
|
-
* Directory for this feature's finish-audit artifacts, e.g.
|
|
197
|
-
* `~/.nax/<project>/finish-audit/<feature>`. Supplied by the plugin, which
|
|
198
|
-
* owns nax's path SSOT (`src/runtime/paths.ts`) that this module may not
|
|
199
|
-
* import. Absent → the flow falls back to a repo-local directory.
|
|
200
|
-
*/
|
|
201
|
-
auditDir?: string;
|
|
202
|
-
/** Run id, used to name this run's audit files. Absent → "run". */
|
|
203
|
-
runId?: string;
|
|
204
|
-
/**
|
|
205
|
-
* True only when Telegram escalation is both enabled *and* credentialed, as
|
|
206
|
-
* determined by the plugin. When true the flow skips the PR/MR comment
|
|
207
|
-
* fallback (and does not open a draft to hold one) — the plugin sends the
|
|
208
|
-
* Telegram message from the result file instead.
|
|
209
|
-
*/
|
|
210
|
-
escalateTelegram: boolean;
|
|
211
|
-
timeouts?: FinishTimeouts;
|
|
212
|
-
/** PR/MR body composition, forwarded from `finish.autoFlow.prBody`. */
|
|
213
|
-
prBody?: FinishPrBodySettings;
|
|
214
|
-
}
|
|
215
|
-
|
|
216
|
-
/**
|
|
217
|
-
* How the repo's own PR/MR template is honoured when composing the body.
|
|
218
|
-
* Absent (and absent fields) mean the defaults in `pr-template-merge.ts`.
|
|
219
|
-
*/
|
|
220
|
-
export interface FinishPrBodySettings {
|
|
221
|
-
/** `merge` (default) · `strict` (keep unfillable headings, empty) · `ignore`. */
|
|
222
|
-
template?: TemplateMode;
|
|
223
|
-
/** Normalised template heading → body-section key, layered over the defaults. */
|
|
224
|
-
sectionMap?: Record<string, string>;
|
|
225
|
-
}
|
|
226
|
-
export interface FinishResult {
|
|
227
|
-
feature: string;
|
|
228
|
-
status: "opened" | "promoted" | "already-ready" | "escalated" | "nothing-to-finish";
|
|
229
|
-
url?: string;
|
|
230
|
-
escalationReason?: string;
|
|
231
|
-
/**
|
|
232
|
-
* The findings behind an escalation. Persisted because the reason alone is a
|
|
233
|
-
* bare count ("3 finding(s) after 3 fix attempts"), and on the Telegram
|
|
234
|
-
* channel the composed PR comment — the only other thing carrying them — is
|
|
235
|
-
* never posted. Without this the findings survived only in acpx's run bundle.
|
|
236
|
-
*/
|
|
237
|
-
findings?: Finding[];
|
|
238
|
-
/**
|
|
239
|
-
* Set when the escalation could not be delivered to its channel (forge
|
|
240
|
-
* comment failed, remote unrecognised). The result file is written before
|
|
241
|
-
* delivery is attempted, so an undelivered escalation is still reported
|
|
242
|
-
* rather than lost.
|
|
243
|
-
*/
|
|
244
|
-
deliveryError?: string;
|
|
245
|
-
/**
|
|
246
|
-
* Every fix round the flow ran, on *all* terminal statuses — not just
|
|
247
|
-
* escalations. A successful finish that took four rounds to get there is the
|
|
248
|
-
* case worth auditing (it says the run's own review gates missed four
|
|
249
|
-
* defects), and it was previously the one case that recorded nothing.
|
|
250
|
-
*/
|
|
251
|
-
rounds?: FinishRound[];
|
|
252
|
-
}
|
|
253
|
-
export interface RunResult {
|
|
254
|
-
exitCode: number;
|
|
255
|
-
stdout: string;
|
|
256
|
-
stderr: string;
|
|
257
|
-
timedOut?: boolean;
|
|
258
|
-
}
|
|
259
|
-
export type RunFn = (cmd: string[], opts: { cwd: string; timeoutMs?: number }) => Promise<RunResult>;
|
|
260
|
-
export type ShellRunFn = (command: string, opts: { cwd: string; timeoutMs?: number }) => Promise<RunResult>;
|