@mmerterden/multi-agent-pipeline 17.3.0 → 17.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +203 -0
- package/README.md +23 -5
- package/README.tr.md +23 -5
- package/docs/adr/0013-lsp-code-intelligence.md +102 -0
- package/docs/adr/README.md +1 -0
- package/docs/token-budget-history.md +1 -1
- package/install/templates/copilot-instructions.md +9 -3
- package/package.json +1 -1
- package/pipeline/agents/code-reviewer.md +35 -1
- package/pipeline/commands/multi-agent/analysis/SKILL.md +3 -3
- package/pipeline/commands/multi-agent/autopilot/SKILL.md +3 -3
- package/pipeline/commands/multi-agent/autopilot-off/SKILL.md +5 -3
- package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/local/SKILL.md +17 -6
- package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +3 -3
- package/pipeline/lib/multi-repo-pipeline.sh +26 -0
- package/pipeline/multi-agent-refs/analysis/locked.md +4 -4
- package/pipeline/multi-agent-refs/analysis/render.md +2 -1
- package/pipeline/multi-agent-refs/channels/pr.md +26 -0
- package/pipeline/multi-agent-refs/cross-cli-contract.md +22 -0
- package/pipeline/multi-agent-refs/features/base-branch-evidence.md +222 -0
- package/pipeline/multi-agent-refs/features/code-graph.md +40 -0
- package/pipeline/multi-agent-refs/features/code-intelligence.md +80 -0
- package/pipeline/multi-agent-refs/features/design-conformance.md +14 -0
- package/pipeline/multi-agent-refs/features/review-file-set.md +132 -0
- package/pipeline/multi-agent-refs/phases/modes.md +23 -3
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +96 -71
- package/pipeline/multi-agent-refs/phases/phase-4-review.md +31 -23
- package/pipeline/multi-agent-refs/phases/phase-7-report.md +1 -1
- package/pipeline/multi-agent-refs/phases.md +7 -2
- package/pipeline/multi-agent-refs/picker-contract.md +37 -5
- package/pipeline/multi-agent-refs/tracker-contract.md +25 -14
- package/pipeline/schemas/agent-state.schema.json +88 -4
- package/pipeline/schemas/prefs.schema.json +22 -0
- package/pipeline/schemas/review-file-exclusions.json +137 -0
- package/pipeline/schemas/reviewer-output.schema.json +27 -1
- package/pipeline/schemas/token-budget.json +2 -2
- package/pipeline/scripts/autopilot-runner.mjs +292 -45
- package/pipeline/scripts/base-branch-candidates.mjs +599 -0
- package/pipeline/scripts/diff-risk-score.mjs +1 -36
- package/pipeline/scripts/gc-abandoned.sh +5 -3
- package/pipeline/scripts/gen-mode-dispatch.mjs +39 -16
- package/pipeline/scripts/git-path.mjs +63 -0
- package/pipeline/scripts/glob-match.mjs +62 -0
- package/pipeline/scripts/graph-mermaid.mjs +251 -0
- package/pipeline/scripts/phase-tracker.sh +39 -2
- package/pipeline/scripts/phase0-exit-gate.mjs +128 -0
- package/pipeline/scripts/review-file-filter.mjs +180 -0
- package/pipeline/scripts/skill-conformance.mjs +1 -31
- package/pipeline/scripts/validate-analysis-doc.mjs +53 -0
- package/pipeline/scripts/validate-reviewer.mjs +90 -1
- package/pipeline/scripts/verify-citations.mjs +428 -0
- package/pipeline/skills/.skill-manifest.json +2 -2
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +1 -1
|
@@ -20,6 +20,20 @@
|
|
|
20
20
|
* `agent-state.json` is the only durable evidence they happened - so the gate asserts
|
|
21
21
|
* their output too, not just `taskType`.
|
|
22
22
|
*
|
|
23
|
+
* A third run got past that, because `baseBranch` and `baseFetchStatus` can be filled
|
|
24
|
+
* in by a branch that was never chosen. The remote had one PR-targetable branch, the
|
|
25
|
+
* one-option `AskUserQuestion` was refused by the host (its schema needs two), and the
|
|
26
|
+
* run announced "only candidate, continuing with it" and carried on. The branch was
|
|
27
|
+
* right; nothing was asked. `baseBranchSource` is the field that separates those two,
|
|
28
|
+
* and the dev-context picker never ran at all - no `siblings`, so Phase 4's parity
|
|
29
|
+
* cross-check had nothing to read and could not tell an empty answer from no answer.
|
|
30
|
+
*
|
|
31
|
+
* A fourth showed the same shape one layer down: `git fetch origin` failed on a
|
|
32
|
+
* restricted network, `git branch -r` printed the remote-tracking cache anyway, and a
|
|
33
|
+
* weeks-old local list was presented as the remote's answer. Degrading to local refs is
|
|
34
|
+
* correct; reporting them as remote is not, so `baseBranchEvidence.refProvenance` has to
|
|
35
|
+
* agree with `baseFetchStatus`.
|
|
36
|
+
*
|
|
23
37
|
* A phase that reports success without its output is worse than one that fails:
|
|
24
38
|
* every later phase then reasons from a field that is not there. So this is a
|
|
25
39
|
* gate, not a lint - the spec already said what to write, and prose alone did
|
|
@@ -157,6 +171,120 @@ export function evaluate(state, extraInput = "") {
|
|
|
157
171
|
);
|
|
158
172
|
}
|
|
159
173
|
|
|
174
|
+
// Which rule decided the base branch. `asked` and `input` are the only two an
|
|
175
|
+
// interactive run can honestly record: `remembered`, `default` and `derived` are the
|
|
176
|
+
// autopilot resolutions, and an interactive run that reaches for them has skipped its
|
|
177
|
+
// picker. This is the assertion `baseBranch` alone cannot make - a branch announced in
|
|
178
|
+
// prose and a branch chosen by the user leave the same value behind. A branch the run
|
|
179
|
+
// derived from the issue and the user then confirmed is still `asked`; the derivation
|
|
180
|
+
// lives in `baseBranchEvidence`, which is a record, not a permission.
|
|
181
|
+
const BRANCH_SOURCES = ["asked", "input", "remembered", "default", "derived"];
|
|
182
|
+
const branchSource =
|
|
183
|
+
typeof state.baseBranchSource === "string" ? state.baseBranchSource.trim() : "";
|
|
184
|
+
if (baseBranch && !BRANCH_SOURCES.includes(branchSource)) {
|
|
185
|
+
failures.push(
|
|
186
|
+
`agent-state.json has baseBranchSource="${branchSource || "<unset>"}"; Step 3 must ` +
|
|
187
|
+
`record one of ${BRANCH_SOURCES.join(" | ")}. Unset means nothing distinguishes a ` +
|
|
188
|
+
`branch the user chose from one the run picked and announced.`,
|
|
189
|
+
);
|
|
190
|
+
}
|
|
191
|
+
const isAutopilot = state.autopilot === true;
|
|
192
|
+
const AUTOPILOT_ONLY_SOURCES = ["remembered", "default", "derived"];
|
|
193
|
+
if (!isAutopilot && AUTOPILOT_ONLY_SOURCES.includes(branchSource)) {
|
|
194
|
+
failures.push(
|
|
195
|
+
`baseBranchSource="${branchSource}" on an interactive run. Those three are autopilot ` +
|
|
196
|
+
`resolutions; an interactive run asks (Step 3 is not skippable) or takes the base ` +
|
|
197
|
+
`from the task reference. A one-candidate filter is still asked, with a second ` +
|
|
198
|
+
`option - picker-contract.md, "Two options or it is not a question".`,
|
|
199
|
+
);
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
// What the base branch was chosen from, and what that list was worth.
|
|
203
|
+
//
|
|
204
|
+
// Two separate silent failures live here. The first: `git fetch origin` can fail on a
|
|
205
|
+
// restricted network while `git branch -r` still prints a full, confident list - the
|
|
206
|
+
// remote-tracking cache - so a weeks-old local guess gets presented as the remote's
|
|
207
|
+
// answer. Falling back to local refs is correct; not saying so is not. The second:
|
|
208
|
+
// `derived` claims the issue named the branch, and a claim with no evidence behind it
|
|
209
|
+
// is `default` wearing a hat.
|
|
210
|
+
const evidence =
|
|
211
|
+
state.baseBranchEvidence && typeof state.baseBranchEvidence === "object"
|
|
212
|
+
? state.baseBranchEvidence
|
|
213
|
+
: null;
|
|
214
|
+
const DEGRADED_FETCH = ["cached-stale", "local-branch"];
|
|
215
|
+
if (DEGRADED_FETCH.includes(fetchStatus)) {
|
|
216
|
+
if (!evidence) {
|
|
217
|
+
failures.push(
|
|
218
|
+
`baseFetchStatus="${fetchStatus}" but state.baseBranchEvidence is absent. A run whose ` +
|
|
219
|
+
`fetch failed listed its branches from local refs; the record of that is what stops a ` +
|
|
220
|
+
`local-only guess being read afterwards as the remote's answer.`,
|
|
221
|
+
);
|
|
222
|
+
} else if (evidence.refProvenance !== "local") {
|
|
223
|
+
failures.push(
|
|
224
|
+
`baseFetchStatus="${fetchStatus}" but baseBranchEvidence.refProvenance=` +
|
|
225
|
+
`"${evidence.refProvenance || "<unset>"}". The fetch failed, so the candidate list came ` +
|
|
226
|
+
`from local refs and must say so - "remote" here is the silent degradation this field exists to catch.`,
|
|
227
|
+
);
|
|
228
|
+
}
|
|
229
|
+
}
|
|
230
|
+
if (branchSource === "derived") {
|
|
231
|
+
const ISSUE_DERIVED = new Set(["issue-version", "linked-release"]);
|
|
232
|
+
const cands = evidence && Array.isArray(evidence.candidates) ? evidence.candidates : [];
|
|
233
|
+
const chosen = cands.find((c) => c && c.branch === baseBranch);
|
|
234
|
+
const hasIssueEvidence =
|
|
235
|
+
chosen &&
|
|
236
|
+
Array.isArray(chosen.evidence) &&
|
|
237
|
+
chosen.evidence.some((e) => e && ISSUE_DERIVED.has(e.kind));
|
|
238
|
+
if (!hasIssueEvidence) {
|
|
239
|
+
failures.push(
|
|
240
|
+
`baseBranchSource="derived" but baseBranchEvidence carries no issue-version or ` +
|
|
241
|
+
`linked-release evidence for "${baseBranch}". "Derived" names a specific claim - the ` +
|
|
242
|
+
`issue's version field or a linked release issue pointed at this branch - and without ` +
|
|
243
|
+
`that record it is the sort-order default under a better name.`,
|
|
244
|
+
);
|
|
245
|
+
}
|
|
246
|
+
if (evidence && evidence.ambiguous === true) {
|
|
247
|
+
failures.push(
|
|
248
|
+
`baseBranchSource="derived" with baseBranchEvidence.ambiguous=true. Two or more ` +
|
|
249
|
+
`candidates tied at the top score; autopilot does not break a tie by picking one.`,
|
|
250
|
+
);
|
|
251
|
+
}
|
|
252
|
+
}
|
|
253
|
+
|
|
254
|
+
// Step 5b decides where the branch lives, and `localMode` alone cannot say
|
|
255
|
+
// whether anyone decided: `false` is both "the user chose a worktree" and
|
|
256
|
+
// "nothing asked and the default stood". Same shape as baseBranchSource, and
|
|
257
|
+
// the same reason - an autopilot run resolves it rather than asking, so
|
|
258
|
+
// `autopilot` is a legal source there and nowhere else.
|
|
259
|
+
const WORKSPACE_SOURCES = ["asked", "command", "autopilot"];
|
|
260
|
+
const workspaceSource =
|
|
261
|
+
typeof state.workspaceSource === "string" ? state.workspaceSource.trim() : "";
|
|
262
|
+
if (!WORKSPACE_SOURCES.includes(workspaceSource)) {
|
|
263
|
+
failures.push(
|
|
264
|
+
`agent-state.json has workspaceSource="${workspaceSource || "<unset>"}"; Step 5b must ` +
|
|
265
|
+
`record one of ${WORKSPACE_SOURCES.join(" | ")}. Unset means nothing distinguishes a ` +
|
|
266
|
+
`worktree the user chose from one no question was asked about.`,
|
|
267
|
+
);
|
|
268
|
+
}
|
|
269
|
+
if (!isAutopilot && workspaceSource === "autopilot") {
|
|
270
|
+
failures.push(
|
|
271
|
+
`workspaceSource="autopilot" on an interactive run. Autopilot resolves the workspace ` +
|
|
272
|
+
`to a worktree because an unattended commit in the user's own checkout is what ` +
|
|
273
|
+
`worktrees prevent; an interactive run asks (Step 5b) or is told by :local / --local.`,
|
|
274
|
+
);
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
// Step 2b's dev-context picker writes siblings[], empty included. The empty array is
|
|
278
|
+
// the record that it ran; absent, a multi-repo task silently became a single-repo one
|
|
279
|
+
// and Phase 4's parity cross-check lost its fourth counterpart source.
|
|
280
|
+
if (!Array.isArray(state.siblings)) {
|
|
281
|
+
failures.push(
|
|
282
|
+
"agent-state.json has no siblings array. Step 2b runs the dev-context picker on " +
|
|
283
|
+
"every input type and persists the result, `[]` included; an absent field means " +
|
|
284
|
+
"the picker never ran, so extra repos and read-only counterparts were never offered.",
|
|
285
|
+
);
|
|
286
|
+
}
|
|
287
|
+
|
|
160
288
|
// Worktree isolation is a standing rule: never develop in the primary checkout.
|
|
161
289
|
const worktree = typeof state.worktreePath === "string" ? state.worktreePath.trim() : "";
|
|
162
290
|
const projectRoot = typeof state.projectRoot === "string" ? state.projectRoot.trim() : "";
|
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* @file review-file-filter.mjs - decide what the reviewers are asked to read.
|
|
5
|
+
*
|
|
6
|
+
* Phase 4 has a size cap and no exclusion list. When the diff exceeds the
|
|
7
|
+
* budget the cap truncates the LARGEST files first (`phase-4-review.md`, Step
|
|
8
|
+
* 1.9), so a regenerated lockfile or a snapshot dump does not merely waste
|
|
9
|
+
* tokens - it is the thing that survives while real code is cut. The cheapest
|
|
10
|
+
* fix is to decide what is worth reading before the cap decides what fits.
|
|
11
|
+
*
|
|
12
|
+
* Every exclusion carries the reason it was excluded, and the two lists
|
|
13
|
+
* partition the input exactly. A file that quietly disappears between the diff
|
|
14
|
+
* and the reviewer is indistinguishable from a file nobody found anything in,
|
|
15
|
+
* and that is the failure this script exists to prevent.
|
|
16
|
+
*
|
|
17
|
+
* Patterns live in `schemas/review-file-exclusions.json` so the list is data,
|
|
18
|
+
* versioned beside the code that reads it, and generic: no stack, project or
|
|
19
|
+
* company name appears in it.
|
|
20
|
+
*
|
|
21
|
+
* Glob matching comes from `glob-match.mjs`, the same module
|
|
22
|
+
* `skill-conformance.mjs` uses. Two matchers would drift, and the day they
|
|
23
|
+
* disagree a file is excluded from the review and still counted in the
|
|
24
|
+
* conformance denominator, with nothing saying so.
|
|
25
|
+
*
|
|
26
|
+
* Inputs:
|
|
27
|
+
* (stdin) One path per line. `git diff --name-only` output.
|
|
28
|
+
* --files <path> Read the path list from a file instead of stdin
|
|
29
|
+
* --patterns <path> Override the pattern file
|
|
30
|
+
* --json Emit the full report (default)
|
|
31
|
+
* --reviewed Print only the reviewed paths, one per line
|
|
32
|
+
*
|
|
33
|
+
* Exit codes:
|
|
34
|
+
* 0 - filtered
|
|
35
|
+
* 2 - the pattern list could not be read. The report still comes out with
|
|
36
|
+
* EVERY file reviewed, because the safe direction is to read too much:
|
|
37
|
+
* a filter that fails closed would silently review nothing.
|
|
38
|
+
* 64 - usage error
|
|
39
|
+
*
|
|
40
|
+
* @module pipeline/scripts/review-file-filter
|
|
41
|
+
*/
|
|
42
|
+
|
|
43
|
+
import { readFileSync } from "node:fs";
|
|
44
|
+
import { join } from "node:path";
|
|
45
|
+
import { globToRegExp } from "./glob-match.mjs";
|
|
46
|
+
import { unquotePathIfNeeded } from "./git-path.mjs";
|
|
47
|
+
|
|
48
|
+
const DEFAULT_PATTERNS = join(import.meta.dirname, "..", "schemas", "review-file-exclusions.json");
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Read and validate the pattern list.
|
|
52
|
+
*
|
|
53
|
+
* A pattern with no reason is rejected rather than defaulted, because the
|
|
54
|
+
* default would be the thing the caller is supposed to print.
|
|
55
|
+
*
|
|
56
|
+
* @param {string} path
|
|
57
|
+
* @returns {{patterns: {glob: string, reason: string, re: RegExp}[], error: string|null}}
|
|
58
|
+
*/
|
|
59
|
+
export function loadPatterns(path) {
|
|
60
|
+
let doc;
|
|
61
|
+
try {
|
|
62
|
+
doc = JSON.parse(readFileSync(path, "utf-8"));
|
|
63
|
+
} catch (e) {
|
|
64
|
+
return { patterns: [], error: `cannot read ${path}: ${e.message}` };
|
|
65
|
+
}
|
|
66
|
+
const rows = Array.isArray(doc?.patterns) ? doc.patterns : null;
|
|
67
|
+
if (!rows) return { patterns: [], error: `${path} has no patterns[] array` };
|
|
68
|
+
|
|
69
|
+
const patterns = [];
|
|
70
|
+
for (const [i, r] of rows.entries()) {
|
|
71
|
+
const glob = typeof r?.glob === "string" ? r.glob.trim() : "";
|
|
72
|
+
const reason = typeof r?.reason === "string" ? r.reason.trim() : "";
|
|
73
|
+
if (!glob) return { patterns: [], error: `patterns[${i}] has no glob` };
|
|
74
|
+
if (!reason) return { patterns: [], error: `patterns[${i}] (${glob}) has no reason` };
|
|
75
|
+
patterns.push({ glob, reason, re: globToRegExp(glob) });
|
|
76
|
+
}
|
|
77
|
+
return { patterns, error: null };
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Split the changed files into what the reviewers read and what they do not.
|
|
82
|
+
*
|
|
83
|
+
* @param {string[]} files
|
|
84
|
+
* @param {{glob: string, reason: string, re: RegExp}[]} patterns
|
|
85
|
+
* @returns {{reviewed: string[], excluded: {path: string, reason: string, pattern: string}[]}}
|
|
86
|
+
*/
|
|
87
|
+
export function partition(files, patterns) {
|
|
88
|
+
const reviewed = [];
|
|
89
|
+
const excluded = [];
|
|
90
|
+
for (const f of files) {
|
|
91
|
+
const hit = patterns.find((p) => p.re.test(f));
|
|
92
|
+
if (hit) excluded.push({ path: f, reason: hit.reason, pattern: hit.glob });
|
|
93
|
+
else reviewed.push(f);
|
|
94
|
+
}
|
|
95
|
+
return { reviewed, excluded };
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
/**
|
|
99
|
+
* @param {string} raw
|
|
100
|
+
* @returns {string[]}
|
|
101
|
+
*/
|
|
102
|
+
export function readPaths(raw) {
|
|
103
|
+
const seen = new Set();
|
|
104
|
+
const out = [];
|
|
105
|
+
for (const line of raw.split("\n")) {
|
|
106
|
+
// Only the line ending is stripped: a leading or trailing space is a real
|
|
107
|
+
// character in a path. The C-style quoting git applies to a non-ASCII name
|
|
108
|
+
// IS undone, because a quoted `"G\303\266..."` matches no glob and would
|
|
109
|
+
// drop the file out of the denominator without anything reporting it -
|
|
110
|
+
// exactly the bug diff-risk-score.mjs already carries a fix for.
|
|
111
|
+
const p = unquotePathIfNeeded(line.replace(/\r$/, ""));
|
|
112
|
+
if (!p || seen.has(p)) continue;
|
|
113
|
+
seen.add(p);
|
|
114
|
+
out.push(p);
|
|
115
|
+
}
|
|
116
|
+
return out;
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
function main() {
|
|
120
|
+
const argv = process.argv.slice(2);
|
|
121
|
+
// A value flag with no value is a usage error, not a silent default. Falling
|
|
122
|
+
// back would hand a caller who typo'd `--patterns` the SHIPPED list while
|
|
123
|
+
// they believed they were running their own.
|
|
124
|
+
const valueOf = (flag) => {
|
|
125
|
+
const i = argv.indexOf(flag);
|
|
126
|
+
if (i === -1) return null;
|
|
127
|
+
const v = argv[i + 1];
|
|
128
|
+
if (v === undefined || v.startsWith("--")) {
|
|
129
|
+
console.error(`review-file-filter: ${flag} needs a path`);
|
|
130
|
+
process.exit(64);
|
|
131
|
+
}
|
|
132
|
+
return v;
|
|
133
|
+
};
|
|
134
|
+
for (const a of argv) {
|
|
135
|
+
if (a.startsWith("--") && !["--files", "--patterns", "--json", "--reviewed"].includes(a)) {
|
|
136
|
+
console.error(`review-file-filter: unknown flag ${a}`);
|
|
137
|
+
process.exit(64);
|
|
138
|
+
}
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
const filesArg = valueOf("--files");
|
|
142
|
+
let raw;
|
|
143
|
+
try {
|
|
144
|
+
raw = filesArg ? readFileSync(filesArg, "utf-8") : readFileSync(0, "utf-8");
|
|
145
|
+
} catch (e) {
|
|
146
|
+
console.error(`review-file-filter: cannot read the file list: ${e.message}`);
|
|
147
|
+
process.exit(64);
|
|
148
|
+
}
|
|
149
|
+
const files = readPaths(raw);
|
|
150
|
+
|
|
151
|
+
const patternsPath = valueOf("--patterns") || DEFAULT_PATTERNS;
|
|
152
|
+
const { patterns, error } = loadPatterns(patternsPath);
|
|
153
|
+
const { reviewed, excluded } = error
|
|
154
|
+
? { reviewed: files.slice(), excluded: [] }
|
|
155
|
+
: partition(files, patterns);
|
|
156
|
+
|
|
157
|
+
const report = {
|
|
158
|
+
reviewed,
|
|
159
|
+
excluded,
|
|
160
|
+
total: files.length,
|
|
161
|
+
patternsSource: patternsPath,
|
|
162
|
+
patternCount: patterns.length,
|
|
163
|
+
...(error ? { patternsError: error } : {}),
|
|
164
|
+
};
|
|
165
|
+
|
|
166
|
+
if (argv.includes("--reviewed")) {
|
|
167
|
+
for (const f of reviewed) console.log(f);
|
|
168
|
+
} else {
|
|
169
|
+
console.log(JSON.stringify(report, null, 2));
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
if (error) {
|
|
173
|
+
console.error(`review-file-filter: ${error} - reviewing all ${files.length} file(s)`);
|
|
174
|
+
process.exit(2);
|
|
175
|
+
}
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
if (import.meta.url === `file://${process.argv[1]}`) {
|
|
179
|
+
main();
|
|
180
|
+
}
|
|
@@ -44,6 +44,7 @@ import { existsSync, readFileSync, writeFileSync, readdirSync, statSync } from "
|
|
|
44
44
|
import { dirname, join, resolve, relative, basename, extname, sep } from "node:path";
|
|
45
45
|
import { fileURLToPath } from "node:url";
|
|
46
46
|
import { homedir } from "node:os";
|
|
47
|
+
import { matchesAnyGlob } from "./glob-match.mjs";
|
|
47
48
|
|
|
48
49
|
const here = dirname(fileURLToPath(import.meta.url));
|
|
49
50
|
|
|
@@ -364,37 +365,6 @@ function classifyLanguage(file) {
|
|
|
364
365
|
return LANG_BY_EXT[extname(base).toLowerCase()] ?? "other";
|
|
365
366
|
}
|
|
366
367
|
|
|
367
|
-
// ---------------------------------------------------------------------------
|
|
368
|
-
// Minimal glob matcher for the `**/x`, `*.ext`, `dir/**` shapes registries use.
|
|
369
|
-
// ---------------------------------------------------------------------------
|
|
370
|
-
function globToRegExp(glob) {
|
|
371
|
-
let re = "";
|
|
372
|
-
for (let i = 0; i < glob.length; i++) {
|
|
373
|
-
const c = glob[i];
|
|
374
|
-
if (c === "*") {
|
|
375
|
-
if (glob[i + 1] === "*") {
|
|
376
|
-
// `**/` matches zero or more path segments.
|
|
377
|
-
if (glob[i + 2] === "/") {
|
|
378
|
-
re += "(?:.*/)?";
|
|
379
|
-
i += 2;
|
|
380
|
-
} else {
|
|
381
|
-
re += ".*";
|
|
382
|
-
i += 1;
|
|
383
|
-
}
|
|
384
|
-
} else {
|
|
385
|
-
re += "[^/]*";
|
|
386
|
-
}
|
|
387
|
-
} else if (c === "?") re += "[^/]";
|
|
388
|
-
else re += c.replace(/[.+^${}()|[\]\\]/g, "\\$&");
|
|
389
|
-
}
|
|
390
|
-
return new RegExp(`^${re}$`);
|
|
391
|
-
}
|
|
392
|
-
|
|
393
|
-
function matchesAnyGlob(file, globs) {
|
|
394
|
-
if (!globs || globs.length === 0) return true;
|
|
395
|
-
return globs.some((g) => globToRegExp(g).test(file));
|
|
396
|
-
}
|
|
397
|
-
|
|
398
368
|
// ---------------------------------------------------------------------------
|
|
399
369
|
// Diff parsing: changed files and added lines (for the exception audit).
|
|
400
370
|
// ---------------------------------------------------------------------------
|
|
@@ -635,6 +635,59 @@ function main() {
|
|
|
635
635
|
);
|
|
636
636
|
}
|
|
637
637
|
|
|
638
|
+
// 2d-2. The diagram has to describe the document it sits in.
|
|
639
|
+
//
|
|
640
|
+
// Until now the only question asked about Section 3 was "is a mermaid block
|
|
641
|
+
// present". A sequence diagram could name a participant the document never
|
|
642
|
+
// mentions, or omit a service the document devotes a section to, and nothing
|
|
643
|
+
// said so - the diagram is the part a reader trusts most and the part nothing
|
|
644
|
+
// checked. Both directions are reported, because they are different mistakes:
|
|
645
|
+
// an invented participant is a claim the document does not support, and a
|
|
646
|
+
// missing one is a service the reader will not see in the picture.
|
|
647
|
+
//
|
|
648
|
+
// Warn, not error: a participant can legitimately be an actor ("User") or an
|
|
649
|
+
// external system named nowhere else, and failing a correct document is how a
|
|
650
|
+
// check gets switched off. The warning names the label so it is one grep to
|
|
651
|
+
// settle.
|
|
652
|
+
const participants = [];
|
|
653
|
+
let inMermaid = false;
|
|
654
|
+
for (const raw of text.split("\n")) {
|
|
655
|
+
const t = raw.trim();
|
|
656
|
+
if (/^```/.test(t)) {
|
|
657
|
+
inMermaid = /^```mermaid\b/.test(t);
|
|
658
|
+
continue;
|
|
659
|
+
}
|
|
660
|
+
if (!inMermaid) continue;
|
|
661
|
+
const m = t.match(/^(?:participant|actor)\s+(\S+)(?:\s+as\s+(.+))?$/);
|
|
662
|
+
if (m) participants.push((m[2] || m[1]).trim());
|
|
663
|
+
}
|
|
664
|
+
|
|
665
|
+
if (participants.length > 0) {
|
|
666
|
+
// The prose is everything outside the fences: a participant that only ever
|
|
667
|
+
// appears inside the diagram has no support in the document.
|
|
668
|
+
const prose = text
|
|
669
|
+
.split(/^```[\s\S]*?^```/gm)
|
|
670
|
+
.join("\n")
|
|
671
|
+
.toLowerCase();
|
|
672
|
+
// Only identifier-shaped labels are evidence. A human actor is an ordinary
|
|
673
|
+
// word in whatever language the document is written in - "User",
|
|
674
|
+
// "Kullanici", "Operator" - and its absence from the prose says nothing. A
|
|
675
|
+
// service or component is written like code: an internal capital
|
|
676
|
+
// (PaymentService) or a separator (payment-service). Testing the SHAPE
|
|
677
|
+
// rather than keeping a word list keeps this working in both languages and
|
|
678
|
+
// stops the list from rotting.
|
|
679
|
+
const looksLikeIdentifier = (x) => /[a-z][A-Z]/.test(x) || (/[._-]/.test(x) && x.length >= 6);
|
|
680
|
+
const orphans = participants.filter((p) => {
|
|
681
|
+
const raw = p.replace(/^["'`]|["'`]$/g, "");
|
|
682
|
+
return looksLikeIdentifier(raw) && !prose.includes(raw.toLowerCase());
|
|
683
|
+
});
|
|
684
|
+
if (orphans.length > 0) {
|
|
685
|
+
warns.push(
|
|
686
|
+
`Section 3 diagram names ${orphans.length} participant(s) the document never mentions: ${orphans.join(", ")}`,
|
|
687
|
+
);
|
|
688
|
+
}
|
|
689
|
+
}
|
|
690
|
+
|
|
638
691
|
mark("flow-chart");
|
|
639
692
|
|
|
640
693
|
// 2e. Locked 16: every Files-to-Add row is tagged. The decision says untagged
|
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
// node validate-reviewer.mjs path/to/reviewer.json
|
|
12
12
|
// cat reviewer.json | node validate-reviewer.mjs -
|
|
13
13
|
// [--criteria path/to/criteria-manifest.json] # enforce the rule-ID checklist
|
|
14
|
+
// [--coverage path/to/review-files.json] # enforce the file denominator
|
|
14
15
|
//
|
|
15
16
|
// With --criteria, the validator also enforces the conformance contract added in
|
|
16
17
|
// schema v1.1.0. Without that enforcement the field is decoration: this validator
|
|
@@ -29,13 +30,16 @@ import { readFileSync } from "node:fs";
|
|
|
29
30
|
|
|
30
31
|
const ALLOWED_SEVERITIES = new Set(["blocking", "important", "suggestion"]);
|
|
31
32
|
const ALLOWED_VERDICTS = new Set(["conformant", "violated", "not-applicable"]);
|
|
33
|
+
const ALLOWED_COVERAGE = new Set(["reviewed", "skipped"]);
|
|
32
34
|
const FINGERPRINT_RE = /^F:[0-9a-f]{8}$/;
|
|
33
35
|
|
|
34
36
|
/** @returns {Promise<string>} */
|
|
35
37
|
function readInput() {
|
|
36
38
|
const arg = process.argv[2];
|
|
37
39
|
if (!arg) {
|
|
38
|
-
console.error(
|
|
40
|
+
console.error(
|
|
41
|
+
"usage: validate-reviewer.mjs <path|-> [--criteria <manifest>] [--coverage <report>]",
|
|
42
|
+
);
|
|
39
43
|
process.exit(64);
|
|
40
44
|
}
|
|
41
45
|
if (arg === "-") {
|
|
@@ -81,6 +85,72 @@ function validateFinding(f, label, errors) {
|
|
|
81
85
|
}
|
|
82
86
|
}
|
|
83
87
|
|
|
88
|
+
/** Files the reviewer was given. Empty set = no denominator supplied. */
|
|
89
|
+
function loadReviewedFiles(path) {
|
|
90
|
+
const report = JSON.parse(readFileSync(path, "utf-8"));
|
|
91
|
+
return new Set((report?.reviewed ?? []).filter((f) => typeof f === "string" && f.length > 0));
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Enforce the file checklist against the fixed denominator.
|
|
96
|
+
*
|
|
97
|
+
* The same three failures the conformance checklist has, on the other axis:
|
|
98
|
+
* a file with no row (silently unread), a row for a file outside the set
|
|
99
|
+
* (a reviewer answering about something it was not given), and `skipped`
|
|
100
|
+
* with no reason. Without this, a reviewer that opened one file of ten and a
|
|
101
|
+
* reviewer that found nothing in all ten return the identical empty findings[].
|
|
102
|
+
*/
|
|
103
|
+
function validateFileCoverage(parsed, reviewedFiles, errors) {
|
|
104
|
+
const rows = parsed.fileCoverage;
|
|
105
|
+
if (!Array.isArray(rows)) {
|
|
106
|
+
errors.push(
|
|
107
|
+
`fileCoverage must be an array with one row per file in the review set (${reviewedFiles.size} expected)`,
|
|
108
|
+
);
|
|
109
|
+
return;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
const seen = new Map();
|
|
113
|
+
rows.forEach((row, i) => {
|
|
114
|
+
const label = `fileCoverage[${i}]`;
|
|
115
|
+
if (typeof row !== "object" || row === null) {
|
|
116
|
+
errors.push(`${label}: not an object`);
|
|
117
|
+
return;
|
|
118
|
+
}
|
|
119
|
+
const path = row.path;
|
|
120
|
+
if (typeof path !== "string" || path.length === 0) {
|
|
121
|
+
errors.push(`${label}: missing path`);
|
|
122
|
+
return;
|
|
123
|
+
}
|
|
124
|
+
if (!ALLOWED_COVERAGE.has(row.verdict)) {
|
|
125
|
+
errors.push(`${label} (${path}): bad verdict ${JSON.stringify(row.verdict)}`);
|
|
126
|
+
}
|
|
127
|
+
if (!reviewedFiles.has(path)) {
|
|
128
|
+
errors.push(
|
|
129
|
+
`${label}: "${path}" is not in the review set - answer only for the files you were given`,
|
|
130
|
+
);
|
|
131
|
+
}
|
|
132
|
+
seen.set(path, (seen.get(path) ?? 0) + 1);
|
|
133
|
+
|
|
134
|
+
if (row.verdict === "skipped" && (typeof row.reason !== "string" || row.reason.length < 4)) {
|
|
135
|
+
errors.push(
|
|
136
|
+
`${label} (${path}): verdict "skipped" needs a reason - a file dropped without one is indistinguishable from one that was read`,
|
|
137
|
+
);
|
|
138
|
+
}
|
|
139
|
+
});
|
|
140
|
+
|
|
141
|
+
for (const [path, count] of seen) {
|
|
142
|
+
if (count > 1)
|
|
143
|
+
errors.push(`fileCoverage: "${path}" appears ${count} times (expected exactly once)`);
|
|
144
|
+
}
|
|
145
|
+
const missing = [...reviewedFiles].filter((f) => !seen.has(f));
|
|
146
|
+
if (missing.length > 0) {
|
|
147
|
+
const shown = missing.slice(0, 8).join(", ");
|
|
148
|
+
errors.push(
|
|
149
|
+
`fileCoverage: ${missing.length} file(s) in the review set have no row (${shown}${missing.length > 8 ? ", ..." : ""}) - neither read nor declared skipped`,
|
|
150
|
+
);
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
|
|
84
154
|
/** Rule IDs the reviewer was told to answer for. Empty set = no criteria supplied. */
|
|
85
155
|
function loadSelectedRuleIds(path) {
|
|
86
156
|
const manifest = JSON.parse(readFileSync(path, "utf-8"));
|
|
@@ -214,6 +284,25 @@ async function main() {
|
|
|
214
284
|
}
|
|
215
285
|
}
|
|
216
286
|
|
|
287
|
+
// File checklist, only when the orchestrator supplied a denominator.
|
|
288
|
+
const covIdx = process.argv.indexOf("--coverage");
|
|
289
|
+
if (covIdx !== -1) {
|
|
290
|
+
const covPath = process.argv[covIdx + 1];
|
|
291
|
+
if (!covPath) {
|
|
292
|
+
errors.push("--coverage given with no report path");
|
|
293
|
+
} else {
|
|
294
|
+
try {
|
|
295
|
+
const reviewedFiles = loadReviewedFiles(covPath);
|
|
296
|
+
// Everything excluded means there is nothing to answer for. Demanding an
|
|
297
|
+
// empty array would fail honest output, and the filter's excluded[] is
|
|
298
|
+
// what carries that information onward.
|
|
299
|
+
if (reviewedFiles.size > 0) validateFileCoverage(parsed, reviewedFiles, errors);
|
|
300
|
+
} catch (err) {
|
|
301
|
+
errors.push(`cannot read review-set report: ${err.message}`);
|
|
302
|
+
}
|
|
303
|
+
}
|
|
304
|
+
}
|
|
305
|
+
|
|
217
306
|
// Internal consistency: approved=true AND blocking findings = contradiction
|
|
218
307
|
if (
|
|
219
308
|
errors.length === 0 &&
|