@mmerterden/multi-agent-pipeline 17.3.0 → 17.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. package/CHANGELOG.md +203 -0
  2. package/README.md +23 -5
  3. package/README.tr.md +23 -5
  4. package/docs/adr/0013-lsp-code-intelligence.md +102 -0
  5. package/docs/adr/README.md +1 -0
  6. package/docs/token-budget-history.md +1 -1
  7. package/install/templates/copilot-instructions.md +9 -3
  8. package/package.json +1 -1
  9. package/pipeline/agents/code-reviewer.md +35 -1
  10. package/pipeline/commands/multi-agent/analysis/SKILL.md +3 -3
  11. package/pipeline/commands/multi-agent/autopilot/SKILL.md +3 -3
  12. package/pipeline/commands/multi-agent/autopilot-off/SKILL.md +5 -3
  13. package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
  14. package/pipeline/commands/multi-agent/local/SKILL.md +17 -6
  15. package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +3 -3
  16. package/pipeline/lib/multi-repo-pipeline.sh +26 -0
  17. package/pipeline/multi-agent-refs/analysis/locked.md +4 -4
  18. package/pipeline/multi-agent-refs/analysis/render.md +2 -1
  19. package/pipeline/multi-agent-refs/channels/pr.md +26 -0
  20. package/pipeline/multi-agent-refs/cross-cli-contract.md +22 -0
  21. package/pipeline/multi-agent-refs/features/base-branch-evidence.md +222 -0
  22. package/pipeline/multi-agent-refs/features/code-graph.md +40 -0
  23. package/pipeline/multi-agent-refs/features/code-intelligence.md +80 -0
  24. package/pipeline/multi-agent-refs/features/design-conformance.md +14 -0
  25. package/pipeline/multi-agent-refs/features/review-file-set.md +132 -0
  26. package/pipeline/multi-agent-refs/phases/modes.md +23 -3
  27. package/pipeline/multi-agent-refs/phases/phase-0-init.md +96 -71
  28. package/pipeline/multi-agent-refs/phases/phase-4-review.md +31 -23
  29. package/pipeline/multi-agent-refs/phases/phase-7-report.md +1 -1
  30. package/pipeline/multi-agent-refs/phases.md +7 -2
  31. package/pipeline/multi-agent-refs/picker-contract.md +37 -5
  32. package/pipeline/multi-agent-refs/tracker-contract.md +25 -14
  33. package/pipeline/schemas/agent-state.schema.json +88 -4
  34. package/pipeline/schemas/prefs.schema.json +22 -0
  35. package/pipeline/schemas/review-file-exclusions.json +137 -0
  36. package/pipeline/schemas/reviewer-output.schema.json +27 -1
  37. package/pipeline/schemas/token-budget.json +2 -2
  38. package/pipeline/scripts/autopilot-runner.mjs +292 -45
  39. package/pipeline/scripts/base-branch-candidates.mjs +599 -0
  40. package/pipeline/scripts/diff-risk-score.mjs +1 -36
  41. package/pipeline/scripts/gc-abandoned.sh +5 -3
  42. package/pipeline/scripts/gen-mode-dispatch.mjs +39 -16
  43. package/pipeline/scripts/git-path.mjs +63 -0
  44. package/pipeline/scripts/glob-match.mjs +62 -0
  45. package/pipeline/scripts/graph-mermaid.mjs +251 -0
  46. package/pipeline/scripts/phase-tracker.sh +39 -2
  47. package/pipeline/scripts/phase0-exit-gate.mjs +128 -0
  48. package/pipeline/scripts/review-file-filter.mjs +180 -0
  49. package/pipeline/scripts/skill-conformance.mjs +1 -31
  50. package/pipeline/scripts/validate-analysis-doc.mjs +53 -0
  51. package/pipeline/scripts/validate-reviewer.mjs +90 -1
  52. package/pipeline/scripts/verify-citations.mjs +428 -0
  53. package/pipeline/skills/.skill-manifest.json +2 -2
  54. package/pipeline/skills/shared/core/multi-agent/SKILL.md +1 -1
@@ -20,6 +20,20 @@
20
20
  * `agent-state.json` is the only durable evidence they happened - so the gate asserts
21
21
  * their output too, not just `taskType`.
22
22
  *
23
+ * A third run got past that, because `baseBranch` and `baseFetchStatus` can be filled
24
+ * in by a branch that was never chosen. The remote had one PR-targetable branch, the
25
+ * one-option `AskUserQuestion` was refused by the host (its schema needs two), and the
26
+ * run announced "only candidate, continuing with it" and carried on. The branch was
27
+ * right; nothing was asked. `baseBranchSource` is the field that separates those two,
28
+ * and the dev-context picker never ran at all - no `siblings`, so Phase 4's parity
29
+ * cross-check had nothing to read and could not tell an empty answer from no answer.
30
+ *
31
+ * A fourth showed the same shape one layer down: `git fetch origin` failed on a
32
+ * restricted network, `git branch -r` printed the remote-tracking cache anyway, and a
33
+ * weeks-old local list was presented as the remote's answer. Degrading to local refs is
34
+ * correct; reporting them as remote is not, so `baseBranchEvidence.refProvenance` has to
35
+ * agree with `baseFetchStatus`.
36
+ *
23
37
  * A phase that reports success without its output is worse than one that fails:
24
38
  * every later phase then reasons from a field that is not there. So this is a
25
39
  * gate, not a lint - the spec already said what to write, and prose alone did
@@ -157,6 +171,120 @@ export function evaluate(state, extraInput = "") {
157
171
  );
158
172
  }
159
173
 
174
+ // Which rule decided the base branch. `asked` and `input` are the only two an
175
+ // interactive run can honestly record: `remembered`, `default` and `derived` are the
176
+ // autopilot resolutions, and an interactive run that reaches for them has skipped its
177
+ // picker. This is the assertion `baseBranch` alone cannot make - a branch announced in
178
+ // prose and a branch chosen by the user leave the same value behind. A branch the run
179
+ // derived from the issue and the user then confirmed is still `asked`; the derivation
180
+ // lives in `baseBranchEvidence`, which is a record, not a permission.
181
+ const BRANCH_SOURCES = ["asked", "input", "remembered", "default", "derived"];
182
+ const branchSource =
183
+ typeof state.baseBranchSource === "string" ? state.baseBranchSource.trim() : "";
184
+ if (baseBranch && !BRANCH_SOURCES.includes(branchSource)) {
185
+ failures.push(
186
+ `agent-state.json has baseBranchSource="${branchSource || "<unset>"}"; Step 3 must ` +
187
+ `record one of ${BRANCH_SOURCES.join(" | ")}. Unset means nothing distinguishes a ` +
188
+ `branch the user chose from one the run picked and announced.`,
189
+ );
190
+ }
191
+ const isAutopilot = state.autopilot === true;
192
+ const AUTOPILOT_ONLY_SOURCES = ["remembered", "default", "derived"];
193
+ if (!isAutopilot && AUTOPILOT_ONLY_SOURCES.includes(branchSource)) {
194
+ failures.push(
195
+ `baseBranchSource="${branchSource}" on an interactive run. Those three are autopilot ` +
196
+ `resolutions; an interactive run asks (Step 3 is not skippable) or takes the base ` +
197
+ `from the task reference. A one-candidate filter is still asked, with a second ` +
198
+ `option - picker-contract.md, "Two options or it is not a question".`,
199
+ );
200
+ }
201
+
202
+ // What the base branch was chosen from, and what that list was worth.
203
+ //
204
+ // Two separate silent failures live here. The first: `git fetch origin` can fail on a
205
+ // restricted network while `git branch -r` still prints a full, confident list - the
206
+ // remote-tracking cache - so a weeks-old local guess gets presented as the remote's
207
+ // answer. Falling back to local refs is correct; not saying so is not. The second:
208
+ // `derived` claims the issue named the branch, and a claim with no evidence behind it
209
+ // is `default` wearing a hat.
210
+ const evidence =
211
+ state.baseBranchEvidence && typeof state.baseBranchEvidence === "object"
212
+ ? state.baseBranchEvidence
213
+ : null;
214
+ const DEGRADED_FETCH = ["cached-stale", "local-branch"];
215
+ if (DEGRADED_FETCH.includes(fetchStatus)) {
216
+ if (!evidence) {
217
+ failures.push(
218
+ `baseFetchStatus="${fetchStatus}" but state.baseBranchEvidence is absent. A run whose ` +
219
+ `fetch failed listed its branches from local refs; the record of that is what stops a ` +
220
+ `local-only guess being read afterwards as the remote's answer.`,
221
+ );
222
+ } else if (evidence.refProvenance !== "local") {
223
+ failures.push(
224
+ `baseFetchStatus="${fetchStatus}" but baseBranchEvidence.refProvenance=` +
225
+ `"${evidence.refProvenance || "<unset>"}". The fetch failed, so the candidate list came ` +
226
+ `from local refs and must say so - "remote" here is the silent degradation this field exists to catch.`,
227
+ );
228
+ }
229
+ }
230
+ if (branchSource === "derived") {
231
+ const ISSUE_DERIVED = new Set(["issue-version", "linked-release"]);
232
+ const cands = evidence && Array.isArray(evidence.candidates) ? evidence.candidates : [];
233
+ const chosen = cands.find((c) => c && c.branch === baseBranch);
234
+ const hasIssueEvidence =
235
+ chosen &&
236
+ Array.isArray(chosen.evidence) &&
237
+ chosen.evidence.some((e) => e && ISSUE_DERIVED.has(e.kind));
238
+ if (!hasIssueEvidence) {
239
+ failures.push(
240
+ `baseBranchSource="derived" but baseBranchEvidence carries no issue-version or ` +
241
+ `linked-release evidence for "${baseBranch}". "Derived" names a specific claim - the ` +
242
+ `issue's version field or a linked release issue pointed at this branch - and without ` +
243
+ `that record it is the sort-order default under a better name.`,
244
+ );
245
+ }
246
+ if (evidence && evidence.ambiguous === true) {
247
+ failures.push(
248
+ `baseBranchSource="derived" with baseBranchEvidence.ambiguous=true. Two or more ` +
249
+ `candidates tied at the top score; autopilot does not break a tie by picking one.`,
250
+ );
251
+ }
252
+ }
253
+
254
+ // Step 5b decides where the branch lives, and `localMode` alone cannot say
255
+ // whether anyone decided: `false` is both "the user chose a worktree" and
256
+ // "nothing asked and the default stood". Same shape as baseBranchSource, and
257
+ // the same reason - an autopilot run resolves it rather than asking, so
258
+ // `autopilot` is a legal source there and nowhere else.
259
+ const WORKSPACE_SOURCES = ["asked", "command", "autopilot"];
260
+ const workspaceSource =
261
+ typeof state.workspaceSource === "string" ? state.workspaceSource.trim() : "";
262
+ if (!WORKSPACE_SOURCES.includes(workspaceSource)) {
263
+ failures.push(
264
+ `agent-state.json has workspaceSource="${workspaceSource || "<unset>"}"; Step 5b must ` +
265
+ `record one of ${WORKSPACE_SOURCES.join(" | ")}. Unset means nothing distinguishes a ` +
266
+ `worktree the user chose from one no question was asked about.`,
267
+ );
268
+ }
269
+ if (!isAutopilot && workspaceSource === "autopilot") {
270
+ failures.push(
271
+ `workspaceSource="autopilot" on an interactive run. Autopilot resolves the workspace ` +
272
+ `to a worktree because an unattended commit in the user's own checkout is what ` +
273
+ `worktrees prevent; an interactive run asks (Step 5b) or is told by :local / --local.`,
274
+ );
275
+ }
276
+
277
+ // Step 2b's dev-context picker writes siblings[], empty included. The empty array is
278
+ // the record that it ran; absent, a multi-repo task silently became a single-repo one
279
+ // and Phase 4's parity cross-check lost its fourth counterpart source.
280
+ if (!Array.isArray(state.siblings)) {
281
+ failures.push(
282
+ "agent-state.json has no siblings array. Step 2b runs the dev-context picker on " +
283
+ "every input type and persists the result, `[]` included; an absent field means " +
284
+ "the picker never ran, so extra repos and read-only counterparts were never offered.",
285
+ );
286
+ }
287
+
160
288
  // Worktree isolation is a standing rule: never develop in the primary checkout.
161
289
  const worktree = typeof state.worktreePath === "string" ? state.worktreePath.trim() : "";
162
290
  const projectRoot = typeof state.projectRoot === "string" ? state.projectRoot.trim() : "";
@@ -0,0 +1,180 @@
1
+ #!/usr/bin/env node
2
+
3
+ /**
4
+ * @file review-file-filter.mjs - decide what the reviewers are asked to read.
5
+ *
6
+ * Phase 4 has a size cap and no exclusion list. When the diff exceeds the
7
+ * budget the cap truncates the LARGEST files first (`phase-4-review.md`, Step
8
+ * 1.9), so a regenerated lockfile or a snapshot dump does not merely waste
9
+ * tokens - it is the thing that survives while real code is cut. The cheapest
10
+ * fix is to decide what is worth reading before the cap decides what fits.
11
+ *
12
+ * Every exclusion carries the reason it was excluded, and the two lists
13
+ * partition the input exactly. A file that quietly disappears between the diff
14
+ * and the reviewer is indistinguishable from a file nobody found anything in,
15
+ * and that is the failure this script exists to prevent.
16
+ *
17
+ * Patterns live in `schemas/review-file-exclusions.json` so the list is data,
18
+ * versioned beside the code that reads it, and generic: no stack, project or
19
+ * company name appears in it.
20
+ *
21
+ * Glob matching comes from `glob-match.mjs`, the same module
22
+ * `skill-conformance.mjs` uses. Two matchers would drift, and the day they
23
+ * disagree a file is excluded from the review and still counted in the
24
+ * conformance denominator, with nothing saying so.
25
+ *
26
+ * Inputs:
27
+ * (stdin) One path per line. `git diff --name-only` output.
28
+ * --files <path> Read the path list from a file instead of stdin
29
+ * --patterns <path> Override the pattern file
30
+ * --json Emit the full report (default)
31
+ * --reviewed Print only the reviewed paths, one per line
32
+ *
33
+ * Exit codes:
34
+ * 0 - filtered
35
+ * 2 - the pattern list could not be read. The report still comes out with
36
+ * EVERY file reviewed, because the safe direction is to read too much:
37
+ * a filter that fails closed would silently review nothing.
38
+ * 64 - usage error
39
+ *
40
+ * @module pipeline/scripts/review-file-filter
41
+ */
42
+
43
+ import { readFileSync } from "node:fs";
44
+ import { join } from "node:path";
45
+ import { globToRegExp } from "./glob-match.mjs";
46
+ import { unquotePathIfNeeded } from "./git-path.mjs";
47
+
48
+ const DEFAULT_PATTERNS = join(import.meta.dirname, "..", "schemas", "review-file-exclusions.json");
49
+
50
+ /**
51
+ * Read and validate the pattern list.
52
+ *
53
+ * A pattern with no reason is rejected rather than defaulted, because the
54
+ * default would be the thing the caller is supposed to print.
55
+ *
56
+ * @param {string} path
57
+ * @returns {{patterns: {glob: string, reason: string, re: RegExp}[], error: string|null}}
58
+ */
59
+ export function loadPatterns(path) {
60
+ let doc;
61
+ try {
62
+ doc = JSON.parse(readFileSync(path, "utf-8"));
63
+ } catch (e) {
64
+ return { patterns: [], error: `cannot read ${path}: ${e.message}` };
65
+ }
66
+ const rows = Array.isArray(doc?.patterns) ? doc.patterns : null;
67
+ if (!rows) return { patterns: [], error: `${path} has no patterns[] array` };
68
+
69
+ const patterns = [];
70
+ for (const [i, r] of rows.entries()) {
71
+ const glob = typeof r?.glob === "string" ? r.glob.trim() : "";
72
+ const reason = typeof r?.reason === "string" ? r.reason.trim() : "";
73
+ if (!glob) return { patterns: [], error: `patterns[${i}] has no glob` };
74
+ if (!reason) return { patterns: [], error: `patterns[${i}] (${glob}) has no reason` };
75
+ patterns.push({ glob, reason, re: globToRegExp(glob) });
76
+ }
77
+ return { patterns, error: null };
78
+ }
79
+
80
+ /**
81
+ * Split the changed files into what the reviewers read and what they do not.
82
+ *
83
+ * @param {string[]} files
84
+ * @param {{glob: string, reason: string, re: RegExp}[]} patterns
85
+ * @returns {{reviewed: string[], excluded: {path: string, reason: string, pattern: string}[]}}
86
+ */
87
+ export function partition(files, patterns) {
88
+ const reviewed = [];
89
+ const excluded = [];
90
+ for (const f of files) {
91
+ const hit = patterns.find((p) => p.re.test(f));
92
+ if (hit) excluded.push({ path: f, reason: hit.reason, pattern: hit.glob });
93
+ else reviewed.push(f);
94
+ }
95
+ return { reviewed, excluded };
96
+ }
97
+
98
+ /**
99
+ * @param {string} raw
100
+ * @returns {string[]}
101
+ */
102
+ export function readPaths(raw) {
103
+ const seen = new Set();
104
+ const out = [];
105
+ for (const line of raw.split("\n")) {
106
+ // Only the line ending is stripped: a leading or trailing space is a real
107
+ // character in a path. The C-style quoting git applies to a non-ASCII name
108
+ // IS undone, because a quoted `"G\303\266..."` matches no glob and would
109
+ // drop the file out of the denominator without anything reporting it -
110
+ // exactly the bug diff-risk-score.mjs already carries a fix for.
111
+ const p = unquotePathIfNeeded(line.replace(/\r$/, ""));
112
+ if (!p || seen.has(p)) continue;
113
+ seen.add(p);
114
+ out.push(p);
115
+ }
116
+ return out;
117
+ }
118
+
119
+ function main() {
120
+ const argv = process.argv.slice(2);
121
+ // A value flag with no value is a usage error, not a silent default. Falling
122
+ // back would hand a caller who typo'd `--patterns` the SHIPPED list while
123
+ // they believed they were running their own.
124
+ const valueOf = (flag) => {
125
+ const i = argv.indexOf(flag);
126
+ if (i === -1) return null;
127
+ const v = argv[i + 1];
128
+ if (v === undefined || v.startsWith("--")) {
129
+ console.error(`review-file-filter: ${flag} needs a path`);
130
+ process.exit(64);
131
+ }
132
+ return v;
133
+ };
134
+ for (const a of argv) {
135
+ if (a.startsWith("--") && !["--files", "--patterns", "--json", "--reviewed"].includes(a)) {
136
+ console.error(`review-file-filter: unknown flag ${a}`);
137
+ process.exit(64);
138
+ }
139
+ }
140
+
141
+ const filesArg = valueOf("--files");
142
+ let raw;
143
+ try {
144
+ raw = filesArg ? readFileSync(filesArg, "utf-8") : readFileSync(0, "utf-8");
145
+ } catch (e) {
146
+ console.error(`review-file-filter: cannot read the file list: ${e.message}`);
147
+ process.exit(64);
148
+ }
149
+ const files = readPaths(raw);
150
+
151
+ const patternsPath = valueOf("--patterns") || DEFAULT_PATTERNS;
152
+ const { patterns, error } = loadPatterns(patternsPath);
153
+ const { reviewed, excluded } = error
154
+ ? { reviewed: files.slice(), excluded: [] }
155
+ : partition(files, patterns);
156
+
157
+ const report = {
158
+ reviewed,
159
+ excluded,
160
+ total: files.length,
161
+ patternsSource: patternsPath,
162
+ patternCount: patterns.length,
163
+ ...(error ? { patternsError: error } : {}),
164
+ };
165
+
166
+ if (argv.includes("--reviewed")) {
167
+ for (const f of reviewed) console.log(f);
168
+ } else {
169
+ console.log(JSON.stringify(report, null, 2));
170
+ }
171
+
172
+ if (error) {
173
+ console.error(`review-file-filter: ${error} - reviewing all ${files.length} file(s)`);
174
+ process.exit(2);
175
+ }
176
+ }
177
+
178
+ if (import.meta.url === `file://${process.argv[1]}`) {
179
+ main();
180
+ }
@@ -44,6 +44,7 @@ import { existsSync, readFileSync, writeFileSync, readdirSync, statSync } from "
44
44
  import { dirname, join, resolve, relative, basename, extname, sep } from "node:path";
45
45
  import { fileURLToPath } from "node:url";
46
46
  import { homedir } from "node:os";
47
+ import { matchesAnyGlob } from "./glob-match.mjs";
47
48
 
48
49
  const here = dirname(fileURLToPath(import.meta.url));
49
50
 
@@ -364,37 +365,6 @@ function classifyLanguage(file) {
364
365
  return LANG_BY_EXT[extname(base).toLowerCase()] ?? "other";
365
366
  }
366
367
 
367
- // ---------------------------------------------------------------------------
368
- // Minimal glob matcher for the `**/x`, `*.ext`, `dir/**` shapes registries use.
369
- // ---------------------------------------------------------------------------
370
- function globToRegExp(glob) {
371
- let re = "";
372
- for (let i = 0; i < glob.length; i++) {
373
- const c = glob[i];
374
- if (c === "*") {
375
- if (glob[i + 1] === "*") {
376
- // `**/` matches zero or more path segments.
377
- if (glob[i + 2] === "/") {
378
- re += "(?:.*/)?";
379
- i += 2;
380
- } else {
381
- re += ".*";
382
- i += 1;
383
- }
384
- } else {
385
- re += "[^/]*";
386
- }
387
- } else if (c === "?") re += "[^/]";
388
- else re += c.replace(/[.+^${}()|[\]\\]/g, "\\$&");
389
- }
390
- return new RegExp(`^${re}$`);
391
- }
392
-
393
- function matchesAnyGlob(file, globs) {
394
- if (!globs || globs.length === 0) return true;
395
- return globs.some((g) => globToRegExp(g).test(file));
396
- }
397
-
398
368
  // ---------------------------------------------------------------------------
399
369
  // Diff parsing: changed files and added lines (for the exception audit).
400
370
  // ---------------------------------------------------------------------------
@@ -635,6 +635,59 @@ function main() {
635
635
  );
636
636
  }
637
637
 
638
+ // 2d-2. The diagram has to describe the document it sits in.
639
+ //
640
+ // Until now the only question asked about Section 3 was "is a mermaid block
641
+ // present". A sequence diagram could name a participant the document never
642
+ // mentions, or omit a service the document devotes a section to, and nothing
643
+ // said so - the diagram is the part a reader trusts most and the part nothing
644
+ // checked. Both directions are reported, because they are different mistakes:
645
+ // an invented participant is a claim the document does not support, and a
646
+ // missing one is a service the reader will not see in the picture.
647
+ //
648
+ // Warn, not error: a participant can legitimately be an actor ("User") or an
649
+ // external system named nowhere else, and failing a correct document is how a
650
+ // check gets switched off. The warning names the label so it is one grep to
651
+ // settle.
652
+ const participants = [];
653
+ let inMermaid = false;
654
+ for (const raw of text.split("\n")) {
655
+ const t = raw.trim();
656
+ if (/^```/.test(t)) {
657
+ inMermaid = /^```mermaid\b/.test(t);
658
+ continue;
659
+ }
660
+ if (!inMermaid) continue;
661
+ const m = t.match(/^(?:participant|actor)\s+(\S+)(?:\s+as\s+(.+))?$/);
662
+ if (m) participants.push((m[2] || m[1]).trim());
663
+ }
664
+
665
+ if (participants.length > 0) {
666
+ // The prose is everything outside the fences: a participant that only ever
667
+ // appears inside the diagram has no support in the document.
668
+ const prose = text
669
+ .split(/^```[\s\S]*?^```/gm)
670
+ .join("\n")
671
+ .toLowerCase();
672
+ // Only identifier-shaped labels are evidence. A human actor is an ordinary
673
+ // word in whatever language the document is written in - "User",
674
+ // "Kullanici", "Operator" - and its absence from the prose says nothing. A
675
+ // service or component is written like code: an internal capital
676
+ // (PaymentService) or a separator (payment-service). Testing the SHAPE
677
+ // rather than keeping a word list keeps this working in both languages and
678
+ // stops the list from rotting.
679
+ const looksLikeIdentifier = (x) => /[a-z][A-Z]/.test(x) || (/[._-]/.test(x) && x.length >= 6);
680
+ const orphans = participants.filter((p) => {
681
+ const raw = p.replace(/^["'`]|["'`]$/g, "");
682
+ return looksLikeIdentifier(raw) && !prose.includes(raw.toLowerCase());
683
+ });
684
+ if (orphans.length > 0) {
685
+ warns.push(
686
+ `Section 3 diagram names ${orphans.length} participant(s) the document never mentions: ${orphans.join(", ")}`,
687
+ );
688
+ }
689
+ }
690
+
638
691
  mark("flow-chart");
639
692
 
640
693
  // 2e. Locked 16: every Files-to-Add row is tagged. The decision says untagged
@@ -11,6 +11,7 @@
11
11
  // node validate-reviewer.mjs path/to/reviewer.json
12
12
  // cat reviewer.json | node validate-reviewer.mjs -
13
13
  // [--criteria path/to/criteria-manifest.json] # enforce the rule-ID checklist
14
+ // [--coverage path/to/review-files.json] # enforce the file denominator
14
15
  //
15
16
  // With --criteria, the validator also enforces the conformance contract added in
16
17
  // schema v1.1.0. Without that enforcement the field is decoration: this validator
@@ -29,13 +30,16 @@ import { readFileSync } from "node:fs";
29
30
 
30
31
  const ALLOWED_SEVERITIES = new Set(["blocking", "important", "suggestion"]);
31
32
  const ALLOWED_VERDICTS = new Set(["conformant", "violated", "not-applicable"]);
33
+ const ALLOWED_COVERAGE = new Set(["reviewed", "skipped"]);
32
34
  const FINGERPRINT_RE = /^F:[0-9a-f]{8}$/;
33
35
 
34
36
  /** @returns {Promise<string>} */
35
37
  function readInput() {
36
38
  const arg = process.argv[2];
37
39
  if (!arg) {
38
- console.error("usage: validate-reviewer.mjs <path|-> [--criteria <manifest>]");
40
+ console.error(
41
+ "usage: validate-reviewer.mjs <path|-> [--criteria <manifest>] [--coverage <report>]",
42
+ );
39
43
  process.exit(64);
40
44
  }
41
45
  if (arg === "-") {
@@ -81,6 +85,72 @@ function validateFinding(f, label, errors) {
81
85
  }
82
86
  }
83
87
 
88
+ /** Files the reviewer was given. Empty set = no denominator supplied. */
89
+ function loadReviewedFiles(path) {
90
+ const report = JSON.parse(readFileSync(path, "utf-8"));
91
+ return new Set((report?.reviewed ?? []).filter((f) => typeof f === "string" && f.length > 0));
92
+ }
93
+
94
+ /**
95
+ * Enforce the file checklist against the fixed denominator.
96
+ *
97
+ * The same three failures the conformance checklist has, on the other axis:
98
+ * a file with no row (silently unread), a row for a file outside the set
99
+ * (a reviewer answering about something it was not given), and `skipped`
100
+ * with no reason. Without this, a reviewer that opened one file of ten and a
101
+ * reviewer that found nothing in all ten return the identical empty findings[].
102
+ */
103
+ function validateFileCoverage(parsed, reviewedFiles, errors) {
104
+ const rows = parsed.fileCoverage;
105
+ if (!Array.isArray(rows)) {
106
+ errors.push(
107
+ `fileCoverage must be an array with one row per file in the review set (${reviewedFiles.size} expected)`,
108
+ );
109
+ return;
110
+ }
111
+
112
+ const seen = new Map();
113
+ rows.forEach((row, i) => {
114
+ const label = `fileCoverage[${i}]`;
115
+ if (typeof row !== "object" || row === null) {
116
+ errors.push(`${label}: not an object`);
117
+ return;
118
+ }
119
+ const path = row.path;
120
+ if (typeof path !== "string" || path.length === 0) {
121
+ errors.push(`${label}: missing path`);
122
+ return;
123
+ }
124
+ if (!ALLOWED_COVERAGE.has(row.verdict)) {
125
+ errors.push(`${label} (${path}): bad verdict ${JSON.stringify(row.verdict)}`);
126
+ }
127
+ if (!reviewedFiles.has(path)) {
128
+ errors.push(
129
+ `${label}: "${path}" is not in the review set - answer only for the files you were given`,
130
+ );
131
+ }
132
+ seen.set(path, (seen.get(path) ?? 0) + 1);
133
+
134
+ if (row.verdict === "skipped" && (typeof row.reason !== "string" || row.reason.length < 4)) {
135
+ errors.push(
136
+ `${label} (${path}): verdict "skipped" needs a reason - a file dropped without one is indistinguishable from one that was read`,
137
+ );
138
+ }
139
+ });
140
+
141
+ for (const [path, count] of seen) {
142
+ if (count > 1)
143
+ errors.push(`fileCoverage: "${path}" appears ${count} times (expected exactly once)`);
144
+ }
145
+ const missing = [...reviewedFiles].filter((f) => !seen.has(f));
146
+ if (missing.length > 0) {
147
+ const shown = missing.slice(0, 8).join(", ");
148
+ errors.push(
149
+ `fileCoverage: ${missing.length} file(s) in the review set have no row (${shown}${missing.length > 8 ? ", ..." : ""}) - neither read nor declared skipped`,
150
+ );
151
+ }
152
+ }
153
+
84
154
  /** Rule IDs the reviewer was told to answer for. Empty set = no criteria supplied. */
85
155
  function loadSelectedRuleIds(path) {
86
156
  const manifest = JSON.parse(readFileSync(path, "utf-8"));
@@ -214,6 +284,25 @@ async function main() {
214
284
  }
215
285
  }
216
286
 
287
+ // File checklist, only when the orchestrator supplied a denominator.
288
+ const covIdx = process.argv.indexOf("--coverage");
289
+ if (covIdx !== -1) {
290
+ const covPath = process.argv[covIdx + 1];
291
+ if (!covPath) {
292
+ errors.push("--coverage given with no report path");
293
+ } else {
294
+ try {
295
+ const reviewedFiles = loadReviewedFiles(covPath);
296
+ // Everything excluded means there is nothing to answer for. Demanding an
297
+ // empty array would fail honest output, and the filter's excluded[] is
298
+ // what carries that information onward.
299
+ if (reviewedFiles.size > 0) validateFileCoverage(parsed, reviewedFiles, errors);
300
+ } catch (err) {
301
+ errors.push(`cannot read review-set report: ${err.message}`);
302
+ }
303
+ }
304
+ }
305
+
217
306
  // Internal consistency: approved=true AND blocking findings = contradiction
218
307
  if (
219
308
  errors.length === 0 &&