@mmerterden/multi-agent-pipeline 20.8.3 → 20.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/CHANGELOG.md +37 -0
  2. package/docs/facts.json +1 -1
  3. package/install/claude.mjs +1 -1
  4. package/manifest.json +37 -28
  5. package/package.json +1 -1
  6. package/pipeline/lib/claude-md-links.mjs +328 -0
  7. package/pipeline/lib/owned-path-gate.mjs +699 -0
  8. package/pipeline/lib/repo-profile-derive.mjs +1771 -0
  9. package/pipeline/lib/repo-profile.mjs +780 -0
  10. package/pipeline/lib/stack-detect.sh +59 -19
  11. package/pipeline/lib/unattended.mjs +17 -0
  12. package/pipeline/multi-agent-refs/features/repo-profile.md +96 -0
  13. package/pipeline/multi-agent-refs/features/review-decision.md +18 -13
  14. package/pipeline/multi-agent-refs/features/stack-skill-routing.md +179 -33
  15. package/pipeline/multi-agent-refs/outside-the-pipeline.md +33 -11
  16. package/pipeline/multi-agent-refs/phases/phase-1-plan.md +26 -12
  17. package/pipeline/multi-agent-refs/phases/phase-2-dev.md +24 -13
  18. package/pipeline/multi-agent-refs/phases/phase-3-review.md +16 -4
  19. package/pipeline/multi-agent-refs/phases/phase-4-commit.md +1 -1
  20. package/pipeline/multi-agent-refs/phases/phase-5-report.md +8 -0
  21. package/pipeline/rules/outside-the-pipeline.md +6 -1
  22. package/pipeline/schemas/agent-state.schema.json +66 -2
  23. package/pipeline/schemas/phases.json +4 -4
  24. package/pipeline/schemas/repo-profile.schema.json +1107 -0
  25. package/pipeline/schemas/token-budget.json +4 -4
  26. package/pipeline/scripts/agent-guard.py +30 -0
  27. package/pipeline/scripts/owned-path-gate.mjs +205 -0
  28. package/pipeline/scripts/pre-commit-check.sh +151 -1
  29. package/pipeline/scripts/repo-profile.mjs +244 -0
  30. package/pipeline/scripts/review-decision-gate.mjs +42 -18
  31. package/pipeline/scripts/skill-candidates.mjs +882 -0
  32. package/pipeline/scripts/unattended_policy.py +90 -0
  33. package/pipeline/scripts/usage-report.mjs +36 -6
  34. package/pipeline/skills/.skill-manifest.json +1 -1
@@ -0,0 +1,328 @@
1
+ #!/usr/bin/env node
2
+ /**
3
+ * claude-md-links.mjs - the in-repo documents a repo's CLAUDE.md points at.
4
+ *
5
+ * A project CLAUDE.md is often an index: "naming rules are in docs/style.md".
6
+ * Phase 1 treats those linked files as project rules, so it needs them, and it
7
+ * needs them bounded. Three forms count as a reference: a markdown link, a
8
+ * document path written as inline code (`` See `.claude/RULES.md` ``), and a
9
+ * host import line (`@docs/rules.md`). A code path is resolved against the
10
+ * CLAUDE.md's own directory first, then the repo root; a link or import against
11
+ * the file's own directory; a leading `/` means the repo root when its first
12
+ * segment exists there, and is otherwise an absolute filesystem path, which is
13
+ * rejected like `~/`.
14
+ *
15
+ * One hop only (links inside a linked file are not followed), at most
16
+ * MAX_FILES files and MAX_BYTES bytes in total. References on a line that calls
17
+ * the target authoritative ("See X for ...", "X is authoritative", "source of
18
+ * truth") are read first, so the caps drop background reading before rules. The
19
+ * file that crosses the byte budget is truncated at a character boundary and
20
+ * reported in `truncated[]`, not dropped. SKILL.md files are skipped: skills are
21
+ * routed by scripts/skill-candidates.mjs. Nothing that resolves outside the repo
22
+ * root is read - checked on the written path and on the real path, for the
23
+ * CLAUDE.md files themselves too, because a symlink inside the repo can point
24
+ * anywhere.
25
+ *
26
+ * Usage:
27
+ * node claude-md-links.mjs <repo-root> [--no-content]
28
+ *
29
+ * Prints JSON: {files:[{path, from, kind, priority, bytes, content, truncated?,
30
+ * fullBytes?}], rejected:[{link, from, kind, reason}], truncated:[path],
31
+ * totalBytes, sources}. `kind` is "link", "code-path", "import" or, for a
32
+ * rejected CLAUDE.md, "source". Exit 0 always when the root is readable; an
33
+ * absent CLAUDE.md is an empty result, not an error. Exit 2 on usage.
34
+ *
35
+ * @module pipeline/lib/claude-md-links
36
+ */
37
+
38
+ import { existsSync, lstatSync, readFileSync, realpathSync, statSync } from "node:fs";
39
+ import { basename, dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
40
+
41
+ import { invokedDirectly } from "./invoked-directly.mjs";
42
+
43
+ export const SOURCES = ["CLAUDE.md", ".claude/CLAUDE.md"];
44
+ export const MAX_FILES = 10;
45
+ export const MAX_BYTES = 200 * 1024;
46
+
47
+ const SCHEME = /^[a-z][a-z0-9+.-]*:/i;
48
+ const AUTHORITATIVE =
49
+ /\bauthoritative\b|\bsource of truth\b|\bcanonical\b|\bmust (?:read|follow)\b|\bsee\b[^\n]*?\bfor\b|\brules? (?:are|live) in\b/i;
50
+
51
+ function blank(m) {
52
+ return m.replace(/[^\n]/g, " ");
53
+ }
54
+
55
+ function stripFences(md) {
56
+ return md.replace(/^(```|~~~)[\s\S]*?^\1[^\n]*$/gm, blank);
57
+ }
58
+
59
+ function stripCode(md) {
60
+ return stripFences(md).replace(/`[^`\n]*`/g, blank);
61
+ }
62
+
63
+ function cleanTarget(raw) {
64
+ let t = raw.trim().replace(/^<|>$/g, "");
65
+ if (t.startsWith("#")) return null;
66
+ t = t.replace(/[#?].*$/, "");
67
+ if (!t) return null;
68
+ try {
69
+ t = decodeURI(t);
70
+ } catch {
71
+ /* keep the undecoded form */
72
+ }
73
+ return t;
74
+ }
75
+
76
+ function lineAt(text, index) {
77
+ const start = text.lastIndexOf("\n", index - 1) + 1;
78
+ const end = text.indexOf("\n", index);
79
+ return text.slice(start, end < 0 ? text.length : end);
80
+ }
81
+
82
+ function linkHits(md) {
83
+ const text = stripCode(String(md ?? ""));
84
+ const out = [];
85
+ const inline = /!?\[[^\]\n]*\]\(\s*(<[^>\n]*>|[^\s)]+)(?:\s+(?:"[^"\n]*"|'[^'\n]*'))?\s*\)/g;
86
+ const reference = /^[ \t]{0,3}\[[^\]\n]+\]:[ \t]*(<[^>\n]*>|\S+)/gm;
87
+ for (const re of [inline, reference]) {
88
+ for (const m of text.matchAll(re)) {
89
+ const t = cleanTarget(m[1]);
90
+ if (t) out.push({ at: m.index, t });
91
+ }
92
+ }
93
+ return out.sort((a, b) => a.at - b.at);
94
+ }
95
+
96
+ /**
97
+ * Link targets in a markdown document, in document order, fragments and titles
98
+ * removed. Anchor-only links and anything inside code are skipped.
99
+ * @param {string} md
100
+ * @returns {string[]}
101
+ */
102
+ export function extractLinks(md) {
103
+ return linkHits(md).map((x) => x.t);
104
+ }
105
+
106
+ const DOC_PATH =
107
+ /^(?:\.{0,2}\/)?[A-Za-z0-9._-]+(?:\/[A-Za-z0-9._-]+)*\.(?:md|mdx|markdown|txt|rst)$/i;
108
+
109
+ function codePathHits(md) {
110
+ const text = stripFences(String(md ?? ""));
111
+ const out = [];
112
+ for (const m of text.matchAll(/`([^`\n]+)`/g)) {
113
+ const t = m[1].trim();
114
+ if (DOC_PATH.test(t)) out.push({ at: m.index, t });
115
+ }
116
+ return out;
117
+ }
118
+
119
+ /**
120
+ * Document paths written as inline code, in document order. Only spans that are
121
+ * nothing but a relative-looking path to a text document count, so a command,
122
+ * a type name or a glob is never taken for a reference.
123
+ * @param {string} md
124
+ * @returns {string[]}
125
+ */
126
+ export function extractCodePaths(md) {
127
+ return codePathHits(md).map((x) => x.t);
128
+ }
129
+
130
+ function importHits(md) {
131
+ const text = stripCode(String(md ?? ""));
132
+ const out = [];
133
+ const re = /(^|[\s(])@((?:~|\.{1,2})?\/?[A-Za-z0-9._-]+(?:\/[A-Za-z0-9._-]+)*)/g;
134
+ for (const m of text.matchAll(re)) {
135
+ const t = m[2].replace(/[.]+$/, "");
136
+ if (/\.[A-Za-z0-9]+$/.test(basename(t))) out.push({ at: m.index + m[1].length, t });
137
+ }
138
+ return out;
139
+ }
140
+
141
+ /**
142
+ * Host import lines (`@path/to/file.md`) outside code, in document order. An
143
+ * `@` inside a word (an e-mail address) is not an import, and a target needs a
144
+ * file extension, so a mention or a package scope is not taken for one.
145
+ * @param {string} md
146
+ * @returns {string[]}
147
+ */
148
+ export function extractImports(md) {
149
+ return importHits(md).map((x) => x.t);
150
+ }
151
+
152
+ function inside(root, p) {
153
+ const rel = relative(root, p);
154
+ return rel === "" || (!rel.startsWith(`..${sep}`) && rel !== ".." && !isAbsolute(rel));
155
+ }
156
+
157
+ function toPosix(p) {
158
+ return p.split(sep).join("/");
159
+ }
160
+
161
+ function locate(root, ref) {
162
+ const { link } = ref;
163
+ if (SCHEME.test(link) || link.startsWith("//")) return { reason: "url" };
164
+ if (link.startsWith("~")) return { reason: "absolute path" };
165
+ let bases = ref.bases;
166
+ let rel = link;
167
+ if (isAbsolute(link)) {
168
+ const first = link.split("/").find(Boolean);
169
+ if (!first || !existsSync(join(root, first))) return { reason: "absolute path" };
170
+ bases = [root];
171
+ rel = link.replace(/^\/+/, "");
172
+ }
173
+ let sawInside = false;
174
+ for (const b of bases) {
175
+ const target = resolve(b, rel);
176
+ if (!inside(root, target)) continue;
177
+ sawInside = true;
178
+ try {
179
+ lstatSync(target);
180
+ return { target };
181
+ } catch {
182
+ /* try the next base */
183
+ }
184
+ }
185
+ return { reason: sawInside ? "missing" : "outside repo" };
186
+ }
187
+
188
+ /** Largest prefix of `buf` no longer than `limit` that ends on a UTF-8 character boundary. */
189
+ function utf8Cut(buf, limit) {
190
+ let end = Math.min(limit, buf.length);
191
+ while (end > 0 && end < buf.length && (buf[end] & 0xc0) === 0x80) end--;
192
+ return end;
193
+ }
194
+
195
+ /**
196
+ * One reference: a file entry, a rejection reason, or null for a silent skip
197
+ * (a duplicate, or one of the CLAUDE.md files themselves).
198
+ */
199
+ function admit(root, ref, seen, result, { maxFiles, maxBytes, withContent }) {
200
+ const found = locate(root, ref);
201
+ if (found.reason) return found.reason;
202
+ let real;
203
+ try {
204
+ real = realpathSync(found.target);
205
+ } catch {
206
+ return "missing";
207
+ }
208
+ if (!inside(root, real)) return "outside repo";
209
+ if (seen.has(real) || SOURCES.some((s) => join(root, s) === real)) return null;
210
+ if (basename(found.target) === "SKILL.md" || basename(real) === "SKILL.md") return "skill file";
211
+ let st;
212
+ try {
213
+ st = statSync(real);
214
+ } catch {
215
+ return "missing";
216
+ }
217
+ if (!st.isFile()) return "not a regular file";
218
+ if (result.files.length >= maxFiles) return "file cap";
219
+ const room = maxBytes - result.totalBytes;
220
+ if (room <= 0) return "byte budget";
221
+ let buf;
222
+ try {
223
+ buf = readFileSync(real);
224
+ } catch {
225
+ return "unreadable";
226
+ }
227
+ if (buf.includes(0)) return "binary";
228
+ const end = buf.length > room ? utf8Cut(buf, room) : buf.length;
229
+ if (end === 0) return "byte budget";
230
+ seen.add(real);
231
+ result.totalBytes += end;
232
+ const path = toPosix(relative(root, real));
233
+ const entry = {
234
+ path,
235
+ from: null,
236
+ kind: ref.kind,
237
+ priority: ref.priority,
238
+ bytes: end,
239
+ };
240
+ if (end < buf.length) {
241
+ entry.truncated = true;
242
+ entry.fullBytes = buf.length;
243
+ result.truncated.push(path);
244
+ }
245
+ if (withContent) entry.content = buf.subarray(0, end).toString("utf8");
246
+ return entry;
247
+ }
248
+
249
+ function refsOf(text, src, base, root) {
250
+ const tag = (kind, bases) => (h) => ({
251
+ link: h.t,
252
+ kind,
253
+ bases,
254
+ from: src,
255
+ at: h.at,
256
+ priority: AUTHORITATIVE.test(lineAt(text, h.at)) ? "authoritative" : "normal",
257
+ });
258
+ return [
259
+ ...linkHits(text).map(tag("link", [base])),
260
+ ...codePathHits(text).map(tag("code-path", base === root ? [base] : [base, root])),
261
+ ...importHits(text).map(tag("import", [base])),
262
+ ].sort((a, b) => a.at - b.at);
263
+ }
264
+
265
+ /**
266
+ * Follow relative links from the repo's CLAUDE.md files, one hop, bounded.
267
+ * @param {string} repoRoot
268
+ * @param {{maxFiles?: number, maxBytes?: number, content?: boolean}} [opts]
269
+ */
270
+ export function followClaudeMdLinks(repoRoot, opts = {}) {
271
+ const maxFiles = opts.maxFiles ?? MAX_FILES;
272
+ const maxBytes = opts.maxBytes ?? MAX_BYTES;
273
+ const withContent = opts.content !== false;
274
+ const result = { files: [], rejected: [], truncated: [], totalBytes: 0, sources: [] };
275
+
276
+ let root;
277
+ try {
278
+ root = realpathSync(resolve(repoRoot));
279
+ } catch {
280
+ return result;
281
+ }
282
+ const seen = new Set();
283
+ const refs = [];
284
+
285
+ for (const src of SOURCES) {
286
+ const srcPath = join(root, src);
287
+ let text;
288
+ try {
289
+ if (!statSync(srcPath).isFile()) continue;
290
+ if (!inside(root, realpathSync(srcPath))) {
291
+ result.rejected.push({ link: src, from: src, kind: "source", reason: "outside repo" });
292
+ continue;
293
+ }
294
+ text = readFileSync(srcPath, "utf8");
295
+ } catch {
296
+ continue;
297
+ }
298
+ result.sources.push(src);
299
+ refs.push(...refsOf(text, src, dirname(srcPath), root));
300
+ }
301
+
302
+ const ordered = [
303
+ ...refs.filter((r) => r.priority === "authoritative"),
304
+ ...refs.filter((r) => r.priority !== "authoritative"),
305
+ ];
306
+ for (const ref of ordered) {
307
+ const verdict = admit(root, ref, seen, result, { maxFiles, maxBytes, withContent });
308
+ if (verdict === null) continue;
309
+ if (typeof verdict === "string") {
310
+ result.rejected.push({ link: ref.link, from: ref.from, kind: ref.kind, reason: verdict });
311
+ } else {
312
+ verdict.from = ref.from;
313
+ result.files.push(verdict);
314
+ }
315
+ }
316
+ return result;
317
+ }
318
+
319
+ if (invokedDirectly(import.meta.url)) {
320
+ const args = process.argv.slice(2);
321
+ const dir = args.find((a) => !a.startsWith("--"));
322
+ if (!dir) {
323
+ console.error("usage: claude-md-links.mjs <repo-root> [--no-content]");
324
+ process.exit(2);
325
+ }
326
+ const out = followClaudeMdLinks(dir, { content: !args.includes("--no-content") });
327
+ process.stdout.write(`${JSON.stringify(out, null, 2)}\n`);
328
+ }