@mmerterden/multi-agent-pipeline 20.8.3 → 20.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +37 -0
- package/docs/facts.json +1 -1
- package/install/claude.mjs +1 -1
- package/manifest.json +37 -28
- package/package.json +1 -1
- package/pipeline/lib/claude-md-links.mjs +328 -0
- package/pipeline/lib/owned-path-gate.mjs +699 -0
- package/pipeline/lib/repo-profile-derive.mjs +1771 -0
- package/pipeline/lib/repo-profile.mjs +780 -0
- package/pipeline/lib/stack-detect.sh +59 -19
- package/pipeline/lib/unattended.mjs +17 -0
- package/pipeline/multi-agent-refs/features/repo-profile.md +96 -0
- package/pipeline/multi-agent-refs/features/review-decision.md +18 -13
- package/pipeline/multi-agent-refs/features/stack-skill-routing.md +179 -33
- package/pipeline/multi-agent-refs/outside-the-pipeline.md +33 -11
- package/pipeline/multi-agent-refs/phases/phase-1-plan.md +26 -12
- package/pipeline/multi-agent-refs/phases/phase-2-dev.md +24 -13
- package/pipeline/multi-agent-refs/phases/phase-3-review.md +16 -4
- package/pipeline/multi-agent-refs/phases/phase-4-commit.md +1 -1
- package/pipeline/multi-agent-refs/phases/phase-5-report.md +8 -0
- package/pipeline/rules/outside-the-pipeline.md +6 -1
- package/pipeline/schemas/agent-state.schema.json +66 -2
- package/pipeline/schemas/phases.json +4 -4
- package/pipeline/schemas/repo-profile.schema.json +1107 -0
- package/pipeline/schemas/token-budget.json +4 -4
- package/pipeline/scripts/agent-guard.py +30 -0
- package/pipeline/scripts/owned-path-gate.mjs +205 -0
- package/pipeline/scripts/pre-commit-check.sh +151 -1
- package/pipeline/scripts/repo-profile.mjs +244 -0
- package/pipeline/scripts/review-decision-gate.mjs +42 -18
- package/pipeline/scripts/skill-candidates.mjs +882 -0
- package/pipeline/scripts/unattended_policy.py +90 -0
- package/pipeline/scripts/usage-report.mjs +36 -6
- package/pipeline/skills/.skill-manifest.json +1 -1
|
@@ -0,0 +1,328 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
/**
|
|
3
|
+
* claude-md-links.mjs - the in-repo documents a repo's CLAUDE.md points at.
|
|
4
|
+
*
|
|
5
|
+
* A project CLAUDE.md is often an index: "naming rules are in docs/style.md".
|
|
6
|
+
* Phase 1 treats those linked files as project rules, so it needs them, and it
|
|
7
|
+
* needs them bounded. Three forms count as a reference: a markdown link, a
|
|
8
|
+
* document path written as inline code (`` See `.claude/RULES.md` ``), and a
|
|
9
|
+
* host import line (`@docs/rules.md`). A code path is resolved against the
|
|
10
|
+
* CLAUDE.md's own directory first, then the repo root; a link or import against
|
|
11
|
+
* the file's own directory; a leading `/` means the repo root when its first
|
|
12
|
+
* segment exists there, and is otherwise an absolute filesystem path, which is
|
|
13
|
+
* rejected like `~/`.
|
|
14
|
+
*
|
|
15
|
+
* One hop only (links inside a linked file are not followed), at most
|
|
16
|
+
* MAX_FILES files and MAX_BYTES bytes in total. References on a line that calls
|
|
17
|
+
* the target authoritative ("See X for ...", "X is authoritative", "source of
|
|
18
|
+
* truth") are read first, so the caps drop background reading before rules. The
|
|
19
|
+
* file that crosses the byte budget is truncated at a character boundary and
|
|
20
|
+
* reported in `truncated[]`, not dropped. SKILL.md files are skipped: skills are
|
|
21
|
+
* routed by scripts/skill-candidates.mjs. Nothing that resolves outside the repo
|
|
22
|
+
* root is read - checked on the written path and on the real path, for the
|
|
23
|
+
* CLAUDE.md files themselves too, because a symlink inside the repo can point
|
|
24
|
+
* anywhere.
|
|
25
|
+
*
|
|
26
|
+
* Usage:
|
|
27
|
+
* node claude-md-links.mjs <repo-root> [--no-content]
|
|
28
|
+
*
|
|
29
|
+
* Prints JSON: {files:[{path, from, kind, priority, bytes, content, truncated?,
|
|
30
|
+
* fullBytes?}], rejected:[{link, from, kind, reason}], truncated:[path],
|
|
31
|
+
* totalBytes, sources}. `kind` is "link", "code-path", "import" or, for a
|
|
32
|
+
* rejected CLAUDE.md, "source". Exit 0 always when the root is readable; an
|
|
33
|
+
* absent CLAUDE.md is an empty result, not an error. Exit 2 on usage.
|
|
34
|
+
*
|
|
35
|
+
* @module pipeline/lib/claude-md-links
|
|
36
|
+
*/
|
|
37
|
+
|
|
38
|
+
import { existsSync, lstatSync, readFileSync, realpathSync, statSync } from "node:fs";
|
|
39
|
+
import { basename, dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
|
|
40
|
+
|
|
41
|
+
import { invokedDirectly } from "./invoked-directly.mjs";
|
|
42
|
+
|
|
43
|
+
export const SOURCES = ["CLAUDE.md", ".claude/CLAUDE.md"];
|
|
44
|
+
export const MAX_FILES = 10;
|
|
45
|
+
export const MAX_BYTES = 200 * 1024;
|
|
46
|
+
|
|
47
|
+
const SCHEME = /^[a-z][a-z0-9+.-]*:/i;
|
|
48
|
+
const AUTHORITATIVE =
|
|
49
|
+
/\bauthoritative\b|\bsource of truth\b|\bcanonical\b|\bmust (?:read|follow)\b|\bsee\b[^\n]*?\bfor\b|\brules? (?:are|live) in\b/i;
|
|
50
|
+
|
|
51
|
+
function blank(m) {
|
|
52
|
+
return m.replace(/[^\n]/g, " ");
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
function stripFences(md) {
|
|
56
|
+
return md.replace(/^(```|~~~)[\s\S]*?^\1[^\n]*$/gm, blank);
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
function stripCode(md) {
|
|
60
|
+
return stripFences(md).replace(/`[^`\n]*`/g, blank);
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
function cleanTarget(raw) {
|
|
64
|
+
let t = raw.trim().replace(/^<|>$/g, "");
|
|
65
|
+
if (t.startsWith("#")) return null;
|
|
66
|
+
t = t.replace(/[#?].*$/, "");
|
|
67
|
+
if (!t) return null;
|
|
68
|
+
try {
|
|
69
|
+
t = decodeURI(t);
|
|
70
|
+
} catch {
|
|
71
|
+
/* keep the undecoded form */
|
|
72
|
+
}
|
|
73
|
+
return t;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
function lineAt(text, index) {
|
|
77
|
+
const start = text.lastIndexOf("\n", index - 1) + 1;
|
|
78
|
+
const end = text.indexOf("\n", index);
|
|
79
|
+
return text.slice(start, end < 0 ? text.length : end);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
function linkHits(md) {
|
|
83
|
+
const text = stripCode(String(md ?? ""));
|
|
84
|
+
const out = [];
|
|
85
|
+
const inline = /!?\[[^\]\n]*\]\(\s*(<[^>\n]*>|[^\s)]+)(?:\s+(?:"[^"\n]*"|'[^'\n]*'))?\s*\)/g;
|
|
86
|
+
const reference = /^[ \t]{0,3}\[[^\]\n]+\]:[ \t]*(<[^>\n]*>|\S+)/gm;
|
|
87
|
+
for (const re of [inline, reference]) {
|
|
88
|
+
for (const m of text.matchAll(re)) {
|
|
89
|
+
const t = cleanTarget(m[1]);
|
|
90
|
+
if (t) out.push({ at: m.index, t });
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
return out.sort((a, b) => a.at - b.at);
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* Link targets in a markdown document, in document order, fragments and titles
|
|
98
|
+
* removed. Anchor-only links and anything inside code are skipped.
|
|
99
|
+
* @param {string} md
|
|
100
|
+
* @returns {string[]}
|
|
101
|
+
*/
|
|
102
|
+
export function extractLinks(md) {
|
|
103
|
+
return linkHits(md).map((x) => x.t);
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
const DOC_PATH =
|
|
107
|
+
/^(?:\.{0,2}\/)?[A-Za-z0-9._-]+(?:\/[A-Za-z0-9._-]+)*\.(?:md|mdx|markdown|txt|rst)$/i;
|
|
108
|
+
|
|
109
|
+
function codePathHits(md) {
|
|
110
|
+
const text = stripFences(String(md ?? ""));
|
|
111
|
+
const out = [];
|
|
112
|
+
for (const m of text.matchAll(/`([^`\n]+)`/g)) {
|
|
113
|
+
const t = m[1].trim();
|
|
114
|
+
if (DOC_PATH.test(t)) out.push({ at: m.index, t });
|
|
115
|
+
}
|
|
116
|
+
return out;
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* Document paths written as inline code, in document order. Only spans that are
|
|
121
|
+
* nothing but a relative-looking path to a text document count, so a command,
|
|
122
|
+
* a type name or a glob is never taken for a reference.
|
|
123
|
+
* @param {string} md
|
|
124
|
+
* @returns {string[]}
|
|
125
|
+
*/
|
|
126
|
+
export function extractCodePaths(md) {
|
|
127
|
+
return codePathHits(md).map((x) => x.t);
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
function importHits(md) {
|
|
131
|
+
const text = stripCode(String(md ?? ""));
|
|
132
|
+
const out = [];
|
|
133
|
+
const re = /(^|[\s(])@((?:~|\.{1,2})?\/?[A-Za-z0-9._-]+(?:\/[A-Za-z0-9._-]+)*)/g;
|
|
134
|
+
for (const m of text.matchAll(re)) {
|
|
135
|
+
const t = m[2].replace(/[.]+$/, "");
|
|
136
|
+
if (/\.[A-Za-z0-9]+$/.test(basename(t))) out.push({ at: m.index + m[1].length, t });
|
|
137
|
+
}
|
|
138
|
+
return out;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* Host import lines (`@path/to/file.md`) outside code, in document order. An
|
|
143
|
+
* `@` inside a word (an e-mail address) is not an import, and a target needs a
|
|
144
|
+
* file extension, so a mention or a package scope is not taken for one.
|
|
145
|
+
* @param {string} md
|
|
146
|
+
* @returns {string[]}
|
|
147
|
+
*/
|
|
148
|
+
export function extractImports(md) {
|
|
149
|
+
return importHits(md).map((x) => x.t);
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
function inside(root, p) {
|
|
153
|
+
const rel = relative(root, p);
|
|
154
|
+
return rel === "" || (!rel.startsWith(`..${sep}`) && rel !== ".." && !isAbsolute(rel));
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
function toPosix(p) {
|
|
158
|
+
return p.split(sep).join("/");
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
function locate(root, ref) {
|
|
162
|
+
const { link } = ref;
|
|
163
|
+
if (SCHEME.test(link) || link.startsWith("//")) return { reason: "url" };
|
|
164
|
+
if (link.startsWith("~")) return { reason: "absolute path" };
|
|
165
|
+
let bases = ref.bases;
|
|
166
|
+
let rel = link;
|
|
167
|
+
if (isAbsolute(link)) {
|
|
168
|
+
const first = link.split("/").find(Boolean);
|
|
169
|
+
if (!first || !existsSync(join(root, first))) return { reason: "absolute path" };
|
|
170
|
+
bases = [root];
|
|
171
|
+
rel = link.replace(/^\/+/, "");
|
|
172
|
+
}
|
|
173
|
+
let sawInside = false;
|
|
174
|
+
for (const b of bases) {
|
|
175
|
+
const target = resolve(b, rel);
|
|
176
|
+
if (!inside(root, target)) continue;
|
|
177
|
+
sawInside = true;
|
|
178
|
+
try {
|
|
179
|
+
lstatSync(target);
|
|
180
|
+
return { target };
|
|
181
|
+
} catch {
|
|
182
|
+
/* try the next base */
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
return { reason: sawInside ? "missing" : "outside repo" };
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
/** Largest prefix of `buf` no longer than `limit` that ends on a UTF-8 character boundary. */
|
|
189
|
+
function utf8Cut(buf, limit) {
|
|
190
|
+
let end = Math.min(limit, buf.length);
|
|
191
|
+
while (end > 0 && end < buf.length && (buf[end] & 0xc0) === 0x80) end--;
|
|
192
|
+
return end;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/**
|
|
196
|
+
* One reference: a file entry, a rejection reason, or null for a silent skip
|
|
197
|
+
* (a duplicate, or one of the CLAUDE.md files themselves).
|
|
198
|
+
*/
|
|
199
|
+
function admit(root, ref, seen, result, { maxFiles, maxBytes, withContent }) {
|
|
200
|
+
const found = locate(root, ref);
|
|
201
|
+
if (found.reason) return found.reason;
|
|
202
|
+
let real;
|
|
203
|
+
try {
|
|
204
|
+
real = realpathSync(found.target);
|
|
205
|
+
} catch {
|
|
206
|
+
return "missing";
|
|
207
|
+
}
|
|
208
|
+
if (!inside(root, real)) return "outside repo";
|
|
209
|
+
if (seen.has(real) || SOURCES.some((s) => join(root, s) === real)) return null;
|
|
210
|
+
if (basename(found.target) === "SKILL.md" || basename(real) === "SKILL.md") return "skill file";
|
|
211
|
+
let st;
|
|
212
|
+
try {
|
|
213
|
+
st = statSync(real);
|
|
214
|
+
} catch {
|
|
215
|
+
return "missing";
|
|
216
|
+
}
|
|
217
|
+
if (!st.isFile()) return "not a regular file";
|
|
218
|
+
if (result.files.length >= maxFiles) return "file cap";
|
|
219
|
+
const room = maxBytes - result.totalBytes;
|
|
220
|
+
if (room <= 0) return "byte budget";
|
|
221
|
+
let buf;
|
|
222
|
+
try {
|
|
223
|
+
buf = readFileSync(real);
|
|
224
|
+
} catch {
|
|
225
|
+
return "unreadable";
|
|
226
|
+
}
|
|
227
|
+
if (buf.includes(0)) return "binary";
|
|
228
|
+
const end = buf.length > room ? utf8Cut(buf, room) : buf.length;
|
|
229
|
+
if (end === 0) return "byte budget";
|
|
230
|
+
seen.add(real);
|
|
231
|
+
result.totalBytes += end;
|
|
232
|
+
const path = toPosix(relative(root, real));
|
|
233
|
+
const entry = {
|
|
234
|
+
path,
|
|
235
|
+
from: null,
|
|
236
|
+
kind: ref.kind,
|
|
237
|
+
priority: ref.priority,
|
|
238
|
+
bytes: end,
|
|
239
|
+
};
|
|
240
|
+
if (end < buf.length) {
|
|
241
|
+
entry.truncated = true;
|
|
242
|
+
entry.fullBytes = buf.length;
|
|
243
|
+
result.truncated.push(path);
|
|
244
|
+
}
|
|
245
|
+
if (withContent) entry.content = buf.subarray(0, end).toString("utf8");
|
|
246
|
+
return entry;
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
function refsOf(text, src, base, root) {
|
|
250
|
+
const tag = (kind, bases) => (h) => ({
|
|
251
|
+
link: h.t,
|
|
252
|
+
kind,
|
|
253
|
+
bases,
|
|
254
|
+
from: src,
|
|
255
|
+
at: h.at,
|
|
256
|
+
priority: AUTHORITATIVE.test(lineAt(text, h.at)) ? "authoritative" : "normal",
|
|
257
|
+
});
|
|
258
|
+
return [
|
|
259
|
+
...linkHits(text).map(tag("link", [base])),
|
|
260
|
+
...codePathHits(text).map(tag("code-path", base === root ? [base] : [base, root])),
|
|
261
|
+
...importHits(text).map(tag("import", [base])),
|
|
262
|
+
].sort((a, b) => a.at - b.at);
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
/**
|
|
266
|
+
* Follow relative links from the repo's CLAUDE.md files, one hop, bounded.
|
|
267
|
+
* @param {string} repoRoot
|
|
268
|
+
* @param {{maxFiles?: number, maxBytes?: number, content?: boolean}} [opts]
|
|
269
|
+
*/
|
|
270
|
+
export function followClaudeMdLinks(repoRoot, opts = {}) {
|
|
271
|
+
const maxFiles = opts.maxFiles ?? MAX_FILES;
|
|
272
|
+
const maxBytes = opts.maxBytes ?? MAX_BYTES;
|
|
273
|
+
const withContent = opts.content !== false;
|
|
274
|
+
const result = { files: [], rejected: [], truncated: [], totalBytes: 0, sources: [] };
|
|
275
|
+
|
|
276
|
+
let root;
|
|
277
|
+
try {
|
|
278
|
+
root = realpathSync(resolve(repoRoot));
|
|
279
|
+
} catch {
|
|
280
|
+
return result;
|
|
281
|
+
}
|
|
282
|
+
const seen = new Set();
|
|
283
|
+
const refs = [];
|
|
284
|
+
|
|
285
|
+
for (const src of SOURCES) {
|
|
286
|
+
const srcPath = join(root, src);
|
|
287
|
+
let text;
|
|
288
|
+
try {
|
|
289
|
+
if (!statSync(srcPath).isFile()) continue;
|
|
290
|
+
if (!inside(root, realpathSync(srcPath))) {
|
|
291
|
+
result.rejected.push({ link: src, from: src, kind: "source", reason: "outside repo" });
|
|
292
|
+
continue;
|
|
293
|
+
}
|
|
294
|
+
text = readFileSync(srcPath, "utf8");
|
|
295
|
+
} catch {
|
|
296
|
+
continue;
|
|
297
|
+
}
|
|
298
|
+
result.sources.push(src);
|
|
299
|
+
refs.push(...refsOf(text, src, dirname(srcPath), root));
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
const ordered = [
|
|
303
|
+
...refs.filter((r) => r.priority === "authoritative"),
|
|
304
|
+
...refs.filter((r) => r.priority !== "authoritative"),
|
|
305
|
+
];
|
|
306
|
+
for (const ref of ordered) {
|
|
307
|
+
const verdict = admit(root, ref, seen, result, { maxFiles, maxBytes, withContent });
|
|
308
|
+
if (verdict === null) continue;
|
|
309
|
+
if (typeof verdict === "string") {
|
|
310
|
+
result.rejected.push({ link: ref.link, from: ref.from, kind: ref.kind, reason: verdict });
|
|
311
|
+
} else {
|
|
312
|
+
verdict.from = ref.from;
|
|
313
|
+
result.files.push(verdict);
|
|
314
|
+
}
|
|
315
|
+
}
|
|
316
|
+
return result;
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
if (invokedDirectly(import.meta.url)) {
|
|
320
|
+
const args = process.argv.slice(2);
|
|
321
|
+
const dir = args.find((a) => !a.startsWith("--"));
|
|
322
|
+
if (!dir) {
|
|
323
|
+
console.error("usage: claude-md-links.mjs <repo-root> [--no-content]");
|
|
324
|
+
process.exit(2);
|
|
325
|
+
}
|
|
326
|
+
const out = followClaudeMdLinks(dir, { content: !args.includes("--no-content") });
|
|
327
|
+
process.stdout.write(`${JSON.stringify(out, null, 2)}\n`);
|
|
328
|
+
}
|