@amenophis1er/foreman 0.1.17 → 0.1.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -1
- package/package.json +1 -1
- package/src/crew.test.ts +376 -0
- package/src/crew.ts +330 -0
- package/src/gitwork.test.ts +84 -2
- package/src/gitwork.ts +143 -2
- package/src/mcp.ts +11 -2
- package/src/orchestrator.test.ts +516 -2
- package/src/orchestrator.ts +514 -69
- package/src/run-crew.test.ts +99 -0
- package/src/run-crew.ts +101 -0
- package/src/server.ts +136 -6
- package/src/types.ts +29 -0
- package/ui/dist/assets/index-0QuGXbFg.js +76 -0
- package/ui/dist/index.html +1 -1
- package/ui/dist/assets/index-Bcc4KMtO.js +0 -68
package/src/crew.ts
ADDED
|
@@ -0,0 +1,330 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Crew presets, and the reviewer gate that stands between a run and "done".
|
|
3
|
+
*
|
|
4
|
+
* A mission that runs unattended has nobody reading its diff. The director is
|
|
5
|
+
* the same kind of agent as the workers, judging its own homework, and asking
|
|
6
|
+
* it nicely to fetch a second opinion is worth exactly as much as any other
|
|
7
|
+
* instruction in a prompt. So a crew preset — a named role a human opts into
|
|
8
|
+
* at compose time — can be marked `requiredForDone`, and Foreman itself refuses
|
|
9
|
+
* to record the run as done until that reviewer has returned PASS on the diff
|
|
10
|
+
* the run actually ends with.
|
|
11
|
+
*
|
|
12
|
+
* Everything here is pure on purpose: the gate is a rule about a run, not a
|
|
13
|
+
* conversation with a model, and a rule that decides whether work counts as
|
|
14
|
+
* finished has to be provable by a test rather than observed in a log.
|
|
15
|
+
*/
|
|
16
|
+
import { createHash } from 'node:crypto';
|
|
17
|
+
import type { ModelChoice, ToolPolicy } from './types.js';
|
|
18
|
+
|
|
19
|
+
export type CrewKind = 'reviewer' | 'specialist';
|
|
20
|
+
|
|
21
|
+
export interface CrewPreset {
|
|
22
|
+
id: string;
|
|
23
|
+
name: string;
|
|
24
|
+
kind: CrewKind;
|
|
25
|
+
model?: ModelChoice;
|
|
26
|
+
providerId?: string;
|
|
27
|
+
brief: string;
|
|
28
|
+
toolPolicy?: 'read-only' | 'default';
|
|
29
|
+
requiredForDone: boolean;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/** One reviewer's answer about one diff. `at` is what makes a later one later. */
|
|
33
|
+
export interface ReviewVerdict {
|
|
34
|
+
presetId: string;
|
|
35
|
+
name: string;
|
|
36
|
+
pass: boolean;
|
|
37
|
+
findings: string;
|
|
38
|
+
diffHash: string;
|
|
39
|
+
workerId: string;
|
|
40
|
+
costUsd?: number;
|
|
41
|
+
at: number;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
const REVIEWER_BRIEF = [
|
|
45
|
+
'You are the reviewer on this mission. Review the diff below against MISSION.md and the',
|
|
46
|
+
'mission brief the way a demanding senior engineer reviews a colleague\'s pull request.',
|
|
47
|
+
'Look for correctness first — does this actually do what was asked, and does it break',
|
|
48
|
+
'anything that already worked. Then missing or shallow tests, security, and anything the',
|
|
49
|
+
'DONE WHEN criteria did not name but a careful reader would insist on.',
|
|
50
|
+
'The verdict is PASS only when you would merge this yourself.',
|
|
51
|
+
].join(' ');
|
|
52
|
+
|
|
53
|
+
const SECURITY_BRIEF = [
|
|
54
|
+
'You are the security reviewer on this mission. Review the diff below for the ways this',
|
|
55
|
+
'change could be abused: secrets or tokens committed or logged, injection of any kind',
|
|
56
|
+
'(shell, SQL, HTML, prompt), path handling that can escape its intended root, and',
|
|
57
|
+
'permissions — anything newly reachable without authentication, or granted more access',
|
|
58
|
+
'than the task needs. Ignore style; say plainly what an attacker could do.',
|
|
59
|
+
'The verdict is PASS only when you would ship this yourself.',
|
|
60
|
+
].join(' ');
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* The crew a human sees before they have edited anything. Defaults, not law:
|
|
64
|
+
* the moment they change one, their list replaces this one whole.
|
|
65
|
+
*
|
|
66
|
+
* Frozen, and every path out of this module hands back copies, because these
|
|
67
|
+
* objects are shared by every project on the server — a caller that "just"
|
|
68
|
+
* flipped requiredForDone on one would be rewriting the gate for all of them.
|
|
69
|
+
*/
|
|
70
|
+
export const BUILT_IN_PRESETS: readonly CrewPreset[] = Object.freeze([
|
|
71
|
+
Object.freeze({
|
|
72
|
+
id: 'reviewer',
|
|
73
|
+
name: 'Reviewer',
|
|
74
|
+
kind: 'reviewer' as const,
|
|
75
|
+
model: 'opus',
|
|
76
|
+
toolPolicy: 'read-only' as const,
|
|
77
|
+
requiredForDone: true,
|
|
78
|
+
brief: REVIEWER_BRIEF,
|
|
79
|
+
}),
|
|
80
|
+
Object.freeze({
|
|
81
|
+
id: 'security-review',
|
|
82
|
+
name: 'Security review',
|
|
83
|
+
kind: 'reviewer' as const,
|
|
84
|
+
model: 'opus',
|
|
85
|
+
toolPolicy: 'read-only' as const,
|
|
86
|
+
requiredForDone: false,
|
|
87
|
+
brief: SECURITY_BRIEF,
|
|
88
|
+
}),
|
|
89
|
+
]) as readonly CrewPreset[];
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* What "read-only" means for a reviewer: these four tools denied outright, and
|
|
93
|
+
* nothing else configured.
|
|
94
|
+
*
|
|
95
|
+
* The alternative — letting Bash through and classifying each command — is the
|
|
96
|
+
* kind of rule that is right until someone writes `sh -c 'cat > f'`. A flat
|
|
97
|
+
* deny is something a test can prove. It costs the reviewer nothing it needs:
|
|
98
|
+
* AUTO_ALLOW_TOOLS in src/policy.ts already lets Read/Glob/Grep/WebFetch/
|
|
99
|
+
* WebSearch through, and the diff it is judging arrives in its brief, so the
|
|
100
|
+
* only thing it loses is the ability to change the code it is judging.
|
|
101
|
+
*/
|
|
102
|
+
export const REVIEWER_TOOL_POLICY: ToolPolicy = Object.freeze({
|
|
103
|
+
Write: 'deny',
|
|
104
|
+
Edit: 'deny',
|
|
105
|
+
NotebookEdit: 'deny',
|
|
106
|
+
Bash: 'deny',
|
|
107
|
+
}) as ToolPolicy;
|
|
108
|
+
|
|
109
|
+
function copy(p: CrewPreset): CrewPreset {
|
|
110
|
+
const out: CrewPreset = { id: p.id, name: p.name, kind: p.kind, brief: p.brief, requiredForDone: p.requiredForDone };
|
|
111
|
+
if (p.model) out.model = p.model;
|
|
112
|
+
if (p.providerId) out.providerId = p.providerId;
|
|
113
|
+
if (p.toolPolicy) out.toolPolicy = p.toolPolicy;
|
|
114
|
+
return out;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
function nonEmptyString(v: unknown): string | undefined {
|
|
118
|
+
return typeof v === 'string' && v.trim() !== '' ? v : undefined;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* A `crewPresets` value read off settings JSON, which nobody validated on the
|
|
123
|
+
* way in. Anything shaped wrongly is dropped rather than repaired: a preset
|
|
124
|
+
* with no id cannot be referenced by a run, and a preset with no name has
|
|
125
|
+
* nothing to put in the sentence that blocks the run.
|
|
126
|
+
*
|
|
127
|
+
* null and [] mean different things, and that difference is why this returns
|
|
128
|
+
* null at all: null is "not configured, fall back", [] is "the human deleted
|
|
129
|
+
* every preset", and answering [] with the built-ins would resurrect a
|
|
130
|
+
* reviewer they had just removed.
|
|
131
|
+
*/
|
|
132
|
+
export function normalizePresets(raw: unknown): CrewPreset[] | null {
|
|
133
|
+
if (!Array.isArray(raw)) return null;
|
|
134
|
+
const out: CrewPreset[] = [];
|
|
135
|
+
const seen = new Set<string>();
|
|
136
|
+
for (const entry of raw) {
|
|
137
|
+
if (!entry || typeof entry !== 'object' || Array.isArray(entry)) continue;
|
|
138
|
+
const e = entry as Record<string, unknown>;
|
|
139
|
+
const id = nonEmptyString(e.id);
|
|
140
|
+
const name = nonEmptyString(e.name);
|
|
141
|
+
if (!id || !name || seen.has(id)) continue;
|
|
142
|
+
seen.add(id);
|
|
143
|
+
const preset: CrewPreset = {
|
|
144
|
+
id,
|
|
145
|
+
name,
|
|
146
|
+
// A preset of an unknown kind is still a worker with a brief; reviewer is
|
|
147
|
+
// the safe reading, since the only thing kind decides is how it is run.
|
|
148
|
+
kind: e.kind === 'specialist' ? 'specialist' : 'reviewer',
|
|
149
|
+
brief: typeof e.brief === 'string' ? e.brief : '',
|
|
150
|
+
// Anything other than a literal true is not consent to block a run.
|
|
151
|
+
requiredForDone: e.requiredForDone === true,
|
|
152
|
+
};
|
|
153
|
+
const model = nonEmptyString(e.model);
|
|
154
|
+
if (model) preset.model = model;
|
|
155
|
+
const providerId = nonEmptyString(e.providerId);
|
|
156
|
+
if (providerId) preset.providerId = providerId;
|
|
157
|
+
if (e.toolPolicy === 'read-only' || e.toolPolicy === 'default') preset.toolPolicy = e.toolPolicy;
|
|
158
|
+
out.push(preset);
|
|
159
|
+
}
|
|
160
|
+
return out;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* The crew presets in force for a project. A project's list replaces the
|
|
165
|
+
* global one whole rather than merging with it — the same way the rest of
|
|
166
|
+
* Settings overlays — because a merge would make "I removed the reviewer here"
|
|
167
|
+
* unsayable.
|
|
168
|
+
*/
|
|
169
|
+
export function crewPresetsFrom(global: unknown, project: unknown): CrewPreset[] {
|
|
170
|
+
const read = (blob: unknown): CrewPreset[] | null => {
|
|
171
|
+
if (!blob || typeof blob !== 'object') return null;
|
|
172
|
+
return normalizePresets((blob as Record<string, unknown>).crewPresets);
|
|
173
|
+
};
|
|
174
|
+
return (read(project) ?? read(global) ?? BUILT_IN_PRESETS).map(copy);
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* The crew a run starts with, copied onto the run record. Deep copies, because
|
|
179
|
+
* editing a preset next week must not change what a run that is still going —
|
|
180
|
+
* or one that finished in March — was reviewed against.
|
|
181
|
+
*/
|
|
182
|
+
export function freezeCrew(ids: readonly string[], presets: readonly CrewPreset[]): CrewPreset[] {
|
|
183
|
+
const wanted = new Set(ids);
|
|
184
|
+
const taken = new Set<string>();
|
|
185
|
+
const out: CrewPreset[] = [];
|
|
186
|
+
for (const p of presets) {
|
|
187
|
+
if (!wanted.has(p.id) || taken.has(p.id)) continue;
|
|
188
|
+
taken.add(p.id);
|
|
189
|
+
out.push(copy(p));
|
|
190
|
+
}
|
|
191
|
+
return out;
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/** Past this, findings are a wall of text nobody reads and a row nobody wants to store. */
|
|
195
|
+
const MAX_FINDINGS = 8000;
|
|
196
|
+
|
|
197
|
+
/**
|
|
198
|
+
* The reviewer's verdict, dug out of its report. It is told to end with a line
|
|
199
|
+
* that is exactly `VERDICT: PASS` or `VERDICT: FAIL`; the last such line wins,
|
|
200
|
+
* because a report that quotes the format while explaining itself and then
|
|
201
|
+
* states its verdict at the end has stated it at the end.
|
|
202
|
+
*
|
|
203
|
+
* null is not FAIL. A report that never says either has not been reviewed —
|
|
204
|
+
* the model ran out of turns, or answered in prose — and the caller has to be
|
|
205
|
+
* able to tell that apart from a reviewer that looked and said no.
|
|
206
|
+
*/
|
|
207
|
+
export function parseVerdict(report: string): { pass: boolean; findings: string } | null {
|
|
208
|
+
if (typeof report !== 'string' || report === '') return null;
|
|
209
|
+
const lines = report.split('\n');
|
|
210
|
+
for (let i = lines.length - 1; i >= 0; i--) {
|
|
211
|
+
// Markdown decoration is what a model reaches for when told to make a line
|
|
212
|
+
// stand out; `**VERDICT: PASS**` is the instruction followed, not broken.
|
|
213
|
+
const line = lines[i].replace(/\*\*/g, '').replace(/^\s*#+\s*/, '');
|
|
214
|
+
const m = /^\s*VERDICT:\s*(PASS|FAIL)\s*$/i.exec(line);
|
|
215
|
+
if (!m) continue;
|
|
216
|
+
let findings = lines.slice(i + 1).join('\n').trim();
|
|
217
|
+
if (findings.length > MAX_FINDINGS) {
|
|
218
|
+
findings = findings.slice(0, MAX_FINDINGS) + '\n\n[findings truncated]';
|
|
219
|
+
}
|
|
220
|
+
return { pass: m[1].toUpperCase() === 'PASS', findings };
|
|
221
|
+
}
|
|
222
|
+
return null;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
/**
|
|
226
|
+
* A fingerprint of what a run has changed, so a PASS can be pinned to the diff
|
|
227
|
+
* it was given. Canonical: sorted by path and field-separated, so the same
|
|
228
|
+
* files listed in a different order are the same hash, while one changed line
|
|
229
|
+
* anywhere is a different one.
|
|
230
|
+
*/
|
|
231
|
+
export function diffHash(deck: {
|
|
232
|
+
files?: Array<{ path: string; status: string; additions: number; deletions: number; diff?: string }>;
|
|
233
|
+
}): string {
|
|
234
|
+
const files = deck?.files ?? [];
|
|
235
|
+
const canon = [...files]
|
|
236
|
+
.sort((a, b) => (a.path < b.path ? -1 : a.path > b.path ? 1 : 0))
|
|
237
|
+
.map((f) => [f.path, f.status, f.additions, f.deletions, f.diff ?? ''].join('\0'))
|
|
238
|
+
.join('\n');
|
|
239
|
+
// A run that changed nothing still has a stable hash, so "reviewed, then
|
|
240
|
+
// nothing moved" holds for an empty diff too.
|
|
241
|
+
return createHash('sha256').update(canon).digest('hex');
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
export interface ReviewBlocker {
|
|
245
|
+
presetId: string;
|
|
246
|
+
name: string;
|
|
247
|
+
reason: 'missing' | 'fail' | 'stale';
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
/**
|
|
251
|
+
* THE GATE. Every required reviewer that has not passed the run's current
|
|
252
|
+
* diff, in crew order. An empty array is the only thing that lets a run be
|
|
253
|
+
* recorded as done.
|
|
254
|
+
*
|
|
255
|
+
* 'stale' is the case worth the machinery: a reviewer passed, then the
|
|
256
|
+
* director kept working. That PASS was about code that no longer exists, and
|
|
257
|
+
* treating it as consent for whatever came afterwards would make the gate
|
|
258
|
+
* trivially walk-around-able — review early, then commit anything.
|
|
259
|
+
*/
|
|
260
|
+
export function reviewBlockers(
|
|
261
|
+
crew: readonly CrewPreset[] | undefined,
|
|
262
|
+
verdicts: readonly ReviewVerdict[] | undefined,
|
|
263
|
+
currentDiffHash: string,
|
|
264
|
+
): ReviewBlocker[] {
|
|
265
|
+
const out: ReviewBlocker[] = [];
|
|
266
|
+
for (const preset of crew ?? []) {
|
|
267
|
+
if (preset.requiredForDone !== true) continue;
|
|
268
|
+
// Latest by `at`, ties going to the one that arrived later in the array —
|
|
269
|
+
// two verdicts written in the same millisecond are in the order they were
|
|
270
|
+
// appended.
|
|
271
|
+
let latest: ReviewVerdict | undefined;
|
|
272
|
+
for (const v of verdicts ?? []) {
|
|
273
|
+
if (v.presetId !== preset.id) continue;
|
|
274
|
+
if (!latest || v.at >= latest.at) latest = v;
|
|
275
|
+
}
|
|
276
|
+
if (!latest) out.push({ presetId: preset.id, name: preset.name, reason: 'missing' });
|
|
277
|
+
else if (!latest.pass) out.push({ presetId: preset.id, name: preset.name, reason: 'fail' });
|
|
278
|
+
else if (latest.diffHash !== currentDiffHash) out.push({ presetId: preset.id, name: preset.name, reason: 'stale' });
|
|
279
|
+
}
|
|
280
|
+
return out;
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
/** One human sentence for the `mission_unreviewed` event. */
|
|
284
|
+
export function unreviewedText(blockers: readonly ReviewBlocker[]): string {
|
|
285
|
+
if (!blockers.length) return '';
|
|
286
|
+
const clause = (b: ReviewBlocker): string => {
|
|
287
|
+
if (b.reason === 'fail') return `${b.name} returned FAIL`;
|
|
288
|
+
if (b.reason === 'stale') return `${b.name} passed an earlier version of the diff; the code changed after it`;
|
|
289
|
+
return `${b.name} has not reviewed this run`;
|
|
290
|
+
};
|
|
291
|
+
const parts = blockers.map(clause);
|
|
292
|
+
const list = parts.length === 1
|
|
293
|
+
? parts[0]
|
|
294
|
+
: `${parts.slice(0, -1).join('; ')}; and ${parts[parts.length - 1]}`;
|
|
295
|
+
const lead = blockers.length === 1 ? 'A required reviewer has not passed this run' : 'Required reviewers have not passed this run';
|
|
296
|
+
return `${lead}: ${list}. This run is not done. Resume to continue it.`;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
/**
|
|
300
|
+
* The prompt the reviewer worker is started with. The diff is handed over in
|
|
301
|
+
* the brief rather than left to be discovered, because a read-only worker with
|
|
302
|
+
* no Bash cannot run `git diff` for itself — and because the thing being
|
|
303
|
+
* reviewed should be the thing Foreman will hash, not whatever the model
|
|
304
|
+
* happened to look at.
|
|
305
|
+
*/
|
|
306
|
+
export function reviewBriefFor(
|
|
307
|
+
preset: CrewPreset,
|
|
308
|
+
opts: { mission: string; doneWhen: string; diff: string; truncated: boolean },
|
|
309
|
+
): string {
|
|
310
|
+
const parts: string[] = [preset.brief.trim()];
|
|
311
|
+
parts.push(`## The mission\n\n${opts.mission.trim() || '(no brief was recorded)'}`);
|
|
312
|
+
parts.push(`## DONE WHEN\n\n${opts.doneWhen.trim() || '(no criteria were recorded)'}`);
|
|
313
|
+
parts.push(`## The diff\n\n\`\`\`diff\n${opts.diff}\n\`\`\``);
|
|
314
|
+
if (opts.truncated) {
|
|
315
|
+
parts.push(
|
|
316
|
+
'That diff was cut short because it is large. Read the files you need directly — ' +
|
|
317
|
+
'Read, Glob and Grep are available to you.',
|
|
318
|
+
);
|
|
319
|
+
}
|
|
320
|
+
parts.push(
|
|
321
|
+
'You cannot modify files: Write, Edit and Bash are denied to you, and attempting them ' +
|
|
322
|
+
'wastes the run. Report, do not fix.',
|
|
323
|
+
);
|
|
324
|
+
parts.push(
|
|
325
|
+
'End your report with a line that is exactly `VERDICT: PASS` or `VERDICT: FAIL`, and ' +
|
|
326
|
+
'then list your findings under it, each with a `file:line` where you can give one. ' +
|
|
327
|
+
'Only what follows that line is kept as your findings.',
|
|
328
|
+
);
|
|
329
|
+
return parts.join('\n\n');
|
|
330
|
+
}
|
package/src/gitwork.test.ts
CHANGED
|
@@ -3,9 +3,10 @@ import assert from 'node:assert/strict';
|
|
|
3
3
|
import os from 'node:os';
|
|
4
4
|
import path from 'node:path';
|
|
5
5
|
import { execFileSync } from 'node:child_process';
|
|
6
|
-
import { mkdtemp, realpath, writeFile } from 'node:fs/promises';
|
|
7
|
-
import { closeMissionBranch, defaultBranch, worktreeGrant, worktreeParent, ensureMissionBranch, gitInfo, missionBranchName, remoteHasBranch, resolvePrBase, startMissionBranch, renameMissionBranch, dirtyPaths,
|
|
6
|
+
import { mkdir, mkdtemp, realpath, symlink, unlink, writeFile } from 'node:fs/promises';
|
|
7
|
+
import { changeFingerprint, closeMissionBranch, defaultBranch, worktreeGrant, worktreeParent, ensureMissionBranch, gitInfo, missionBranchName, remoteHasBranch, resolvePrBase, startMissionBranch, renameMissionBranch, dirtyPaths,
|
|
8
8
|
} from './gitwork.js';
|
|
9
|
+
import type { ReviewVerdict } from './crew.js';
|
|
9
10
|
|
|
10
11
|
const sh = (cwd: string, ...args: string[]) => execFileSync('git', args, { cwd, stdio: 'pipe', env: { ...process.env, GIT_CONFIG_GLOBAL: '/dev/null' } }).toString();
|
|
11
12
|
|
|
@@ -93,6 +94,26 @@ test('prDraft: the run title, the brief, the boxes as the mission left them, and
|
|
|
93
94
|
assert.match(d.body, /## Mission\n\nAdd a footer to the page\.\nKeep it small\./);
|
|
94
95
|
assert.match(d.body, /## Done when\n\n- \[x\] footer\.html exists\n- \[ \] linked from index/);
|
|
95
96
|
assert.match(d.body, /branch `foreman\/add-a-footer-ab12` from `main` · spend \$0\.42/);
|
|
97
|
+
assert.ok(!/## Review/.test(d.body), 'a run nobody reviewed says nothing about review');
|
|
98
|
+
});
|
|
99
|
+
|
|
100
|
+
test('prDraft: the reviewers and their verdicts, with the head of the findings', async () => {
|
|
101
|
+
const { prDraft } = await import('./gitwork.js');
|
|
102
|
+
const verdict = (over: Partial<ReviewVerdict>): ReviewVerdict => ({
|
|
103
|
+
presetId: 'reviewer', name: 'Reviewer', pass: true, findings: '', diffHash: 'h', workerId: 'w1', at: 1, ...over,
|
|
104
|
+
});
|
|
105
|
+
const d = prDraft(
|
|
106
|
+
{ mission: 'Add a footer.', costUsd: 1, costBasis: 'priced', git: { branch: 'b', base: 'main', baseHead: null } },
|
|
107
|
+
null,
|
|
108
|
+
[
|
|
109
|
+
verdict({ findings: '- footer.html:12 the year is hard-coded' }),
|
|
110
|
+
verdict({ presetId: 'security-review', name: 'Security review', pass: false, findings: `x${'y'.repeat(2000)}` }),
|
|
111
|
+
],
|
|
112
|
+
);
|
|
113
|
+
assert.match(d.body, /## Review\n\n\*\*Reviewer: PASS\*\*\n\n- footer\.html:12 the year is hard-coded/);
|
|
114
|
+
assert.match(d.body, /\*\*Security review: FAIL\*\*/);
|
|
115
|
+
assert.match(d.body, /… the rest is in the run's record\./, 'a long findings list is cut, not pasted whole');
|
|
116
|
+
assert.ok(d.body.length < 2000, 'the body stays a pull request, not an archive');
|
|
96
117
|
});
|
|
97
118
|
|
|
98
119
|
|
|
@@ -195,3 +216,64 @@ test('a worktree kept inside the repository is not opened by opening the reposit
|
|
|
195
216
|
assert.equal(decision.grant, null, 'so the parent is not opened automatically');
|
|
196
217
|
assert.match(decision.reason ?? '', /would also open/);
|
|
197
218
|
});
|
|
219
|
+
|
|
220
|
+
test('the fingerprint copes with awkward filenames and with a repository that has no commit', async () => {
|
|
221
|
+
const dir = await mkdtemp(path.join(os.tmpdir(), 'fingerprint-git-'));
|
|
222
|
+
sh(dir, 'init', '-q', '-b', 'main');
|
|
223
|
+
sh(dir, 'config', 'user.email', 'me@example.com');
|
|
224
|
+
sh(dir, 'config', 'user.name', 'Me');
|
|
225
|
+
|
|
226
|
+
// No commit yet: `git diff HEAD` has nothing to diff against, and a run that
|
|
227
|
+
// starts a repository from nothing is an ordinary mission.
|
|
228
|
+
await writeFile(path.join(dir, 'first.txt'), 'work\n');
|
|
229
|
+
const empty = await changeFingerprint(dir);
|
|
230
|
+
assert.ok(empty, 'a repository with no HEAD still has a fingerprint');
|
|
231
|
+
|
|
232
|
+
// A name git C-quotes in `ls-files`: a quoted path handed to hash-object
|
|
233
|
+
// fails, which used to silently drop the file's content from the hash.
|
|
234
|
+
const awkward = path.join(dir, 'répertoire "odd" name.txt');
|
|
235
|
+
await writeFile(awkward, 'one\n');
|
|
236
|
+
const withAwkward = await changeFingerprint(dir);
|
|
237
|
+
assert.ok(withAwkward);
|
|
238
|
+
assert.notEqual(withAwkward, empty);
|
|
239
|
+
|
|
240
|
+
await writeFile(awkward, 'two\n');
|
|
241
|
+
assert.notEqual(await changeFingerprint(dir), withAwkward, 'editing it moves the fingerprint');
|
|
242
|
+
|
|
243
|
+
// And once there is a commit, the same file is tracked and still counts.
|
|
244
|
+
sh(dir, 'add', '-A'); sh(dir, 'commit', '-q', '-m', 'one');
|
|
245
|
+
const committed = await changeFingerprint(dir);
|
|
246
|
+
await writeFile(awkward, 'three\n');
|
|
247
|
+
assert.notEqual(await changeFingerprint(dir), committed);
|
|
248
|
+
});
|
|
249
|
+
|
|
250
|
+
test('the fingerprint is scoped to the project folder, and sees a retargeted symlink', async () => {
|
|
251
|
+
// A project that is a subdirectory of a bigger repository: a sibling's
|
|
252
|
+
// changes are not this mission's, and must not invalidate its review.
|
|
253
|
+
const repoRoot = await repo();
|
|
254
|
+
const project = path.join(repoRoot, 'packages', 'app');
|
|
255
|
+
await mkdir(project, { recursive: true });
|
|
256
|
+
await writeFile(path.join(project, 'index.ts'), 'export const a = 1;\n');
|
|
257
|
+
await mkdir(path.join(repoRoot, 'packages', 'other'), { recursive: true });
|
|
258
|
+
await writeFile(path.join(repoRoot, 'packages', 'other', 'index.ts'), 'export const b = 1;\n');
|
|
259
|
+
sh(repoRoot, 'add', '-A'); sh(repoRoot, 'commit', '-q', '-m', 'two packages');
|
|
260
|
+
|
|
261
|
+
const reviewed = await changeFingerprint(project);
|
|
262
|
+
assert.ok(reviewed);
|
|
263
|
+
await writeFile(path.join(repoRoot, 'packages', 'other', 'index.ts'), 'export const b = 2;\n');
|
|
264
|
+
assert.equal(await changeFingerprint(project), reviewed, 'a sibling package is not this mission');
|
|
265
|
+
await writeFile(path.join(project, 'index.ts'), 'export const a = 2;\n');
|
|
266
|
+
assert.notEqual(await changeFingerprint(project), reviewed, 'its own change still counts');
|
|
267
|
+
|
|
268
|
+
// Non-git folder: a symlink is neither file nor directory, and retargeting
|
|
269
|
+
// one changes the project without changing any content.
|
|
270
|
+
const plain = await mkdtemp(path.join(os.tmpdir(), 'links-'));
|
|
271
|
+
await writeFile(path.join(plain, 'one.txt'), 'one\n');
|
|
272
|
+
await writeFile(path.join(plain, 'two.txt'), 'two\n');
|
|
273
|
+
await symlink('one.txt', path.join(plain, 'current'));
|
|
274
|
+
const linked = await changeFingerprint(plain);
|
|
275
|
+
assert.ok(linked);
|
|
276
|
+
await unlink(path.join(plain, 'current'));
|
|
277
|
+
await symlink('two.txt', path.join(plain, 'current'));
|
|
278
|
+
assert.notEqual(await changeFingerprint(plain), linked, 'a retargeted link moves the fingerprint');
|
|
279
|
+
});
|
package/src/gitwork.ts
CHANGED
|
@@ -9,7 +9,10 @@
|
|
|
9
9
|
* the button that says so — once, for that branch, to open the pull request.
|
|
10
10
|
*/
|
|
11
11
|
import path from 'node:path';
|
|
12
|
+
import { readdir, readFile, readlink, stat } from 'node:fs/promises';
|
|
13
|
+
import { createHash } from 'node:crypto';
|
|
12
14
|
import { execFile } from 'node:child_process';
|
|
15
|
+
import type { ReviewVerdict } from './crew.js';
|
|
13
16
|
|
|
14
17
|
export interface GitInfo {
|
|
15
18
|
repo: boolean;
|
|
@@ -160,6 +163,116 @@ export function worktreeGrant(
|
|
|
160
163
|
return { grant: parent };
|
|
161
164
|
}
|
|
162
165
|
|
|
166
|
+
/**
|
|
167
|
+
* A fingerprint of everything this checkout has changed, for pinning a
|
|
168
|
+
* reviewer's PASS to the code it actually read.
|
|
169
|
+
*
|
|
170
|
+
* Deliberately not computed from the deck: the deck is a *view* — it stops at
|
|
171
|
+
* 200 files and carries no binary content — so a change to the 201st file, or
|
|
172
|
+
* a swapped image, would leave a deck-derived hash identical and a stale PASS
|
|
173
|
+
* looking current. This asks git instead: the full diff against HEAD including
|
|
174
|
+
* binary deltas, plus the blob hash of every untracked file. Null when the
|
|
175
|
+
* folder is not a repository or git will not answer, which callers must treat
|
|
176
|
+
* as "cannot verify", never as "nothing changed".
|
|
177
|
+
*/
|
|
178
|
+
/** git's empty tree, for diffing a repository that has no commit yet. */
|
|
179
|
+
const EMPTY_TREE = '4b825dc642cb6eb9a060e54bf8d69288fbee4904';
|
|
180
|
+
const FINGERPRINT_FILE_CAP = 20_000;
|
|
181
|
+
const FINGERPRINT_HASH_MAX_BYTES = 1024 * 1024;
|
|
182
|
+
const FINGERPRINT_SKIP = new Set(['.git', '.foreman', 'node_modules']);
|
|
183
|
+
|
|
184
|
+
/**
|
|
185
|
+
* The same fingerprint for a folder that is not a repository — Foreman links
|
|
186
|
+
* plain folders too, and a gate that only worked in git would make every
|
|
187
|
+
* mission in one impossible to finish.
|
|
188
|
+
*
|
|
189
|
+
* Content-hashed up to a megabyte a file, size and mtime beyond that, since
|
|
190
|
+
* reading a large binary on every gate check costs more than it proves.
|
|
191
|
+
* Dependency trees and Foreman's own directory are skipped: they are not the
|
|
192
|
+
* work under review. Null past the file cap, which the caller reads as
|
|
193
|
+
* "cannot verify" — the honest answer for a tree too large to pin.
|
|
194
|
+
*/
|
|
195
|
+
async function walkFingerprint(folder: string): Promise<string | null> {
|
|
196
|
+
const parts: string[] = [];
|
|
197
|
+
const walk = async (dir: string, rel: string): Promise<boolean> => {
|
|
198
|
+
const entries = await readdir(dir, { withFileTypes: true }).catch(() => null);
|
|
199
|
+
// A directory we cannot read may be where the change is. Failing open
|
|
200
|
+
// would let a PASS stand over work nobody could see.
|
|
201
|
+
if (!entries) return false;
|
|
202
|
+
for (const e of entries.sort((a, b) => (a.name < b.name ? -1 : 1))) {
|
|
203
|
+
if (FINGERPRINT_SKIP.has(e.name)) continue;
|
|
204
|
+
const full = path.join(dir, e.name);
|
|
205
|
+
const here = rel ? `${rel}/${e.name}` : e.name;
|
|
206
|
+
if (e.isDirectory()) {
|
|
207
|
+
if (!await walk(full, here)) return false;
|
|
208
|
+
continue;
|
|
209
|
+
}
|
|
210
|
+
// A symlink is neither a file nor a directory to readdir, and retargeting
|
|
211
|
+
// one changes what the project is without touching a byte of content.
|
|
212
|
+
if (e.isSymbolicLink()) {
|
|
213
|
+
const target = await readlink(full).catch(() => null);
|
|
214
|
+
if (target === null) return false;
|
|
215
|
+
parts.push(`${here}\0link\0${target}`);
|
|
216
|
+
continue;
|
|
217
|
+
}
|
|
218
|
+
if (!e.isFile()) continue;
|
|
219
|
+
if (parts.length >= FINGERPRINT_FILE_CAP) return false;
|
|
220
|
+
const st = await stat(full).catch(() => null);
|
|
221
|
+
if (!st) continue;
|
|
222
|
+
if (st.size <= FINGERPRINT_HASH_MAX_BYTES) {
|
|
223
|
+
const buf = await readFile(full).catch(() => null);
|
|
224
|
+
parts.push(`${here}\0${st.size}\0${buf ? createHash('sha256').update(buf).digest('hex') : 'unreadable'}`);
|
|
225
|
+
} else {
|
|
226
|
+
parts.push(`${here}\0${st.size}\0${st.mtimeMs}`);
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
return true;
|
|
230
|
+
};
|
|
231
|
+
if (!await walk(folder, '')) return null;
|
|
232
|
+
return createHash('sha256').update(parts.join('\n')).digest('hex');
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
export async function changeFingerprint(folder: string): Promise<string | null> {
|
|
236
|
+
try {
|
|
237
|
+
const inside = await git(['rev-parse', '--is-inside-work-tree'], folder).catch(() => '');
|
|
238
|
+
if (inside.trim() !== 'true') return walkFingerprint(folder);
|
|
239
|
+
// HEAD is part of the fingerprint, not just the dirty tree: a director
|
|
240
|
+
// that commits its work after a PASS leaves `git diff HEAD` empty, and a
|
|
241
|
+
// fingerprint of the diff alone would call the new commit unchanged and
|
|
242
|
+
// let the old PASS stand.
|
|
243
|
+
// A repository with no commit yet has no HEAD to diff against, and a
|
|
244
|
+
// mission that starts one is ordinary — so the comparison falls back to
|
|
245
|
+
// git's empty tree rather than failing, which would make every run with a
|
|
246
|
+
// required reviewer impossible to finish until someone committed.
|
|
247
|
+
const head = (await git(['rev-parse', 'HEAD'], folder).catch(() => '')).trim();
|
|
248
|
+
// Scoped to this folder, like the deck: a project linked as a subdirectory
|
|
249
|
+
// of a bigger repository must not have its review invalidated because a
|
|
250
|
+
// sibling project changed. `ls-files` below is already limited to the cwd.
|
|
251
|
+
const tracked = await git(
|
|
252
|
+
['diff', head || EMPTY_TREE, '--binary', '--no-color', '--no-ext-diff', '--', '.'], folder, 60_000,
|
|
253
|
+
);
|
|
254
|
+
// -z, because `ls-files` C-quotes any path with a quote, a tab or a
|
|
255
|
+
// non-ASCII character, and a quoted path handed back to `hash-object`
|
|
256
|
+
// fails — which used to leave those files with no content in the hash at
|
|
257
|
+
// all, so edits to them were invisible to the gate.
|
|
258
|
+
const untracked = (await git(['ls-files', '--others', '--exclude-standard', '-z'], folder))
|
|
259
|
+
.split('\0')
|
|
260
|
+
// The mission doc and the crew's scratch space are Foreman's own and
|
|
261
|
+
// change constantly; they are not the work under review.
|
|
262
|
+
.filter((p) => p && !p.startsWith('.foreman/'));
|
|
263
|
+
const parts: string[] = [`HEAD\0${head || 'none'}`, tracked];
|
|
264
|
+
for (const p of untracked.sort()) {
|
|
265
|
+
// No catch: a file whose hash cannot be read is a fingerprint that
|
|
266
|
+
// cannot be trusted, and the honest answer is "cannot verify".
|
|
267
|
+
const blob = await git(['hash-object', '--', p], folder);
|
|
268
|
+
parts.push(`${p}\0${blob.trim()}`);
|
|
269
|
+
}
|
|
270
|
+
return createHash('sha256').update(parts.join('\n')).digest('hex');
|
|
271
|
+
} catch {
|
|
272
|
+
return null;
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
|
|
163
276
|
export async function dirtyPaths(folder: string, limit = 8): Promise<string[]> {
|
|
164
277
|
try {
|
|
165
278
|
const out = await git(['status', '--porcelain', '--untracked-files=normal'], folder);
|
|
@@ -275,8 +388,22 @@ export function compareUrl(remote: string | undefined, base: string, branch: str
|
|
|
275
388
|
return `${r.web}`;
|
|
276
389
|
}
|
|
277
390
|
|
|
278
|
-
/**
|
|
279
|
-
|
|
391
|
+
/** How much of a reviewer's findings go in the body; the rest is in the run's record. */
|
|
392
|
+
const FINDINGS_HEAD = 800;
|
|
393
|
+
|
|
394
|
+
/**
|
|
395
|
+
* The pull request as Foreman drafts it: the run's title, and a body a reviewer
|
|
396
|
+
* can read without opening Foreman.
|
|
397
|
+
*
|
|
398
|
+
* `reviews` is passed in rather than read from the run's record here, because
|
|
399
|
+
* this module knows about git and nothing else — and because the caller is the
|
|
400
|
+
* only one that knows which verdicts are the ones this branch was judged by.
|
|
401
|
+
*/
|
|
402
|
+
export function prDraft(
|
|
403
|
+
run: { title?: string; mission: string; costUsd: number; costBasis?: string; git?: MissionGit },
|
|
404
|
+
missionDoc: string | null,
|
|
405
|
+
reviews?: readonly ReviewVerdict[],
|
|
406
|
+
): { title: string; body: string } {
|
|
280
407
|
const first = run.mission.split('\n').find((l) => l.trim())?.trim() ?? 'Mission';
|
|
281
408
|
const title = (run.title || first).slice(0, 120);
|
|
282
409
|
const boxes = (missionDoc ?? '').split('\n').filter((l) => /^\s*[-*] \[[ xX]\]/.test(l)).map((l) => l.trim());
|
|
@@ -285,6 +412,20 @@ export function prDraft(run: { title?: string; mission: string; costUsd: number;
|
|
|
285
412
|
'## Mission', '', run.mission.trim(), '',
|
|
286
413
|
];
|
|
287
414
|
if (boxes.length) parts.push('## Done when', '', ...boxes, '');
|
|
415
|
+
// Who reviewed this before it was offered to a human, and what they said.
|
|
416
|
+
// The whole point of the reviewer gate is that the answer travels with the
|
|
417
|
+
// work; a PASS nobody outside Foreman can see is worth nothing on a branch.
|
|
418
|
+
if (reviews?.length) {
|
|
419
|
+
parts.push('## Review', '');
|
|
420
|
+
for (const v of reviews) {
|
|
421
|
+
parts.push(`**${v.name}: ${v.pass ? 'PASS' : 'FAIL'}**`);
|
|
422
|
+
const head = v.findings.trim();
|
|
423
|
+
if (head) {
|
|
424
|
+
parts.push('', head.length > FINDINGS_HEAD ? `${head.slice(0, FINDINGS_HEAD).trimEnd()}\n\n… the rest is in the run's record.` : head);
|
|
425
|
+
}
|
|
426
|
+
parts.push('');
|
|
427
|
+
}
|
|
428
|
+
}
|
|
288
429
|
parts.push('---', `Run by [Foreman](https://github.com/amenophis1er/foreman) on branch \`${run.git?.branch ?? ''}\` from \`${run.git?.base ?? ''}\` · spend ${spend}.`);
|
|
289
430
|
return { title, body: parts.join('\n') };
|
|
290
431
|
}
|
package/src/mcp.ts
CHANGED
|
@@ -24,6 +24,8 @@ import fs from 'node:fs';
|
|
|
24
24
|
import path from 'node:path';
|
|
25
25
|
import { z } from 'zod';
|
|
26
26
|
import { describeCadence, type Cadence } from './schedule.js';
|
|
27
|
+
import type { CrewPreset, ReviewVerdict } from './crew.js';
|
|
28
|
+
import { reviewReportLines } from './run-crew.js';
|
|
27
29
|
|
|
28
30
|
export interface ToolResult {
|
|
29
31
|
/** What the model reads. */
|
|
@@ -65,6 +67,9 @@ interface RunSummary {
|
|
|
65
67
|
workers?: Array<{ id: string; status: string; costUsd: number; task: string }>;
|
|
66
68
|
git?: { branch: string; base: string; commits?: number; pr?: string; prState?: string };
|
|
67
69
|
usage?: { inputTokens: number; outputTokens: number };
|
|
70
|
+
/** Frozen at dispatch; read here only to say who was meant to review. */
|
|
71
|
+
crew?: CrewPreset[];
|
|
72
|
+
reviews?: ReviewVerdict[];
|
|
68
73
|
}
|
|
69
74
|
interface Need { kind: string; id: string; runId?: string; text: string; options?: string[]; toolName?: string; since?: number }
|
|
70
75
|
interface ProjectCard {
|
|
@@ -339,7 +344,7 @@ export function foremanTools(opts: ForemanClientOptions): ToolDef[] {
|
|
|
339
344
|
|
|
340
345
|
const runReport: ToolDef = {
|
|
341
346
|
name: 'run_report',
|
|
342
|
-
description: 'What a finished run produced, in one call: the director\'s final report, DONE WHEN ticks, the files it changed with +/− counts, the branch, commit and pull request, spend and
|
|
347
|
+
description: 'What a finished run produced, in one call: the director\'s final report, DONE WHEN ticks, the files it changed with +/− counts, the branch, commit and pull request, spend, crew and any review verdicts. For a running run it reports the state so far.',
|
|
343
348
|
schema: { runId: z.string() },
|
|
344
349
|
run: async ({ runId }) => {
|
|
345
350
|
const id = String(runId);
|
|
@@ -369,9 +374,13 @@ export function foremanTools(opts: ForemanClientOptions): ToolDef[] {
|
|
|
369
374
|
deck ? `changed: ${own.length} file${own.length === 1 ? '' : 's'} · +${deck.totals.additions} −${deck.totals.deletions}${images ? ` · ${images} screenshot${images === 1 ? '' : 's'}` : ''}${files.length > own.length ? ` · ${files.length - own.length} already dirty before the run` : ''}` : null,
|
|
370
375
|
fileLines.length ? fileLines.join('\n') + (own.length > 40 ? `\n … ${own.length - 40} more` : '') : null,
|
|
371
376
|
`crew: ${(r.workers ?? []).length} worker${(r.workers ?? []).length === 1 ? '' : 's'} · ${(r.workers ?? []).filter((w) => w.status === 'done').length} done`,
|
|
377
|
+
// The verdicts, and any required reviewer standing between this run and
|
|
378
|
+
// done. Read-only, like everything else here: presets are configuration
|
|
379
|
+
// and configuration is edited on the dashboard.
|
|
380
|
+
...reviewReportLines(r.crew, r.reviews),
|
|
372
381
|
report ? `\nDirector's report:\n${report.slice(0, 4000)}` : '\nNo final report from the director yet.',
|
|
373
382
|
].filter(Boolean);
|
|
374
|
-
return { text: lines.join('\n'), data: { run: r, doneWhen: dw, files: own, totals: deck?.totals, report } };
|
|
383
|
+
return { text: lines.join('\n'), data: { run: r, doneWhen: dw, files: own, totals: deck?.totals, report, reviews: r.reviews ?? [] } };
|
|
375
384
|
},
|
|
376
385
|
};
|
|
377
386
|
|