@amenophis1er/foreman 0.1.17 → 0.1.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -1
- package/package.json +1 -1
- package/src/crew.test.ts +376 -0
- package/src/crew.ts +330 -0
- package/src/gitwork.test.ts +84 -2
- package/src/gitwork.ts +143 -2
- package/src/mcp.ts +11 -2
- package/src/orchestrator.test.ts +516 -2
- package/src/orchestrator.ts +514 -69
- package/src/run-crew.test.ts +99 -0
- package/src/run-crew.ts +101 -0
- package/src/server.ts +136 -6
- package/src/types.ts +29 -0
- package/ui/dist/assets/index-0QuGXbFg.js +76 -0
- package/ui/dist/index.html +1 -1
- package/ui/dist/assets/index-Bcc4KMtO.js +0 -68
package/README.md
CHANGED
|
@@ -171,6 +171,37 @@ a human's judgement is required *and* being wrong is expensive.
|
|
|
171
171
|
phone with buttons. The default is what happens when you truly cannot
|
|
172
172
|
answer, not what happens because you never knew.
|
|
173
173
|
|
|
174
|
+
## Crew, and the reviewer gate
|
|
175
|
+
|
|
176
|
+
A mission's verification is normally the director's own word. A **crew preset**
|
|
177
|
+
is a named role you can put on a mission instead — a worker with a fixed brief,
|
|
178
|
+
a model of its own, and, for a reviewer, a read-only tool policy: it may read,
|
|
179
|
+
grep and search, and it may not write, edit or run a command. Two ship with
|
|
180
|
+
Foreman, a **Reviewer** and a **Security review**, and you edit them in
|
|
181
|
+
Settings → Crew, globally or per project.
|
|
182
|
+
|
|
183
|
+
You opt a mission in at compose time, with toggles under the model pickers.
|
|
184
|
+
The presets you pick are frozen onto the run, so editing a preset afterwards
|
|
185
|
+
cannot change a mission that already used it.
|
|
186
|
+
|
|
187
|
+
A reviewer marked **required for done** is a gate Foreman enforces, not a
|
|
188
|
+
request the director may skip:
|
|
189
|
+
|
|
190
|
+
- The director calls `request_review`, which hands the reviewer the mission
|
|
191
|
+
brief, the DONE WHEN criteria and the run's diff, and takes back a verdict —
|
|
192
|
+
`PASS` or `FAIL` — with findings.
|
|
193
|
+
- The verdict is pinned to a hash of the diff it actually read.
|
|
194
|
+
- At the end of the run, Foreman recomputes that hash. A run may be recorded
|
|
195
|
+
**done** only if every required reviewer's latest verdict is a PASS on the
|
|
196
|
+
diff the run *ends with*. Missing, FAIL, or a PASS that went stale because
|
|
197
|
+
the code moved afterwards: the run is `interrupted`, and says which reviewer
|
|
198
|
+
and why. A run stopped at its budget cap obeys the same rule.
|
|
199
|
+
|
|
200
|
+
So the reviewer goes last. It costs what a worker costs and comes out of the
|
|
201
|
+
same budget. Its verdicts show up in the transcript, on the run, in the run
|
|
202
|
+
report on your phone and over `foreman mcp`, and in the body of the pull
|
|
203
|
+
request Foreman drafts.
|
|
204
|
+
|
|
174
205
|
## Schedules
|
|
175
206
|
|
|
176
207
|
A schedule is a mission that starts itself. It lives in Foreman, not in the
|
|
@@ -268,7 +299,7 @@ result.
|
|
|
268
299
|
```sh
|
|
269
300
|
npm ci && npm run setup # dependencies, then the dashboard build
|
|
270
301
|
npm start # serves http://localhost:4177
|
|
271
|
-
npm test #
|
|
302
|
+
npm test # 466 tests, node:test
|
|
272
303
|
npm run typecheck # server and dashboard
|
|
273
304
|
npm run dev # API + Vite together
|
|
274
305
|
scripts/dev-restart.sh # restarts the server only when nothing would be lost
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@amenophis1er/foreman",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.18",
|
|
4
4
|
"description": "Autonomous mission runner on the Claude Agent SDK: a director plans, delegates to workers, verifies, and reports — from one dashboard, your phone, or the CLI.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"claude",
|
package/src/crew.test.ts
ADDED
|
@@ -0,0 +1,376 @@
|
|
|
1
|
+
import { test } from 'node:test';
|
|
2
|
+
import assert from 'node:assert/strict';
|
|
3
|
+
import {
|
|
4
|
+
BUILT_IN_PRESETS,
|
|
5
|
+
REVIEWER_TOOL_POLICY,
|
|
6
|
+
normalizePresets,
|
|
7
|
+
crewPresetsFrom,
|
|
8
|
+
freezeCrew,
|
|
9
|
+
parseVerdict,
|
|
10
|
+
diffHash,
|
|
11
|
+
reviewBlockers,
|
|
12
|
+
unreviewedText,
|
|
13
|
+
reviewBriefFor,
|
|
14
|
+
type CrewPreset,
|
|
15
|
+
type ReviewVerdict,
|
|
16
|
+
} from './crew.js';
|
|
17
|
+
|
|
18
|
+
/** A preset with only the fields a given test cares about spelled out. */
|
|
19
|
+
function preset(over: Partial<CrewPreset> = {}): CrewPreset {
|
|
20
|
+
return { id: 'reviewer', name: 'Reviewer', kind: 'reviewer', brief: 'look hard', requiredForDone: true, ...over };
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
/** A verdict about diff 'h1' unless told otherwise. */
|
|
24
|
+
function verdict(over: Partial<ReviewVerdict> = {}): ReviewVerdict {
|
|
25
|
+
return {
|
|
26
|
+
presetId: 'reviewer', name: 'Reviewer', pass: true, findings: '',
|
|
27
|
+
diffHash: 'h1', workerId: 'w1', at: 1000, ...over,
|
|
28
|
+
};
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
/** A deck file. */
|
|
32
|
+
function file(over: Partial<{ path: string; status: string; additions: number; deletions: number; diff: string }> = {}) {
|
|
33
|
+
return { path: 'src/a.ts', status: 'modified', additions: 3, deletions: 1, diff: '@@ -1 +1 @@\n-a\n+b', ...over };
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
test('BUILT_IN_PRESETS: two reviewers, only the first one gates', () => {
|
|
37
|
+
assert.deepEqual(BUILT_IN_PRESETS.map((p) => p.id), ['reviewer', 'security-review']);
|
|
38
|
+
assert.deepEqual(BUILT_IN_PRESETS.map((p) => p.requiredForDone), [true, false]);
|
|
39
|
+
for (const p of BUILT_IN_PRESETS) {
|
|
40
|
+
assert.equal(p.kind, 'reviewer');
|
|
41
|
+
assert.equal(p.model, 'opus');
|
|
42
|
+
assert.equal(p.toolPolicy, 'read-only');
|
|
43
|
+
assert.ok(p.brief.length > 40, `${p.id} needs a real brief`);
|
|
44
|
+
}
|
|
45
|
+
// Shared by every project on the server, so nobody gets to edit them in place.
|
|
46
|
+
assert.ok(Object.isFrozen(BUILT_IN_PRESETS));
|
|
47
|
+
assert.throws(() => { (BUILT_IN_PRESETS[0] as CrewPreset).requiredForDone = false; });
|
|
48
|
+
});
|
|
49
|
+
|
|
50
|
+
test('REVIEWER_TOOL_POLICY: the four writing tools, denied flat', () => {
|
|
51
|
+
assert.deepEqual({ ...REVIEWER_TOOL_POLICY }, {
|
|
52
|
+
Write: 'deny', Edit: 'deny', NotebookEdit: 'deny', Bash: 'deny',
|
|
53
|
+
});
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
test('parseVerdict: PASS and FAIL, with everything after the line as findings', () => {
|
|
57
|
+
assert.deepEqual(parseVerdict('VERDICT: PASS'), { pass: true, findings: '' });
|
|
58
|
+
assert.deepEqual(parseVerdict('looked at it\nVERDICT: FAIL\n\nsrc/a.ts:12 no test\n'), {
|
|
59
|
+
pass: false, findings: 'src/a.ts:12 no test',
|
|
60
|
+
});
|
|
61
|
+
// Surrounding whitespace on the line itself is tolerated.
|
|
62
|
+
assert.deepEqual(parseVerdict(' VERDICT: PASS \nfine'), { pass: true, findings: 'fine' });
|
|
63
|
+
});
|
|
64
|
+
|
|
65
|
+
test('parseVerdict: lowercase and markdown decoration still count', () => {
|
|
66
|
+
assert.deepEqual(parseVerdict('verdict: pass'), { pass: true, findings: '' });
|
|
67
|
+
assert.deepEqual(parseVerdict('**VERDICT: FAIL**\nbad'), { pass: false, findings: 'bad' });
|
|
68
|
+
assert.deepEqual(parseVerdict('## VERDICT: PASS'), { pass: true, findings: '' });
|
|
69
|
+
assert.deepEqual(parseVerdict('### **verdict: Fail**'), { pass: false, findings: '' });
|
|
70
|
+
});
|
|
71
|
+
|
|
72
|
+
test('parseVerdict: no verdict is null, not FAIL', () => {
|
|
73
|
+
// The caller has to tell "the reviewer said no" from "the reviewer never
|
|
74
|
+
// answered" — they are different problems with different fixes.
|
|
75
|
+
assert.equal(parseVerdict(''), null);
|
|
76
|
+
assert.equal(parseVerdict('I ran out of turns before I could finish.'), null);
|
|
77
|
+
// Malformed is malformed: a word that is neither, and a verdict with nothing
|
|
78
|
+
// after the colon.
|
|
79
|
+
assert.equal(parseVerdict('VERDICT: MAYBE'), null);
|
|
80
|
+
assert.equal(parseVerdict('VERDICT:'), null);
|
|
81
|
+
assert.equal(parseVerdict('VERDICT: PASS with reservations'), null);
|
|
82
|
+
// Mentioned mid-sentence rather than on its own line.
|
|
83
|
+
assert.equal(parseVerdict('my VERDICT: PASS on this one'), null);
|
|
84
|
+
});
|
|
85
|
+
|
|
86
|
+
test('parseVerdict: the last verdict line wins', () => {
|
|
87
|
+
const report = [
|
|
88
|
+
'I will end with VERDICT: PASS or VERDICT: FAIL as instructed.',
|
|
89
|
+
'Now the review.',
|
|
90
|
+
'VERDICT: FAIL',
|
|
91
|
+
'src/a.ts:3 off by one',
|
|
92
|
+
].join('\n');
|
|
93
|
+
// The first two are inside a sentence, so they are not lines; the real one is.
|
|
94
|
+
assert.deepEqual(parseVerdict(report), { pass: false, findings: 'src/a.ts:3 off by one' });
|
|
95
|
+
|
|
96
|
+
// Two genuine verdict lines: the model changed its mind, and the one it
|
|
97
|
+
// ended on is the one it means.
|
|
98
|
+
assert.deepEqual(parseVerdict('VERDICT: PASS\nthen I looked again\nVERDICT: FAIL\nsrc/b.ts:9'), {
|
|
99
|
+
pass: false, findings: 'src/b.ts:9',
|
|
100
|
+
});
|
|
101
|
+
});
|
|
102
|
+
|
|
103
|
+
test('parseVerdict: a runaway report is cut and says so', () => {
|
|
104
|
+
const parsed = parseVerdict('VERDICT: FAIL\n' + 'x'.repeat(20000));
|
|
105
|
+
assert.ok(parsed);
|
|
106
|
+
assert.equal(parsed.pass, false);
|
|
107
|
+
assert.ok(parsed.findings.length < 8200, 'findings should be capped');
|
|
108
|
+
assert.match(parsed.findings, /truncated/);
|
|
109
|
+
// Just under the cap is kept whole.
|
|
110
|
+
const short = parseVerdict('VERDICT: PASS\n' + 'y'.repeat(100));
|
|
111
|
+
assert.equal(short?.findings, 'y'.repeat(100));
|
|
112
|
+
});
|
|
113
|
+
|
|
114
|
+
test('diffHash: deterministic, order-independent, and empty is legal', () => {
|
|
115
|
+
const a = file({ path: 'src/a.ts' });
|
|
116
|
+
const b = file({ path: 'src/b.ts', status: 'added', additions: 9, deletions: 0, diff: '+new' });
|
|
117
|
+
assert.equal(diffHash({ files: [a, b] }), diffHash({ files: [a, b] }));
|
|
118
|
+
// Same files, different order: the same run, so the same hash.
|
|
119
|
+
assert.equal(diffHash({ files: [a, b] }), diffHash({ files: [b, a] }));
|
|
120
|
+
assert.match(diffHash({ files: [a] }), /^[0-9a-f]{64}$/);
|
|
121
|
+
|
|
122
|
+
// No changes at all is a stable hash, not a crash.
|
|
123
|
+
assert.equal(diffHash({}), diffHash({ files: [] }));
|
|
124
|
+
assert.match(diffHash({}), /^[0-9a-f]{64}$/);
|
|
125
|
+
assert.notEqual(diffHash({}), diffHash({ files: [a] }));
|
|
126
|
+
});
|
|
127
|
+
|
|
128
|
+
test('diffHash: any change to a file moves the hash', () => {
|
|
129
|
+
const base = diffHash({ files: [file()] });
|
|
130
|
+
assert.notEqual(base, diffHash({ files: [file({ diff: '@@ -1 +1 @@\n-a\n+c' })] }));
|
|
131
|
+
assert.notEqual(base, diffHash({ files: [file({ path: 'src/z.ts' })] }));
|
|
132
|
+
assert.notEqual(base, diffHash({ files: [file({ status: 'deleted' })] }));
|
|
133
|
+
assert.notEqual(base, diffHash({ files: [file({ additions: 4 })] }));
|
|
134
|
+
assert.notEqual(base, diffHash({ files: [file({ deletions: 2 })] }));
|
|
135
|
+
// A missing diff body hashes as empty, and is not the same as any body.
|
|
136
|
+
const nodiff = { path: 'src/a.ts', status: 'modified', additions: 3, deletions: 1 };
|
|
137
|
+
assert.equal(diffHash({ files: [nodiff] }), diffHash({ files: [{ ...nodiff, diff: '' }] }));
|
|
138
|
+
assert.notEqual(diffHash({ files: [nodiff] }), base);
|
|
139
|
+
});
|
|
140
|
+
|
|
141
|
+
test('normalizePresets: null when it is not an array, [] when the human emptied it', () => {
|
|
142
|
+
// The whole reason this returns null: "nothing configured" and "nothing
|
|
143
|
+
// wanted" must not be the same answer.
|
|
144
|
+
for (const junk of [undefined, null, {}, 'reviewer', 7, true]) {
|
|
145
|
+
assert.equal(normalizePresets(junk), null);
|
|
146
|
+
}
|
|
147
|
+
assert.deepEqual(normalizePresets([]), []);
|
|
148
|
+
});
|
|
149
|
+
|
|
150
|
+
test('normalizePresets: junk entries are dropped, not repaired', () => {
|
|
151
|
+
const got = normalizePresets([
|
|
152
|
+
null, 7, 'reviewer', [], {},
|
|
153
|
+
{ id: 'no-name' },
|
|
154
|
+
{ name: 'no id' },
|
|
155
|
+
{ id: ' ', name: 'blank id' },
|
|
156
|
+
{ id: 'blank-name', name: ' ' },
|
|
157
|
+
{ id: 'ok', name: 'Ok' },
|
|
158
|
+
]);
|
|
159
|
+
assert.deepEqual(got?.map((p) => p.id), ['ok']);
|
|
160
|
+
});
|
|
161
|
+
|
|
162
|
+
test('normalizePresets: defaults filled, optionals only when they are real', () => {
|
|
163
|
+
const [p] = normalizePresets([{ id: 'a', name: 'A' }])!;
|
|
164
|
+
assert.deepEqual(p, { id: 'a', name: 'A', kind: 'reviewer', brief: '', requiredForDone: false });
|
|
165
|
+
|
|
166
|
+
const [q] = normalizePresets([{
|
|
167
|
+
id: 'b', name: 'B', kind: 'specialist', brief: 'do the thing',
|
|
168
|
+
model: 'sonnet', providerId: 'prov1', toolPolicy: 'default', requiredForDone: true,
|
|
169
|
+
}])!;
|
|
170
|
+
assert.deepEqual(q, {
|
|
171
|
+
id: 'b', name: 'B', kind: 'specialist', brief: 'do the thing',
|
|
172
|
+
model: 'sonnet', providerId: 'prov1', toolPolicy: 'default', requiredForDone: true,
|
|
173
|
+
});
|
|
174
|
+
|
|
175
|
+
// Unknown kind falls back to reviewer; empty strings are not values;
|
|
176
|
+
// anything short of a literal true is not consent to block a run.
|
|
177
|
+
const [r] = normalizePresets([{
|
|
178
|
+
id: 'c', name: 'C', kind: 'wizard', brief: 42,
|
|
179
|
+
model: '', providerId: ' ', toolPolicy: 'yolo', requiredForDone: 'yes',
|
|
180
|
+
}])!;
|
|
181
|
+
assert.deepEqual(r, { id: 'c', name: 'C', kind: 'reviewer', brief: '', requiredForDone: false });
|
|
182
|
+
assert.equal(normalizePresets([{ id: 'd', name: 'D', requiredForDone: 1 }])![0].requiredForDone, false);
|
|
183
|
+
});
|
|
184
|
+
|
|
185
|
+
test('normalizePresets: duplicate ids, first wins', () => {
|
|
186
|
+
const got = normalizePresets([
|
|
187
|
+
{ id: 'r', name: 'First', requiredForDone: true },
|
|
188
|
+
{ id: 'r', name: 'Second' },
|
|
189
|
+
{ id: 's', name: 'Other' },
|
|
190
|
+
])!;
|
|
191
|
+
assert.deepEqual(got.map((p) => [p.id, p.name]), [['r', 'First'], ['s', 'Other']]);
|
|
192
|
+
assert.equal(got[0].requiredForDone, true);
|
|
193
|
+
});
|
|
194
|
+
|
|
195
|
+
test('crewPresetsFrom: built-ins, then global, then the project', () => {
|
|
196
|
+
assert.deepEqual(crewPresetsFrom(undefined, undefined).map((p) => p.id), ['reviewer', 'security-review']);
|
|
197
|
+
assert.deepEqual(crewPresetsFrom({}, {}).map((p) => p.id), ['reviewer', 'security-review']);
|
|
198
|
+
|
|
199
|
+
const global = { crewPresets: [{ id: 'g', name: 'G' }, { id: 'h', name: 'H' }] };
|
|
200
|
+
assert.deepEqual(crewPresetsFrom(global, undefined).map((p) => p.id), ['g', 'h']);
|
|
201
|
+
assert.deepEqual(crewPresetsFrom(global, {}).map((p) => p.id), ['g', 'h']);
|
|
202
|
+
|
|
203
|
+
// The project's list REPLACES the global one — it does not merge, or 'g'
|
|
204
|
+
// would still be here and "I removed the reviewer on this project" would be
|
|
205
|
+
// unsayable.
|
|
206
|
+
const project = { crewPresets: [{ id: 'p', name: 'P' }] };
|
|
207
|
+
assert.deepEqual(crewPresetsFrom(global, project).map((p) => p.id), ['p']);
|
|
208
|
+
// Including down to nothing.
|
|
209
|
+
assert.deepEqual(crewPresetsFrom(global, { crewPresets: [] }), []);
|
|
210
|
+
assert.deepEqual(crewPresetsFrom({ crewPresets: [] }, undefined), []);
|
|
211
|
+
// A project blob with junk in crewPresets is "not configured", so the global
|
|
212
|
+
// list still speaks.
|
|
213
|
+
assert.deepEqual(crewPresetsFrom(global, { crewPresets: 'reviewer' }).map((p) => p.id), ['g', 'h']);
|
|
214
|
+
});
|
|
215
|
+
|
|
216
|
+
test('crewPresetsFrom: hands out copies of the built-ins', () => {
|
|
217
|
+
const crew = crewPresetsFrom(undefined, undefined);
|
|
218
|
+
crew[0].requiredForDone = false;
|
|
219
|
+
crew[0].name = 'Tampered';
|
|
220
|
+
assert.equal(BUILT_IN_PRESETS[0].requiredForDone, true);
|
|
221
|
+
assert.equal(BUILT_IN_PRESETS[0].name, 'Reviewer');
|
|
222
|
+
assert.equal(crewPresetsFrom(undefined, undefined)[0].requiredForDone, true);
|
|
223
|
+
});
|
|
224
|
+
|
|
225
|
+
test('freezeCrew: presets order, deduped, unknown ids ignored', () => {
|
|
226
|
+
const presets = [preset({ id: 'a', name: 'A' }), preset({ id: 'b', name: 'B' }), preset({ id: 'c', name: 'C' })];
|
|
227
|
+
// The order of `presets` decides, not the order of `ids`.
|
|
228
|
+
assert.deepEqual(freezeCrew(['c', 'a'], presets).map((p) => p.id), ['a', 'c']);
|
|
229
|
+
assert.deepEqual(freezeCrew(['a', 'a', 'b'], presets).map((p) => p.id), ['a', 'b']);
|
|
230
|
+
assert.deepEqual(freezeCrew(['nope'], presets), []);
|
|
231
|
+
assert.deepEqual(freezeCrew([], presets), []);
|
|
232
|
+
assert.deepEqual(freezeCrew(['a', 'ghost'], presets).map((p) => p.id), ['a']);
|
|
233
|
+
});
|
|
234
|
+
|
|
235
|
+
test('freezeCrew: the run keeps a copy, so a later edit cannot rewrite its gate', () => {
|
|
236
|
+
const source = preset({ id: 'a', name: 'A', model: 'opus', providerId: 'prov', toolPolicy: 'read-only' });
|
|
237
|
+
const [frozen] = freezeCrew(['a'], [source]);
|
|
238
|
+
assert.deepEqual(frozen, source);
|
|
239
|
+
assert.notEqual(frozen, source);
|
|
240
|
+
frozen.requiredForDone = false;
|
|
241
|
+
frozen.brief = 'do nothing';
|
|
242
|
+
frozen.name = 'Tampered';
|
|
243
|
+
assert.equal(source.requiredForDone, true);
|
|
244
|
+
assert.equal(source.brief, 'look hard');
|
|
245
|
+
assert.equal(source.name, 'A');
|
|
246
|
+
// And the other way: editing the preset afterwards does not reach the run.
|
|
247
|
+
source.model = 'haiku';
|
|
248
|
+
assert.equal(frozen.model, 'opus');
|
|
249
|
+
});
|
|
250
|
+
|
|
251
|
+
test('reviewBlockers: nothing required means nothing blocks', () => {
|
|
252
|
+
assert.deepEqual(reviewBlockers(undefined, undefined, 'h1'), []);
|
|
253
|
+
assert.deepEqual(reviewBlockers([], [], 'h1'), []);
|
|
254
|
+
// A reviewer the human did not mark as required never blocks, however badly
|
|
255
|
+
// it went.
|
|
256
|
+
const optional = [preset({ id: 'sec', name: 'Security review', requiredForDone: false })];
|
|
257
|
+
assert.deepEqual(reviewBlockers(optional, [], 'h1'), []);
|
|
258
|
+
assert.deepEqual(reviewBlockers(optional, [verdict({ presetId: 'sec', pass: false })], 'h1'), []);
|
|
259
|
+
});
|
|
260
|
+
|
|
261
|
+
test('reviewBlockers: missing, fail, stale, pass', () => {
|
|
262
|
+
const crew = [preset()];
|
|
263
|
+
assert.deepEqual(reviewBlockers(crew, [], 'h1'), [{ presetId: 'reviewer', name: 'Reviewer', reason: 'missing' }]);
|
|
264
|
+
// Someone else's verdict is not this reviewer's verdict.
|
|
265
|
+
assert.deepEqual(reviewBlockers(crew, [verdict({ presetId: 'other' })], 'h1'), [
|
|
266
|
+
{ presetId: 'reviewer', name: 'Reviewer', reason: 'missing' },
|
|
267
|
+
]);
|
|
268
|
+
assert.deepEqual(reviewBlockers(crew, [verdict({ pass: false })], 'h1'), [
|
|
269
|
+
{ presetId: 'reviewer', name: 'Reviewer', reason: 'fail' },
|
|
270
|
+
]);
|
|
271
|
+
// Passed, then the code moved: that PASS was about a diff that no longer
|
|
272
|
+
// exists.
|
|
273
|
+
assert.deepEqual(reviewBlockers(crew, [verdict({ diffHash: 'h0' })], 'h1'), [
|
|
274
|
+
{ presetId: 'reviewer', name: 'Reviewer', reason: 'stale' },
|
|
275
|
+
]);
|
|
276
|
+
assert.deepEqual(reviewBlockers(crew, [verdict()], 'h1'), []);
|
|
277
|
+
});
|
|
278
|
+
|
|
279
|
+
test('reviewBlockers: the latest verdict decides', () => {
|
|
280
|
+
const crew = [preset()];
|
|
281
|
+
// A later FAIL overrules an earlier PASS, whatever order they are stored in.
|
|
282
|
+
const passThenFail = [verdict({ at: 1 }), verdict({ at: 2, pass: false })];
|
|
283
|
+
assert.deepEqual(reviewBlockers(crew, passThenFail, 'h1').map((b) => b.reason), ['fail']);
|
|
284
|
+
assert.deepEqual(reviewBlockers(crew, [...passThenFail].reverse(), 'h1').map((b) => b.reason), ['fail']);
|
|
285
|
+
// And a re-review clears an earlier FAIL.
|
|
286
|
+
assert.deepEqual(reviewBlockers(crew, [verdict({ at: 1, pass: false }), verdict({ at: 2 })], 'h1'), []);
|
|
287
|
+
// Same millisecond: the one appended later is the later one.
|
|
288
|
+
assert.deepEqual(
|
|
289
|
+
reviewBlockers(crew, [verdict({ at: 5, pass: false }), verdict({ at: 5 })], 'h1'),
|
|
290
|
+
[],
|
|
291
|
+
);
|
|
292
|
+
assert.deepEqual(
|
|
293
|
+
reviewBlockers(crew, [verdict({ at: 5 }), verdict({ at: 5, pass: false })], 'h1').map((b) => b.reason),
|
|
294
|
+
['fail'],
|
|
295
|
+
);
|
|
296
|
+
});
|
|
297
|
+
|
|
298
|
+
test('reviewBlockers: several reviewers, in crew order', () => {
|
|
299
|
+
const crew = [
|
|
300
|
+
preset({ id: 'a', name: 'A' }),
|
|
301
|
+
preset({ id: 'sec', name: 'Security review', requiredForDone: false }),
|
|
302
|
+
preset({ id: 'b', name: 'B' }),
|
|
303
|
+
preset({ id: 'c', name: 'C' }),
|
|
304
|
+
];
|
|
305
|
+
const verdicts = [
|
|
306
|
+
verdict({ presetId: 'c', name: 'C' }), // passed the current diff
|
|
307
|
+
verdict({ presetId: 'b', name: 'B', pass: false }), // said no
|
|
308
|
+
verdict({ presetId: 'sec', pass: false }), // not required, so silent
|
|
309
|
+
// 'a' never answered.
|
|
310
|
+
];
|
|
311
|
+
assert.deepEqual(reviewBlockers(crew, verdicts, 'h1'), [
|
|
312
|
+
{ presetId: 'a', name: 'A', reason: 'missing' },
|
|
313
|
+
{ presetId: 'b', name: 'B', reason: 'fail' },
|
|
314
|
+
]);
|
|
315
|
+
// Move the diff and the one that had passed goes stale too.
|
|
316
|
+
assert.deepEqual(reviewBlockers(crew, verdicts, 'h2').map((b) => [b.presetId, b.reason]), [
|
|
317
|
+
['a', 'missing'], ['b', 'fail'], ['c', 'stale'],
|
|
318
|
+
]);
|
|
319
|
+
});
|
|
320
|
+
|
|
321
|
+
test('unreviewedText: one reviewer, named, with what is wrong with it', () => {
|
|
322
|
+
assert.equal(unreviewedText([]), '');
|
|
323
|
+
const one = unreviewedText([{ presetId: 'reviewer', name: 'Reviewer', reason: 'missing' }]);
|
|
324
|
+
assert.match(one, /^A required reviewer has not passed this run: Reviewer has not reviewed this run\./);
|
|
325
|
+
assert.match(one, /not done\. Resume to continue it\.$/);
|
|
326
|
+
assert.match(unreviewedText([{ presetId: 'r', name: 'Reviewer', reason: 'fail' }]), /Reviewer returned FAIL/);
|
|
327
|
+
assert.match(
|
|
328
|
+
unreviewedText([{ presetId: 'r', name: 'Reviewer', reason: 'stale' }]),
|
|
329
|
+
/Reviewer passed an earlier version of the diff; the code changed after it/,
|
|
330
|
+
);
|
|
331
|
+
});
|
|
332
|
+
|
|
333
|
+
test('unreviewedText: several reviewers, plural and joined', () => {
|
|
334
|
+
const text = unreviewedText([
|
|
335
|
+
{ presetId: 'a', name: 'Reviewer', reason: 'missing' },
|
|
336
|
+
{ presetId: 'b', name: 'Security review', reason: 'fail' },
|
|
337
|
+
{ presetId: 'c', name: 'Perf', reason: 'stale' },
|
|
338
|
+
]);
|
|
339
|
+
assert.match(text, /^Required reviewers have not passed this run:/);
|
|
340
|
+
assert.match(text, /Reviewer has not reviewed this run; Security review returned FAIL; and Perf passed an earlier version/);
|
|
341
|
+
assert.match(text, /This run is not done\. Resume to continue it\.$/);
|
|
342
|
+
// Singular and plural are the only difference in the lead.
|
|
343
|
+
assert.ok(!/Required reviewers/.test(unreviewedText([{ presetId: 'a', name: 'A', reason: 'fail' }])));
|
|
344
|
+
});
|
|
345
|
+
|
|
346
|
+
test('reviewBriefFor: carries the brief, the mission, the criteria, the diff and the format', () => {
|
|
347
|
+
const brief = reviewBriefFor(preset({ brief: 'Be demanding.' }), {
|
|
348
|
+
mission: 'Add crew presets',
|
|
349
|
+
doneWhen: '- [x] tests pass',
|
|
350
|
+
diff: '@@ -1 +1 @@\n+ok',
|
|
351
|
+
truncated: false,
|
|
352
|
+
});
|
|
353
|
+
assert.match(brief, /Be demanding\./);
|
|
354
|
+
assert.match(brief, /Add crew presets/);
|
|
355
|
+
assert.match(brief, /- \[x\] tests pass/);
|
|
356
|
+
assert.match(brief, /```diff\n@@ -1 \+1 @@\n\+ok\n```/);
|
|
357
|
+
assert.match(brief, /VERDICT: PASS/);
|
|
358
|
+
assert.match(brief, /VERDICT: FAIL/);
|
|
359
|
+
assert.match(brief, /file:line/);
|
|
360
|
+
// It is told plainly that it cannot write, so it does not burn the run
|
|
361
|
+
// discovering the denials one tool call at a time.
|
|
362
|
+
assert.match(brief, /cannot modify files/);
|
|
363
|
+
// Nothing about reading files directly, because it was given the whole diff.
|
|
364
|
+
assert.ok(!/cut short/.test(brief));
|
|
365
|
+
});
|
|
366
|
+
|
|
367
|
+
test('reviewBriefFor: a cut diff says so and points at the files', () => {
|
|
368
|
+
const brief = reviewBriefFor(preset(), { mission: 'm', doneWhen: 'd', diff: 'x', truncated: true });
|
|
369
|
+
assert.match(brief, /cut short/);
|
|
370
|
+
assert.match(brief, /Read, Glob and Grep/);
|
|
371
|
+
// An empty mission or criteria section is labelled rather than left blank,
|
|
372
|
+
// so the reviewer is not left guessing whether it was dropped.
|
|
373
|
+
const bare = reviewBriefFor(preset(), { mission: '', doneWhen: ' ', diff: '', truncated: false });
|
|
374
|
+
assert.match(bare, /no brief was recorded/);
|
|
375
|
+
assert.match(bare, /no criteria were recorded/);
|
|
376
|
+
});
|