@amenophis1er/foreman 0.1.16 → 0.1.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -3
- package/package.json +1 -1
- package/src/crew.test.ts +376 -0
- package/src/crew.ts +330 -0
- package/src/fleet-planner.test.ts +39 -1
- package/src/fleet-planner.ts +83 -3
- package/src/gitwork.test.ts +133 -2
- package/src/gitwork.ts +211 -2
- package/src/mcp.test.ts +38 -1
- package/src/mcp.ts +86 -5
- package/src/notify/commands.test.ts +2 -0
- package/src/notify/commands.ts +6 -0
- package/src/notify/telegram.ts +1 -0
- package/src/notify.test.ts +61 -0
- package/src/notify.ts +45 -1
- package/src/orchestrator.test.ts +516 -2
- package/src/orchestrator.ts +514 -69
- package/src/run-crew.test.ts +99 -0
- package/src/run-crew.ts +101 -0
- package/src/schedule-guards.test.ts +236 -0
- package/src/schedule-guards.ts +149 -0
- package/src/schedule.test.ts +240 -0
- package/src/schedule.ts +343 -0
- package/src/server.ts +674 -11
- package/src/store.test.ts +81 -1
- package/src/store.ts +117 -3
- package/src/types.ts +92 -0
- package/ui/dist/assets/index-0QuGXbFg.js +76 -0
- package/ui/dist/index.html +1 -1
- package/ui/dist/assets/index-DOVnExqF.js +0 -68
package/src/crew.ts
ADDED
|
@@ -0,0 +1,330 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Crew presets, and the reviewer gate that stands between a run and "done".
|
|
3
|
+
*
|
|
4
|
+
* A mission that runs unattended has nobody reading its diff. The director is
|
|
5
|
+
* the same kind of agent as the workers, judging its own homework, and asking
|
|
6
|
+
* it nicely to fetch a second opinion is worth exactly as much as any other
|
|
7
|
+
* instruction in a prompt. So a crew preset — a named role a human opts into
|
|
8
|
+
* at compose time — can be marked `requiredForDone`, and Foreman itself refuses
|
|
9
|
+
* to record the run as done until that reviewer has returned PASS on the diff
|
|
10
|
+
* the run actually ends with.
|
|
11
|
+
*
|
|
12
|
+
* Everything here is pure on purpose: the gate is a rule about a run, not a
|
|
13
|
+
* conversation with a model, and a rule that decides whether work counts as
|
|
14
|
+
* finished has to be provable by a test rather than observed in a log.
|
|
15
|
+
*/
|
|
16
|
+
import { createHash } from 'node:crypto';
|
|
17
|
+
import type { ModelChoice, ToolPolicy } from './types.js';
|
|
18
|
+
|
|
19
|
+
export type CrewKind = 'reviewer' | 'specialist';
|
|
20
|
+
|
|
21
|
+
export interface CrewPreset {
|
|
22
|
+
id: string;
|
|
23
|
+
name: string;
|
|
24
|
+
kind: CrewKind;
|
|
25
|
+
model?: ModelChoice;
|
|
26
|
+
providerId?: string;
|
|
27
|
+
brief: string;
|
|
28
|
+
toolPolicy?: 'read-only' | 'default';
|
|
29
|
+
requiredForDone: boolean;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
/** One reviewer's answer about one diff. `at` is what makes a later one later. */
|
|
33
|
+
export interface ReviewVerdict {
|
|
34
|
+
presetId: string;
|
|
35
|
+
name: string;
|
|
36
|
+
pass: boolean;
|
|
37
|
+
findings: string;
|
|
38
|
+
diffHash: string;
|
|
39
|
+
workerId: string;
|
|
40
|
+
costUsd?: number;
|
|
41
|
+
at: number;
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
const REVIEWER_BRIEF = [
|
|
45
|
+
'You are the reviewer on this mission. Review the diff below against MISSION.md and the',
|
|
46
|
+
'mission brief the way a demanding senior engineer reviews a colleague\'s pull request.',
|
|
47
|
+
'Look for correctness first — does this actually do what was asked, and does it break',
|
|
48
|
+
'anything that already worked. Then missing or shallow tests, security, and anything the',
|
|
49
|
+
'DONE WHEN criteria did not name but a careful reader would insist on.',
|
|
50
|
+
'The verdict is PASS only when you would merge this yourself.',
|
|
51
|
+
].join(' ');
|
|
52
|
+
|
|
53
|
+
const SECURITY_BRIEF = [
|
|
54
|
+
'You are the security reviewer on this mission. Review the diff below for the ways this',
|
|
55
|
+
'change could be abused: secrets or tokens committed or logged, injection of any kind',
|
|
56
|
+
'(shell, SQL, HTML, prompt), path handling that can escape its intended root, and',
|
|
57
|
+
'permissions — anything newly reachable without authentication, or granted more access',
|
|
58
|
+
'than the task needs. Ignore style; say plainly what an attacker could do.',
|
|
59
|
+
'The verdict is PASS only when you would ship this yourself.',
|
|
60
|
+
].join(' ');
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* The crew a human sees before they have edited anything. Defaults, not law:
|
|
64
|
+
* the moment they change one, their list replaces this one whole.
|
|
65
|
+
*
|
|
66
|
+
* Frozen, and every path out of this module hands back copies, because these
|
|
67
|
+
* objects are shared by every project on the server — a caller that "just"
|
|
68
|
+
* flipped requiredForDone on one would be rewriting the gate for all of them.
|
|
69
|
+
*/
|
|
70
|
+
export const BUILT_IN_PRESETS: readonly CrewPreset[] = Object.freeze([
|
|
71
|
+
Object.freeze({
|
|
72
|
+
id: 'reviewer',
|
|
73
|
+
name: 'Reviewer',
|
|
74
|
+
kind: 'reviewer' as const,
|
|
75
|
+
model: 'opus',
|
|
76
|
+
toolPolicy: 'read-only' as const,
|
|
77
|
+
requiredForDone: true,
|
|
78
|
+
brief: REVIEWER_BRIEF,
|
|
79
|
+
}),
|
|
80
|
+
Object.freeze({
|
|
81
|
+
id: 'security-review',
|
|
82
|
+
name: 'Security review',
|
|
83
|
+
kind: 'reviewer' as const,
|
|
84
|
+
model: 'opus',
|
|
85
|
+
toolPolicy: 'read-only' as const,
|
|
86
|
+
requiredForDone: false,
|
|
87
|
+
brief: SECURITY_BRIEF,
|
|
88
|
+
}),
|
|
89
|
+
]) as readonly CrewPreset[];
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* What "read-only" means for a reviewer: these four tools denied outright, and
|
|
93
|
+
* nothing else configured.
|
|
94
|
+
*
|
|
95
|
+
* The alternative — letting Bash through and classifying each command — is the
|
|
96
|
+
* kind of rule that is right until someone writes `sh -c 'cat > f'`. A flat
|
|
97
|
+
* deny is something a test can prove. It costs the reviewer nothing it needs:
|
|
98
|
+
* AUTO_ALLOW_TOOLS in src/policy.ts already lets Read/Glob/Grep/WebFetch/
|
|
99
|
+
* WebSearch through, and the diff it is judging arrives in its brief, so the
|
|
100
|
+
* only thing it loses is the ability to change the code it is judging.
|
|
101
|
+
*/
|
|
102
|
+
export const REVIEWER_TOOL_POLICY: ToolPolicy = Object.freeze({
|
|
103
|
+
Write: 'deny',
|
|
104
|
+
Edit: 'deny',
|
|
105
|
+
NotebookEdit: 'deny',
|
|
106
|
+
Bash: 'deny',
|
|
107
|
+
}) as ToolPolicy;
|
|
108
|
+
|
|
109
|
+
function copy(p: CrewPreset): CrewPreset {
|
|
110
|
+
const out: CrewPreset = { id: p.id, name: p.name, kind: p.kind, brief: p.brief, requiredForDone: p.requiredForDone };
|
|
111
|
+
if (p.model) out.model = p.model;
|
|
112
|
+
if (p.providerId) out.providerId = p.providerId;
|
|
113
|
+
if (p.toolPolicy) out.toolPolicy = p.toolPolicy;
|
|
114
|
+
return out;
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
function nonEmptyString(v: unknown): string | undefined {
|
|
118
|
+
return typeof v === 'string' && v.trim() !== '' ? v : undefined;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/**
|
|
122
|
+
* A `crewPresets` value read off settings JSON, which nobody validated on the
|
|
123
|
+
* way in. Anything shaped wrongly is dropped rather than repaired: a preset
|
|
124
|
+
* with no id cannot be referenced by a run, and a preset with no name has
|
|
125
|
+
* nothing to put in the sentence that blocks the run.
|
|
126
|
+
*
|
|
127
|
+
* null and [] mean different things, and that difference is why this returns
|
|
128
|
+
* null at all: null is "not configured, fall back", [] is "the human deleted
|
|
129
|
+
* every preset", and answering [] with the built-ins would resurrect a
|
|
130
|
+
* reviewer they had just removed.
|
|
131
|
+
*/
|
|
132
|
+
export function normalizePresets(raw: unknown): CrewPreset[] | null {
|
|
133
|
+
if (!Array.isArray(raw)) return null;
|
|
134
|
+
const out: CrewPreset[] = [];
|
|
135
|
+
const seen = new Set<string>();
|
|
136
|
+
for (const entry of raw) {
|
|
137
|
+
if (!entry || typeof entry !== 'object' || Array.isArray(entry)) continue;
|
|
138
|
+
const e = entry as Record<string, unknown>;
|
|
139
|
+
const id = nonEmptyString(e.id);
|
|
140
|
+
const name = nonEmptyString(e.name);
|
|
141
|
+
if (!id || !name || seen.has(id)) continue;
|
|
142
|
+
seen.add(id);
|
|
143
|
+
const preset: CrewPreset = {
|
|
144
|
+
id,
|
|
145
|
+
name,
|
|
146
|
+
// A preset of an unknown kind is still a worker with a brief; reviewer is
|
|
147
|
+
// the safe reading, since the only thing kind decides is how it is run.
|
|
148
|
+
kind: e.kind === 'specialist' ? 'specialist' : 'reviewer',
|
|
149
|
+
brief: typeof e.brief === 'string' ? e.brief : '',
|
|
150
|
+
// Anything other than a literal true is not consent to block a run.
|
|
151
|
+
requiredForDone: e.requiredForDone === true,
|
|
152
|
+
};
|
|
153
|
+
const model = nonEmptyString(e.model);
|
|
154
|
+
if (model) preset.model = model;
|
|
155
|
+
const providerId = nonEmptyString(e.providerId);
|
|
156
|
+
if (providerId) preset.providerId = providerId;
|
|
157
|
+
if (e.toolPolicy === 'read-only' || e.toolPolicy === 'default') preset.toolPolicy = e.toolPolicy;
|
|
158
|
+
out.push(preset);
|
|
159
|
+
}
|
|
160
|
+
return out;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* The crew presets in force for a project. A project's list replaces the
|
|
165
|
+
* global one whole rather than merging with it — the same way the rest of
|
|
166
|
+
* Settings overlays — because a merge would make "I removed the reviewer here"
|
|
167
|
+
* unsayable.
|
|
168
|
+
*/
|
|
169
|
+
export function crewPresetsFrom(global: unknown, project: unknown): CrewPreset[] {
|
|
170
|
+
const read = (blob: unknown): CrewPreset[] | null => {
|
|
171
|
+
if (!blob || typeof blob !== 'object') return null;
|
|
172
|
+
return normalizePresets((blob as Record<string, unknown>).crewPresets);
|
|
173
|
+
};
|
|
174
|
+
return (read(project) ?? read(global) ?? BUILT_IN_PRESETS).map(copy);
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/**
|
|
178
|
+
* The crew a run starts with, copied onto the run record. Deep copies, because
|
|
179
|
+
* editing a preset next week must not change what a run that is still going —
|
|
180
|
+
* or one that finished in March — was reviewed against.
|
|
181
|
+
*/
|
|
182
|
+
export function freezeCrew(ids: readonly string[], presets: readonly CrewPreset[]): CrewPreset[] {
|
|
183
|
+
const wanted = new Set(ids);
|
|
184
|
+
const taken = new Set<string>();
|
|
185
|
+
const out: CrewPreset[] = [];
|
|
186
|
+
for (const p of presets) {
|
|
187
|
+
if (!wanted.has(p.id) || taken.has(p.id)) continue;
|
|
188
|
+
taken.add(p.id);
|
|
189
|
+
out.push(copy(p));
|
|
190
|
+
}
|
|
191
|
+
return out;
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/** Past this, findings are a wall of text nobody reads and a row nobody wants to store. */
|
|
195
|
+
const MAX_FINDINGS = 8000;
|
|
196
|
+
|
|
197
|
+
/**
|
|
198
|
+
* The reviewer's verdict, dug out of its report. It is told to end with a line
|
|
199
|
+
* that is exactly `VERDICT: PASS` or `VERDICT: FAIL`; the last such line wins,
|
|
200
|
+
* because a report that quotes the format while explaining itself and then
|
|
201
|
+
* states its verdict at the end has stated it at the end.
|
|
202
|
+
*
|
|
203
|
+
* null is not FAIL. A report that never says either has not been reviewed —
|
|
204
|
+
* the model ran out of turns, or answered in prose — and the caller has to be
|
|
205
|
+
* able to tell that apart from a reviewer that looked and said no.
|
|
206
|
+
*/
|
|
207
|
+
export function parseVerdict(report: string): { pass: boolean; findings: string } | null {
|
|
208
|
+
if (typeof report !== 'string' || report === '') return null;
|
|
209
|
+
const lines = report.split('\n');
|
|
210
|
+
for (let i = lines.length - 1; i >= 0; i--) {
|
|
211
|
+
// Markdown decoration is what a model reaches for when told to make a line
|
|
212
|
+
// stand out; `**VERDICT: PASS**` is the instruction followed, not broken.
|
|
213
|
+
const line = lines[i].replace(/\*\*/g, '').replace(/^\s*#+\s*/, '');
|
|
214
|
+
const m = /^\s*VERDICT:\s*(PASS|FAIL)\s*$/i.exec(line);
|
|
215
|
+
if (!m) continue;
|
|
216
|
+
let findings = lines.slice(i + 1).join('\n').trim();
|
|
217
|
+
if (findings.length > MAX_FINDINGS) {
|
|
218
|
+
findings = findings.slice(0, MAX_FINDINGS) + '\n\n[findings truncated]';
|
|
219
|
+
}
|
|
220
|
+
return { pass: m[1].toUpperCase() === 'PASS', findings };
|
|
221
|
+
}
|
|
222
|
+
return null;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
/**
|
|
226
|
+
* A fingerprint of what a run has changed, so a PASS can be pinned to the diff
|
|
227
|
+
* it was given. Canonical: sorted by path and field-separated, so the same
|
|
228
|
+
* files listed in a different order are the same hash, while one changed line
|
|
229
|
+
* anywhere is a different one.
|
|
230
|
+
*/
|
|
231
|
+
export function diffHash(deck: {
|
|
232
|
+
files?: Array<{ path: string; status: string; additions: number; deletions: number; diff?: string }>;
|
|
233
|
+
}): string {
|
|
234
|
+
const files = deck?.files ?? [];
|
|
235
|
+
const canon = [...files]
|
|
236
|
+
.sort((a, b) => (a.path < b.path ? -1 : a.path > b.path ? 1 : 0))
|
|
237
|
+
.map((f) => [f.path, f.status, f.additions, f.deletions, f.diff ?? ''].join('\0'))
|
|
238
|
+
.join('\n');
|
|
239
|
+
// A run that changed nothing still has a stable hash, so "reviewed, then
|
|
240
|
+
// nothing moved" holds for an empty diff too.
|
|
241
|
+
return createHash('sha256').update(canon).digest('hex');
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
export interface ReviewBlocker {
|
|
245
|
+
presetId: string;
|
|
246
|
+
name: string;
|
|
247
|
+
reason: 'missing' | 'fail' | 'stale';
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
/**
|
|
251
|
+
* THE GATE. Every required reviewer that has not passed the run's current
|
|
252
|
+
* diff, in crew order. An empty array is the only thing that lets a run be
|
|
253
|
+
* recorded as done.
|
|
254
|
+
*
|
|
255
|
+
* 'stale' is the case worth the machinery: a reviewer passed, then the
|
|
256
|
+
* director kept working. That PASS was about code that no longer exists, and
|
|
257
|
+
* treating it as consent for whatever came afterwards would make the gate
|
|
258
|
+
* trivially walk-around-able — review early, then commit anything.
|
|
259
|
+
*/
|
|
260
|
+
export function reviewBlockers(
|
|
261
|
+
crew: readonly CrewPreset[] | undefined,
|
|
262
|
+
verdicts: readonly ReviewVerdict[] | undefined,
|
|
263
|
+
currentDiffHash: string,
|
|
264
|
+
): ReviewBlocker[] {
|
|
265
|
+
const out: ReviewBlocker[] = [];
|
|
266
|
+
for (const preset of crew ?? []) {
|
|
267
|
+
if (preset.requiredForDone !== true) continue;
|
|
268
|
+
// Latest by `at`, ties going to the one that arrived later in the array —
|
|
269
|
+
// two verdicts written in the same millisecond are in the order they were
|
|
270
|
+
// appended.
|
|
271
|
+
let latest: ReviewVerdict | undefined;
|
|
272
|
+
for (const v of verdicts ?? []) {
|
|
273
|
+
if (v.presetId !== preset.id) continue;
|
|
274
|
+
if (!latest || v.at >= latest.at) latest = v;
|
|
275
|
+
}
|
|
276
|
+
if (!latest) out.push({ presetId: preset.id, name: preset.name, reason: 'missing' });
|
|
277
|
+
else if (!latest.pass) out.push({ presetId: preset.id, name: preset.name, reason: 'fail' });
|
|
278
|
+
else if (latest.diffHash !== currentDiffHash) out.push({ presetId: preset.id, name: preset.name, reason: 'stale' });
|
|
279
|
+
}
|
|
280
|
+
return out;
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
/** One human sentence for the `mission_unreviewed` event. */
|
|
284
|
+
export function unreviewedText(blockers: readonly ReviewBlocker[]): string {
|
|
285
|
+
if (!blockers.length) return '';
|
|
286
|
+
const clause = (b: ReviewBlocker): string => {
|
|
287
|
+
if (b.reason === 'fail') return `${b.name} returned FAIL`;
|
|
288
|
+
if (b.reason === 'stale') return `${b.name} passed an earlier version of the diff; the code changed after it`;
|
|
289
|
+
return `${b.name} has not reviewed this run`;
|
|
290
|
+
};
|
|
291
|
+
const parts = blockers.map(clause);
|
|
292
|
+
const list = parts.length === 1
|
|
293
|
+
? parts[0]
|
|
294
|
+
: `${parts.slice(0, -1).join('; ')}; and ${parts[parts.length - 1]}`;
|
|
295
|
+
const lead = blockers.length === 1 ? 'A required reviewer has not passed this run' : 'Required reviewers have not passed this run';
|
|
296
|
+
return `${lead}: ${list}. This run is not done. Resume to continue it.`;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
/**
|
|
300
|
+
* The prompt the reviewer worker is started with. The diff is handed over in
|
|
301
|
+
* the brief rather than left to be discovered, because a read-only worker with
|
|
302
|
+
* no Bash cannot run `git diff` for itself — and because the thing being
|
|
303
|
+
* reviewed should be the thing Foreman will hash, not whatever the model
|
|
304
|
+
* happened to look at.
|
|
305
|
+
*/
|
|
306
|
+
export function reviewBriefFor(
|
|
307
|
+
preset: CrewPreset,
|
|
308
|
+
opts: { mission: string; doneWhen: string; diff: string; truncated: boolean },
|
|
309
|
+
): string {
|
|
310
|
+
const parts: string[] = [preset.brief.trim()];
|
|
311
|
+
parts.push(`## The mission\n\n${opts.mission.trim() || '(no brief was recorded)'}`);
|
|
312
|
+
parts.push(`## DONE WHEN\n\n${opts.doneWhen.trim() || '(no criteria were recorded)'}`);
|
|
313
|
+
parts.push(`## The diff\n\n\`\`\`diff\n${opts.diff}\n\`\`\``);
|
|
314
|
+
if (opts.truncated) {
|
|
315
|
+
parts.push(
|
|
316
|
+
'That diff was cut short because it is large. Read the files you need directly — ' +
|
|
317
|
+
'Read, Glob and Grep are available to you.',
|
|
318
|
+
);
|
|
319
|
+
}
|
|
320
|
+
parts.push(
|
|
321
|
+
'You cannot modify files: Write, Edit and Bash are denied to you, and attempting them ' +
|
|
322
|
+
'wastes the run. Report, do not fix.',
|
|
323
|
+
);
|
|
324
|
+
parts.push(
|
|
325
|
+
'End your report with a line that is exactly `VERDICT: PASS` or `VERDICT: FAIL`, and ' +
|
|
326
|
+
'then list your findings under it, each with a `file:line` where you can give one. ' +
|
|
327
|
+
'Only what follows that line is kept as your findings.',
|
|
328
|
+
);
|
|
329
|
+
return parts.join('\n\n');
|
|
330
|
+
}
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
import { test } from 'node:test';
|
|
2
2
|
import assert from 'node:assert/strict';
|
|
3
|
-
import { FLEET_CHAT_ID, PHONE_CONTEXT_MS, fleetSummary, phoneRoute, situation } from './fleet-planner.js';
|
|
3
|
+
import { FLEET_CHAT_ID, PHONE_CONTEXT_MS, fleetSummary, phoneRoute, scheduleSummary, scheduleTools, situation, type FleetHost, type FleetScheduleView } from './fleet-planner.js';
|
|
4
|
+
import type { Schedule } from './types.js';
|
|
4
5
|
|
|
5
6
|
test('plain phone text goes to the fleet planner when no project conversation is open', () => {
|
|
6
7
|
assert.equal(phoneRoute(null), 'fleet');
|
|
@@ -41,6 +42,43 @@ test('fleetSummary says what is running, what is waiting, and what finished last
|
|
|
41
42
|
assert.equal(fleetSummary([]), 'No projects are linked yet.');
|
|
42
43
|
});
|
|
43
44
|
|
|
45
|
+
const schedule = (o: Partial<Schedule> = {}): Schedule => ({
|
|
46
|
+
id: 's1', projectId: 'a', name: 'nightly deps', brief: 'Update the dependencies',
|
|
47
|
+
cadence: { kind: 'daily', at: '07:30' }, budgetUsd: 3, enabled: true, createdAt: 0,
|
|
48
|
+
nextRunAt: null, consecutiveFailures: 0, pausedReason: null, ...o,
|
|
49
|
+
});
|
|
50
|
+
|
|
51
|
+
test('scheduleSummary says the cadence in words, the next run both ways, and why one is paused', () => {
|
|
52
|
+
const now = 1_700_000_000_000;
|
|
53
|
+
const views: FleetScheduleView[] = [
|
|
54
|
+
{ projectName: 'P5', monthSpendUsd: 4.2, monthlyCapUsd: 25, schedule: schedule({ nextRunAt: now + 15 * 3_600_000, lastOutcome: 'done', lastRunId: 'r9', lastRunAt: now - 9 * 3_600_000 }) },
|
|
55
|
+
{ projectName: 'P7', schedule: schedule({ id: 's2', name: 'audit', cadence: { kind: 'interval', everyMinutes: 360 }, budgetUsd: 2, enabled: false, pausedReason: 'failures', consecutiveFailures: 2 }) },
|
|
56
|
+
{ projectName: 'P7', schedule: schedule({ id: 's3', name: 'weekly report', cadence: { kind: 'weekly', day: 1, at: '09:00' }, enabled: false, pausedReason: 'monthly-cap' }) },
|
|
57
|
+
];
|
|
58
|
+
const text = scheduleSummary(views, now);
|
|
59
|
+
assert.match(text, /- P5: "nightly deps" · daily 07:30 · next .* \(in 15 h\)\n enabled · \$3\.00 per run · last done \(r9\) 9 h ago\n scheduled spend this month: \$4\.20 of \$25\.00/);
|
|
60
|
+
assert.match(text, /- P7: "audit" · every 6 hours · no next run while paused\n paused after 2 failed scheduled runs in a row · \$2\.00 per run · never run yet/);
|
|
61
|
+
assert.match(text, /- P7: "weekly report" · every Monday 09:00 .*\n paused at the project's monthly cap for scheduled spend/);
|
|
62
|
+
assert.match(text, /read-only from here: created, edited, paused and resumed on the dashboard/);
|
|
63
|
+
assert.match(scheduleSummary([]), /No schedules\. They are created on the dashboard/);
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
test('the front desk reads schedules and has no tool that changes one', async () => {
|
|
67
|
+
const asked: Array<string | undefined> = [];
|
|
68
|
+
const host = {
|
|
69
|
+
listSchedules: async (ref?: string) => { asked.push(ref); return [{ projectName: 'P5', schedule: schedule({ nextRunAt: Date.now() + 3_600_000 }) }]; },
|
|
70
|
+
} as unknown as FleetHost;
|
|
71
|
+
const tools = scheduleTools(host);
|
|
72
|
+
assert.deepEqual(tools.map((t) => t.name), ['list_schedules']);
|
|
73
|
+
assert.match(tools[0].description, /Read-only.*dashboard/s);
|
|
74
|
+
const out = await tools[0].handler({ project: 'P5' } as never, undefined);
|
|
75
|
+
assert.deepEqual(asked, ['P5']);
|
|
76
|
+
assert.match(String((out.content as Array<{ text: string }>)[0].text), /"nightly deps" · daily 07:30/);
|
|
77
|
+
// A host from before schedules existed says so instead of guessing.
|
|
78
|
+
const bare = await scheduleTools({} as FleetHost)[0].handler({} as never, undefined);
|
|
79
|
+
assert.match(String((bare.content as Array<{ text: string }>)[0].text), /cannot list schedules/);
|
|
80
|
+
});
|
|
81
|
+
|
|
44
82
|
test('situation carries the clock, the channel, and the news since the last message', () => {
|
|
45
83
|
const now = new Date(2026, 8, 6, 14, 5);
|
|
46
84
|
const s = situation({ via: 'telegram', sinceMs: 12 * 60_000, news: ['13:58 P7 — mission ended: done ($0.78)'] }, now);
|
package/src/fleet-planner.ts
CHANGED
|
@@ -14,8 +14,9 @@
|
|
|
14
14
|
* - **No resident process.** A message resumes a stored session, runs one
|
|
15
15
|
* turn, exits. Continuity is the session id on disk.
|
|
16
16
|
* - **Read-only, always.** Its tools are the fleet's verbs — list, inspect,
|
|
17
|
-
* create or link a project, open a planning conversation, propose, steer
|
|
18
|
-
* No shell, no file access, no starting missions
|
|
17
|
+
* create or link a project, open a planning conversation, propose, steer,
|
|
18
|
+
* read the schedules. No shell, no file access, no starting missions — and
|
|
19
|
+
* no change to a schedule, which is standing configuration.
|
|
19
20
|
* - **It never answers for the human.** Open approvals and questions are
|
|
20
21
|
* described, not resolved. The buttons on the card are the human's, and an
|
|
21
22
|
* agent that presses them is a hole through `canUseTool`.
|
|
@@ -27,7 +28,8 @@ import {
|
|
|
27
28
|
} from '@anthropic-ai/claude-agent-sdk';
|
|
28
29
|
import type { AgentEnv } from './provider.js';
|
|
29
30
|
import { modelsSection, needsBrowser, pickKnownModel, type PlannerModel } from './planner.js';
|
|
30
|
-
import
|
|
31
|
+
import { describeCadence } from './schedule.js';
|
|
32
|
+
import type { MissionProposal, Schedule } from './types.js';
|
|
31
33
|
|
|
32
34
|
/** Where the fleet conversation is stored, beside the project chats. The underscore keeps it out of project listings. */
|
|
33
35
|
export const FLEET_CHAT_ID = '_fleet';
|
|
@@ -85,6 +87,9 @@ const ago = (ms: number): string => {
|
|
|
85
87
|
return h < 48 ? `${h} h ago` : `${Math.round(h / 24)} days ago`;
|
|
86
88
|
};
|
|
87
89
|
|
|
90
|
+
/** The same distance, forwards: "in 15 h". */
|
|
91
|
+
const ahead = (ms: number): string => (ms < 60_000 ? 'in under a minute' : `in ${ago(ms).replace(/ ago$/, '')}`);
|
|
92
|
+
|
|
88
93
|
/** The fleet in plain lines, as the list_projects tool returns it. */
|
|
89
94
|
export function fleetSummary(views: FleetProjectView[], now = Date.now()): string {
|
|
90
95
|
if (!views.length) return 'No projects are linked yet.';
|
|
@@ -104,6 +109,68 @@ export function fleetSummary(views: FleetProjectView[], now = Date.now()): strin
|
|
|
104
109
|
}).join('\n');
|
|
105
110
|
}
|
|
106
111
|
|
|
112
|
+
/** One schedule as the front desk sees it: the record, plus whose it is. */
|
|
113
|
+
export interface FleetScheduleView {
|
|
114
|
+
projectName: string;
|
|
115
|
+
schedule: Schedule;
|
|
116
|
+
/** Scheduled spend this month against the project's ceiling, when the host knows it. */
|
|
117
|
+
monthSpendUsd?: number;
|
|
118
|
+
monthlyCapUsd?: number;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
/** Why a schedule is not going to fire, in the words that say what would undo it. */
|
|
122
|
+
function pausedPhrase(s: Schedule): string {
|
|
123
|
+
switch (s.pausedReason) {
|
|
124
|
+
case 'failures': return `paused after ${s.consecutiveFailures || 2} failed scheduled runs in a row`;
|
|
125
|
+
case 'monthly-cap': return 'paused at the project\'s monthly cap for scheduled spend';
|
|
126
|
+
case 'human': return 'paused by hand';
|
|
127
|
+
default: return s.enabled ? 'enabled' : 'disabled';
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
/**
|
|
132
|
+
* The schedules in plain lines, as the list_schedules tool returns them. The
|
|
133
|
+
* next run is said absolutely and relatively both: a schedule is a wall-clock
|
|
134
|
+
* promise, and "in 15 h" is what the person on the phone actually asked.
|
|
135
|
+
*/
|
|
136
|
+
export function scheduleSummary(views: FleetScheduleView[], now = Date.now()): string {
|
|
137
|
+
if (!views.length) return 'No schedules. They are created on the dashboard, in a project\'s view.';
|
|
138
|
+
const lines = views.map((v) => {
|
|
139
|
+
const s = v.schedule;
|
|
140
|
+
const next = s.pausedReason || !s.enabled ? 'no next run while paused'
|
|
141
|
+
: s.nextRunAt ? `next ${new Date(s.nextRunAt).toLocaleString()} (${ahead(s.nextRunAt - now)})`
|
|
142
|
+
: 'next never — this cadence has no future firing';
|
|
143
|
+
const last = s.lastOutcome
|
|
144
|
+
? `last ${s.lastOutcome}${s.lastRunId ? ` (${s.lastRunId})` : ''}${s.lastRunAt ? ` ${ago(now - s.lastRunAt)}` : ''}`
|
|
145
|
+
: 'never run yet';
|
|
146
|
+
const month = typeof v.monthSpendUsd === 'number' && typeof v.monthlyCapUsd === 'number'
|
|
147
|
+
? `\n scheduled spend this month: $${v.monthSpendUsd.toFixed(2)} of $${v.monthlyCapUsd.toFixed(2)}` : '';
|
|
148
|
+
return `- ${v.projectName}: "${s.name}" · ${describeCadence(s.cadence)} · ${next}\n ${pausedPhrase(s)} · $${s.budgetUsd.toFixed(2)} per run · ${last}${month}`;
|
|
149
|
+
});
|
|
150
|
+
lines.push('Schedules are read-only from here: created, edited, paused and resumed on the dashboard.');
|
|
151
|
+
return lines.join('\n');
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* The front desk's one schedule verb. Listing only, and there is no sibling:
|
|
156
|
+
* a schedule is standing configuration, and the phone never grants standing
|
|
157
|
+
* changes — the same rule that keeps "always allow" off the buttons.
|
|
158
|
+
*/
|
|
159
|
+
export function scheduleTools(host: FleetHost) {
|
|
160
|
+
return [
|
|
161
|
+
tool('list_schedules', 'The standing schedules — missions that start themselves on a cadence — for the whole fleet or one project: the cadence in words, the next run, enabled or paused and why, the per-run cap, and how the last firing ended. Read-only, and the only schedule tool: creating, editing, pausing, resuming or running one now is done on the dashboard, never from here.',
|
|
162
|
+
{ project: z.string().optional().describe('Project name, id, or folder name; omit for the whole fleet') },
|
|
163
|
+
async ({ project }) => {
|
|
164
|
+
if (!host.listSchedules) return text('This Foreman cannot list schedules.');
|
|
165
|
+
try {
|
|
166
|
+
return text(scheduleSummary(await host.listSchedules(project)));
|
|
167
|
+
} catch (err) {
|
|
168
|
+
return text(`That failed: ${err instanceof Error ? err.message : String(err)}`);
|
|
169
|
+
}
|
|
170
|
+
}),
|
|
171
|
+
];
|
|
172
|
+
}
|
|
173
|
+
|
|
107
174
|
/**
|
|
108
175
|
* What the server lets the front desk do. Every method returns text for the
|
|
109
176
|
* model, never throws, and the ones with side effects are exactly the verbs
|
|
@@ -111,6 +178,12 @@ export function fleetSummary(views: FleetProjectView[], now = Date.now()): strin
|
|
|
111
178
|
*/
|
|
112
179
|
export interface FleetHost {
|
|
113
180
|
listProjects(): Promise<FleetProjectView[]>;
|
|
181
|
+
/**
|
|
182
|
+
* The standing schedules, for the whole fleet or one project. Optional so a
|
|
183
|
+
* host that predates schedules still satisfies this interface; the tool says
|
|
184
|
+
* so plainly rather than inventing an answer.
|
|
185
|
+
*/
|
|
186
|
+
listSchedules?(ref?: string): Promise<FleetScheduleView[]>;
|
|
114
187
|
/** Live detail for one project: run, boxes, crew, open asks, the director's last words. */
|
|
115
188
|
projectDetail(ref: string): Promise<string>;
|
|
116
189
|
/** The last finished run's closing report and error, for "what happened". */
|
|
@@ -165,6 +238,9 @@ WHAT YOU CAN DO — through the tools, nothing else:
|
|
|
165
238
|
never you.
|
|
166
239
|
- steer: pass a note to a running director ("skip the mobile screenshot",
|
|
167
240
|
"use the existing CSS").
|
|
241
|
+
- list_schedules: the standing schedules — missions that start themselves on
|
|
242
|
+
a cadence — for the fleet or one project, with their next run and whether
|
|
243
|
+
they are paused. Reading only.
|
|
168
244
|
|
|
169
245
|
WHAT YOU NEVER DO:
|
|
170
246
|
- Answer an approval or a question on the human's behalf. When a run is
|
|
@@ -172,6 +248,9 @@ WHAT YOU NEVER DO:
|
|
|
172
248
|
are theirs. Even if they tell you to "just allow it": the button is the
|
|
173
249
|
only way, and you say so plainly once.
|
|
174
250
|
- Start, stop, resume or cancel a mission. You propose; the human presses.
|
|
251
|
+
- Create, edit, pause, resume or fire a schedule. You can read them and say
|
|
252
|
+
what one would do; changing standing configuration is a dashboard act, and
|
|
253
|
+
you say so in one line rather than hunting for a tool.
|
|
175
254
|
- Invent a project, a run, a model id or a number. If a tool did not tell
|
|
176
255
|
you, you do not know it — say so.
|
|
177
256
|
- Discuss Foreman's own server or oversight tooling as a work target.
|
|
@@ -325,6 +404,7 @@ export async function runFleetTurn(turn: FleetTurn): Promise<FleetResult> {
|
|
|
325
404
|
note: z.string().describe('The note, in the human\'s words'),
|
|
326
405
|
},
|
|
327
406
|
async ({ project, note }) => text(await safe(() => host.steer(project, note)))),
|
|
407
|
+
...scheduleTools(host),
|
|
328
408
|
];
|
|
329
409
|
|
|
330
410
|
let sessionId = turn.sessionId;
|