@amenophis1er/foreman 0.1.12 → 0.1.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -1
- package/bin/foreman.mjs +2 -1
- package/package.json +2 -1
- package/skills/director/SKILL.md +5 -0
- package/src/cli.ts +7 -0
- package/src/completion.ts +1 -0
- package/src/deck.ts +4 -2
- package/src/gitwork.test.ts +16 -1
- package/src/gitwork.ts +27 -1
- package/src/mcp.test.ts +181 -0
- package/src/mcp.ts +481 -0
- package/src/orchestrator.ts +28 -1
- package/src/server.ts +53 -5
- package/src/snapshot.test.ts +68 -0
- package/src/snapshot.ts +116 -0
- package/src/store.test.ts +16 -0
- package/src/store.ts +17 -0
- package/src/types.ts +9 -0
- package/ui/dist/assets/{index--dvfX3Mo.js → index-Ct_GJThl.js} +9 -9
- package/ui/dist/index.html +1 -1
package/src/mcp.ts
ADDED
|
@@ -0,0 +1,481 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `foreman mcp` — Foreman for other agents.
|
|
3
|
+
*
|
|
4
|
+
* A stdio MCP server another agent starts on demand (Claude Code, Codex,
|
|
5
|
+
* Antigravity: `<client> mcp add foreman -- foreman mcp`). Deliberately thin:
|
|
6
|
+
* every tool is a call to the running Foreman server over the same REST and
|
|
7
|
+
* event stream the dashboard uses, so policy lives in one place — the server
|
|
8
|
+
* — and this file has no business logic of its own.
|
|
9
|
+
*
|
|
10
|
+
* What it offers is what an agent watching or launching missions needs: the
|
|
11
|
+
* fleet, runs, a run's status with a `wait` (one call that returns when
|
|
12
|
+
* something changes, instead of a polling loop), the transcript, the mission
|
|
13
|
+
* doc and memory, linking a project, starting a mission, steering a director.
|
|
14
|
+
*
|
|
15
|
+
* What it does not offer, on purpose: approving or denying, answering the
|
|
16
|
+
* director's questions, interrupt, resume, raising a budget, opening a pull
|
|
17
|
+
* request, settings and keys. Those are the moments Foreman exists to put a
|
|
18
|
+
* human in; `run_status` says when a run needs one, and with what, so the
|
|
19
|
+
* agent's job is to send the human to decide, not to decide.
|
|
20
|
+
*/
|
|
21
|
+
import fs from 'node:fs';
|
|
22
|
+
import path from 'node:path';
|
|
23
|
+
import { z } from 'zod';
|
|
24
|
+
|
|
25
|
+
export interface ToolResult {
|
|
26
|
+
/** What the model reads. */
|
|
27
|
+
text: string;
|
|
28
|
+
/** The same facts, structured. */
|
|
29
|
+
data?: unknown;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
export interface ToolDef {
|
|
33
|
+
name: string;
|
|
34
|
+
description: string;
|
|
35
|
+
schema: z.ZodRawShape;
|
|
36
|
+
run: (args: Record<string, unknown>) => Promise<ToolResult>;
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
export interface ForemanClientOptions {
|
|
40
|
+
/** Where Foreman answers; `FOREMAN_URL` or http://127.0.0.1:4177. */
|
|
41
|
+
base: string;
|
|
42
|
+
fetchImpl?: typeof fetch;
|
|
43
|
+
/** Longest a `wait` may block, in seconds. */
|
|
44
|
+
maxWaitSeconds?: number;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
const usd = (n: number) => `$${n.toFixed(2)}`;
|
|
48
|
+
|
|
49
|
+
/** A folder that is Foreman itself: never a mission target (the oversight rule). */
|
|
50
|
+
export function isForemanCheckout(folder: string): boolean {
|
|
51
|
+
try {
|
|
52
|
+
const pkg = JSON.parse(fs.readFileSync(path.join(folder, 'package.json'), 'utf8')) as { bin?: Record<string, string> | string };
|
|
53
|
+
if (pkg.bin && typeof pkg.bin === 'object' && 'foreman' in pkg.bin) return true;
|
|
54
|
+
} catch { /* no package.json, or not ours */ }
|
|
55
|
+
return fs.existsSync(path.join(folder, 'src', 'orchestrator.ts')) && fs.existsSync(path.join(folder, 'src', 'policy.ts'));
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
interface RunSummary {
|
|
59
|
+
id: string; projectId?: string; mission: string; title?: string; status: string;
|
|
60
|
+
costUsd: number; budgetUsd: number; costBasis?: string; createdAt: number; endedAt?: number;
|
|
61
|
+
directorModel?: string; workerModel?: string; resumes?: number; stopReason?: string;
|
|
62
|
+
workers?: Array<{ id: string; status: string; costUsd: number; task: string }>;
|
|
63
|
+
git?: { branch: string; base: string; commits?: number; pr?: string; prState?: string };
|
|
64
|
+
usage?: { inputTokens: number; outputTokens: number };
|
|
65
|
+
}
|
|
66
|
+
interface Need { kind: string; id: string; runId?: string; text: string; options?: string[]; toolName?: string; since?: number }
|
|
67
|
+
interface ProjectCard {
|
|
68
|
+
id: string; name: string; folder: string; activeRun: RunSummary | null;
|
|
69
|
+
lastRun: { id: string; title?: string; mission: string; status: string; costUsd: number; createdAt: number } | null;
|
|
70
|
+
pendingPermissions: number; pendingQuestions: number; needs?: Need[]; git?: { branch?: string; dirty?: boolean } | null;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/** One line for a run, the way the fleet board says it. */
|
|
74
|
+
function runLine(r: RunSummary): string {
|
|
75
|
+
const cost = r.costBasis && r.costBasis !== 'priced' ? `${r.costBasis}` : `${usd(r.costUsd)} of ${usd(r.budgetUsd)}`;
|
|
76
|
+
return `${r.id} · ${r.status}${r.stopReason ? ` (stopped at its ${r.stopReason} cap)` : ''} · ${cost} · ${r.title || r.mission.slice(0, 80)}`;
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/** A transcript event as one line; null for noise. */
|
|
80
|
+
export function eventLine(event: string, d: Record<string, unknown>): string | null {
|
|
81
|
+
const s = (k: string) => (typeof d[k] === 'string' ? (d[k] as string) : '');
|
|
82
|
+
switch (event) {
|
|
83
|
+
case 'run_started': return 'run started';
|
|
84
|
+
case 'run_resumed': return 'run resumed';
|
|
85
|
+
case 'run_finished': return `run finished: ${s('status')}`;
|
|
86
|
+
case 'worker_started': return `${s('id')} started: ${s('task').slice(0, 120)}`;
|
|
87
|
+
case 'worker_finished': return `${s('id')} finished: ${s('status')}`;
|
|
88
|
+
case 'worker_progress': return `${s('id')}: ${d.blocked ? `BLOCKED — ${String(d.blocked)}` : s('status')}`;
|
|
89
|
+
case 'permission_request': return `NEEDS YOU — ${s('agent')} wants ${s('toolName')}${s('decisionReason') ? ` (${s('decisionReason')})` : ''}`;
|
|
90
|
+
case 'permission_resolved': return `approval ${s('behavior')}`;
|
|
91
|
+
case 'permission_timeout': return 'approval timed out (unattended default applied)';
|
|
92
|
+
case 'question': return `NEEDS YOU — director asks: ${s('question')}`;
|
|
93
|
+
case 'question_answered': return 'question answered';
|
|
94
|
+
case 'budget_alert': case 'budget_stop': case 'models_changed': case 'settings_changed': case 'mission_incomplete':
|
|
95
|
+
return s('text') || s('reason') || (Array.isArray(d.changes) ? (d.changes as string[]).join('; ') : null) || event;
|
|
96
|
+
case 'git_branch': return `on branch ${s('branch')}`;
|
|
97
|
+
case 'git_committed': return s('text') || 'branch closed';
|
|
98
|
+
case 'pull_request': return s('url') ? `pull request: ${s('url')}` : null;
|
|
99
|
+
case 'memory_updated': return 'project memory rewritten';
|
|
100
|
+
case 'steer': return `operator → ${s('to')}: ${s('text')}`;
|
|
101
|
+
case 'message': {
|
|
102
|
+
const msg = d.msg as { type?: string; message?: { content?: Array<{ type: string; text?: string; name?: string }> } } | undefined;
|
|
103
|
+
if (msg?.type !== 'assistant') return null;
|
|
104
|
+
const block = msg.message?.content?.find((c) => c.type === 'text' && c.text) ?? msg.message?.content?.find((c) => c.type === 'tool_use');
|
|
105
|
+
if (!block) return null;
|
|
106
|
+
return block.type === 'text' ? `${s('agent')}: ${(block.text ?? '').replace(/\s+/g, ' ').slice(0, 200)}` : `${s('agent')} → ${block.name}`;
|
|
107
|
+
}
|
|
108
|
+
default: return null;
|
|
109
|
+
}
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/** DONE WHEN lines from a mission doc: ticked and unticked. */
|
|
113
|
+
export function doneWhen(doc: string): { done: string[]; open: string[] } {
|
|
114
|
+
const done: string[] = []; const open: string[] = [];
|
|
115
|
+
for (const line of doc.split('\n')) {
|
|
116
|
+
const m = /^\s*[-*]\s*\[( |x|X)\]\s+(.*)$/.exec(line);
|
|
117
|
+
if (!m) continue;
|
|
118
|
+
(m[1].trim() ? done : open).push(m[2].trim());
|
|
119
|
+
}
|
|
120
|
+
return { done, open };
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* Waits for the next event on a run, over the server's own SSE stream. One
|
|
125
|
+
* subscription per call, filtered to the run; resolves true on the first
|
|
126
|
+
* event that names it, false at the timeout. Never polls.
|
|
127
|
+
*/
|
|
128
|
+
export async function waitForRunEvent(base: string, runId: string, seconds: number, fetchImpl: typeof fetch): Promise<boolean> {
|
|
129
|
+
const ctl = new AbortController();
|
|
130
|
+
let reader: ReadableStreamDefaultReader<Uint8Array> | null = null;
|
|
131
|
+
// The clock wins whatever the stream does: a fetch that ignores the abort
|
|
132
|
+
// signal, or a body that never yields, must not hold the caller past the
|
|
133
|
+
// seconds it asked for.
|
|
134
|
+
const timeout = new Promise<false>((resolve) => setTimeout(() => { ctl.abort(); void reader?.cancel().catch(() => {}); resolve(false); }, Math.max(1, seconds) * 1000));
|
|
135
|
+
const watch = (async (): Promise<boolean> => {
|
|
136
|
+
try {
|
|
137
|
+
const res = await fetchImpl(`${base}/events`, { signal: ctl.signal, headers: { accept: 'text/event-stream' } });
|
|
138
|
+
if (!res.ok || !res.body) return false;
|
|
139
|
+
reader = res.body.getReader();
|
|
140
|
+
const decoder = new TextDecoder();
|
|
141
|
+
let buf = '';
|
|
142
|
+
for (;;) {
|
|
143
|
+
const { value, done } = await reader.read();
|
|
144
|
+
if (done) return false;
|
|
145
|
+
buf += decoder.decode(value, { stream: true });
|
|
146
|
+
let nl: number;
|
|
147
|
+
while ((nl = buf.indexOf('\n')) >= 0) {
|
|
148
|
+
const line = buf.slice(0, nl).trimEnd();
|
|
149
|
+
buf = buf.slice(nl + 1);
|
|
150
|
+
if (!line.startsWith('data:')) continue;
|
|
151
|
+
try {
|
|
152
|
+
const env = JSON.parse(line.slice(5).trim()) as { runId?: string | null };
|
|
153
|
+
if (env.runId === runId) { ctl.abort(); void reader.cancel().catch(() => {}); return true; }
|
|
154
|
+
} catch { /* a comment or a partial frame */ }
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
} catch {
|
|
158
|
+
return false;
|
|
159
|
+
}
|
|
160
|
+
})();
|
|
161
|
+
return Promise.race([watch, timeout]);
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
/** The tools, as data — the MCP server registers them; the tests call them. */
|
|
165
|
+
export function foremanTools(opts: ForemanClientOptions): ToolDef[] {
|
|
166
|
+
const base = opts.base.replace(/\/+$/, '');
|
|
167
|
+
const f = opts.fetchImpl ?? fetch;
|
|
168
|
+
const maxWait = opts.maxWaitSeconds ?? 300;
|
|
169
|
+
|
|
170
|
+
async function get<T>(p: string): Promise<T> {
|
|
171
|
+
let res: Response;
|
|
172
|
+
try { res = await f(`${base}${p}`); } catch (err) {
|
|
173
|
+
throw new Error(`Foreman is not answering at ${base} (${err instanceof Error ? err.message : String(err)}). Start it with \`foreman\`, or set FOREMAN_URL.`);
|
|
174
|
+
}
|
|
175
|
+
if (!res.ok) throw new Error(`${p}: HTTP ${res.status} ${(await res.text()).slice(0, 200)}`);
|
|
176
|
+
return res.json() as Promise<T>;
|
|
177
|
+
}
|
|
178
|
+
async function post<T>(p: string, body: unknown): Promise<{ ok: boolean; status: number; json: T }> {
|
|
179
|
+
let res: Response;
|
|
180
|
+
try { res = await f(`${base}${p}`, { method: 'POST', headers: { 'content-type': 'application/json' }, body: JSON.stringify(body) }); } catch (err) {
|
|
181
|
+
throw new Error(`Foreman is not answering at ${base} (${err instanceof Error ? err.message : String(err)}). Start it with \`foreman\`, or set FOREMAN_URL.`);
|
|
182
|
+
}
|
|
183
|
+
const json = await res.json().catch(() => ({})) as T;
|
|
184
|
+
return { ok: res.ok, status: res.status, json };
|
|
185
|
+
}
|
|
186
|
+
async function findRun(runId: string): Promise<RunSummary> {
|
|
187
|
+
const { runs } = await get<{ runs: RunSummary[] }>('/runs');
|
|
188
|
+
const r = runs.find((x) => x.id === runId);
|
|
189
|
+
if (!r) throw new Error(`No run ${runId}. list_runs shows what exists.`);
|
|
190
|
+
return r;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
const fleet: ToolDef = {
|
|
194
|
+
name: 'fleet_status',
|
|
195
|
+
description: 'Every linked project with what it is doing: the active mission (status, spend against cap, crew), what needs a human, and the last finished run. Start here.',
|
|
196
|
+
schema: {},
|
|
197
|
+
run: async () => {
|
|
198
|
+
const d = await get<{ projects: ProjectCard[]; authMode?: string; version?: string }>('/projects');
|
|
199
|
+
const lines = d.projects.map((p) => {
|
|
200
|
+
const a = p.activeRun;
|
|
201
|
+
const needs = (p.pendingPermissions ?? 0) + (p.pendingQuestions ?? 0);
|
|
202
|
+
const head = `${p.name} (${p.id}) — ${p.folder}${p.git?.branch ? ` · ${p.git.branch}` : ''}`;
|
|
203
|
+
if (a) return `${head}\n running: ${runLine(a)}${needs ? `\n NEEDS YOU: ${needs} pending — the human decides these on the dashboard or phone` : ''}`;
|
|
204
|
+
if (p.lastRun) return `${head}\n idle · last: ${p.lastRun.status} · ${usd(p.lastRun.costUsd)} · ${p.lastRun.title || p.lastRun.mission.slice(0, 80)}`;
|
|
205
|
+
return `${head}\n idle · no runs yet`;
|
|
206
|
+
});
|
|
207
|
+
return {
|
|
208
|
+
text: lines.length ? lines.join('\n') : 'No projects linked. link_project adds one.',
|
|
209
|
+
data: { version: d.version, authMode: d.authMode, projects: d.projects.map((p) => ({ id: p.id, name: p.name, folder: p.folder, branch: p.git?.branch, activeRun: p.activeRun && { id: p.activeRun.id, status: p.activeRun.status, costUsd: p.activeRun.costUsd, budgetUsd: p.activeRun.budgetUsd }, needs: p.needs ?? [], lastRun: p.lastRun })) },
|
|
210
|
+
};
|
|
211
|
+
},
|
|
212
|
+
};
|
|
213
|
+
|
|
214
|
+
const listRuns: ToolDef = {
|
|
215
|
+
name: 'list_runs',
|
|
216
|
+
description: 'Run summaries, newest first, optionally for one project.',
|
|
217
|
+
schema: { projectId: z.string().optional(), limit: z.number().int().min(1).max(100).default(20) },
|
|
218
|
+
run: async ({ projectId, limit }) => {
|
|
219
|
+
const { runs } = await get<{ runs: RunSummary[] }>(`/runs${projectId ? `?projectId=${encodeURIComponent(String(projectId))}` : ''}`);
|
|
220
|
+
const shown = runs.slice(0, Number(limit ?? 20));
|
|
221
|
+
return { text: shown.length ? shown.map(runLine).join('\n') : 'No runs.', data: { runs: shown } };
|
|
222
|
+
},
|
|
223
|
+
};
|
|
224
|
+
|
|
225
|
+
/**
|
|
226
|
+
* What a reader of run_status would notice changing. `until: 'any'` waits
|
|
227
|
+
* for THIS to move, not for the next event: a fresh run emits a cost frame
|
|
228
|
+
* per SDK message, and returning on those meant "changed: true" twice in a
|
|
229
|
+
* row with identical payloads — a client had to poll after all.
|
|
230
|
+
*/
|
|
231
|
+
async function digestOf(id: string): Promise<string> {
|
|
232
|
+
const r = await findRun(id);
|
|
233
|
+
const needs = await needsOf(id);
|
|
234
|
+
const doc = await get<{ doc: string }>(`/missiondoc?run=${encodeURIComponent(id)}`).then((d) => d.doc ?? '').catch(() => '');
|
|
235
|
+
const dw = doneWhen(doc);
|
|
236
|
+
return JSON.stringify([r.status, r.stopReason ?? null, (r.costUsd ?? 0).toFixed(2), (r.workers ?? []).map((w) => `${w.id}:${w.status}`), dw.done.length, dw.open.length, needs.map((n) => n.id), r.git?.commits ?? 0, r.git?.pr ?? null]);
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
/** What is pending on a run right now, from the fleet payload. */
|
|
240
|
+
async function needsOf(id: string): Promise<Need[]> {
|
|
241
|
+
const projects = (await get<{ projects: ProjectCard[] }>('/projects')).projects;
|
|
242
|
+
const card = projects.find((p) => p.activeRun?.id === id);
|
|
243
|
+
return (card?.needs ?? []).filter((n) => !n.runId || n.runId === id);
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
const runStatus: ToolDef = {
|
|
247
|
+
name: 'run_status',
|
|
248
|
+
description: 'One run: status, spend against cap, crew and their states, DONE WHEN ticks, what needs a human, branch and pull request. With wait_seconds > 0 it blocks until `until` is met — "any" change on the run, the run "finished", or it "needs_you" (a pending approval or question, or finished) — or until the timeout. Use it instead of polling; call again if it timed out.',
|
|
249
|
+
schema: {
|
|
250
|
+
runId: z.string(),
|
|
251
|
+
wait_seconds: z.number().int().min(0).max(maxWait).default(0),
|
|
252
|
+
until: z.enum(['any', 'finished', 'needs_you']).default('any'),
|
|
253
|
+
},
|
|
254
|
+
run: async ({ runId, wait_seconds, until }) => {
|
|
255
|
+
const id = String(runId);
|
|
256
|
+
const cond = String(until ?? 'any');
|
|
257
|
+
let changed: boolean | undefined;
|
|
258
|
+
if (Number(wait_seconds) > 0) {
|
|
259
|
+
// Wait in rounds: each round ends on the first event for the run,
|
|
260
|
+
// then the condition is checked against fresh state; an event that
|
|
261
|
+
// does not satisfy it (a cost tick, under "finished") starts another
|
|
262
|
+
// round with the time that is left. The clock is the outer bound.
|
|
263
|
+
const deadline = Date.now() + Number(wait_seconds) * 1000;
|
|
264
|
+
changed = false;
|
|
265
|
+
const before = await digestOf(id);
|
|
266
|
+
for (;;) {
|
|
267
|
+
const now = await findRun(id);
|
|
268
|
+
const satisfied = now.status !== 'running'
|
|
269
|
+
|| (cond === 'needs_you' && (await needsOf(id)).length > 0)
|
|
270
|
+
|| (cond === 'any' && changed);
|
|
271
|
+
if (satisfied) break;
|
|
272
|
+
const left = Math.ceil((deadline - Date.now()) / 1000);
|
|
273
|
+
if (left <= 0) break;
|
|
274
|
+
const got = await waitForRunEvent(base, id, left, f);
|
|
275
|
+
if (!got) break;
|
|
276
|
+
// An event arrived; only count it when the picture it paints differs.
|
|
277
|
+
if ((await digestOf(id)) !== before) changed = true;
|
|
278
|
+
}
|
|
279
|
+
}
|
|
280
|
+
const r = await findRun(id);
|
|
281
|
+
const needs = await needsOf(id);
|
|
282
|
+
const doc = await get<{ doc: string }>(`/missiondoc?run=${encodeURIComponent(id)}`).then((d) => d.doc).catch(() => '');
|
|
283
|
+
const dw = doneWhen(doc);
|
|
284
|
+
const workers = (r.workers ?? []).map((w) => ` ${w.id} · ${w.status} · ${usd(w.costUsd)} · ${w.task.slice(0, 80)}`);
|
|
285
|
+
const lines = [
|
|
286
|
+
runLine(r),
|
|
287
|
+
`director ${r.directorModel ?? 'default'} · workers ${r.workerModel ?? 'default'} · resumes ${r.resumes ?? 0}`,
|
|
288
|
+
r.git ? `branch ${r.git.branch} from ${r.git.base}${r.git.pr ? ` · PR ${r.git.pr}${r.git.prState ? ` (${r.git.prState})` : ''}` : ''}` : null,
|
|
289
|
+
dw.done.length + dw.open.length ? `DONE WHEN ${dw.done.length}/${dw.done.length + dw.open.length}${dw.open.length ? `\n open: ${dw.open.join('\n open: ')}` : ''}` : null,
|
|
290
|
+
workers.length ? `crew:\n${workers.join('\n')}` : null,
|
|
291
|
+
needs.length ? `NEEDS YOU (${needs.length}) — only a human can answer these, on the dashboard or the phone:\n${needs.map((n) => ` [${n.kind}] ${n.text}`).join('\n')}` : null,
|
|
292
|
+
changed !== undefined ? (r.status !== 'running' ? 'finished' : needs.length && cond === 'needs_you' ? 'needs you' : changed ? 'changed: yes' : 'changed: no (timeout — call again)') : null,
|
|
293
|
+
].filter(Boolean);
|
|
294
|
+
return { text: lines.join('\n'), data: { run: r, doneWhen: dw, needs, changed, waitedFor: Number(wait_seconds) > 0 ? cond : undefined } };
|
|
295
|
+
},
|
|
296
|
+
};
|
|
297
|
+
|
|
298
|
+
const runReport: ToolDef = {
|
|
299
|
+
name: 'run_report',
|
|
300
|
+
description: 'What a finished run produced, in one call: the director\'s final report, DONE WHEN ticks, the files it changed with +/− counts, the branch, commit and pull request, spend and crew. For a running run it reports the state so far.',
|
|
301
|
+
schema: { runId: z.string() },
|
|
302
|
+
run: async ({ runId }) => {
|
|
303
|
+
const id = String(runId);
|
|
304
|
+
const r = await findRun(id);
|
|
305
|
+
const [events, docRes, deck] = await Promise.all([
|
|
306
|
+
get<{ events: Array<{ ts: number; event: string; data: Record<string, unknown> }> }>(`/runs/${encodeURIComponent(id)}/events`).then((d) => d.events).catch(() => []),
|
|
307
|
+
get<{ doc: string }>(`/missiondoc?run=${encodeURIComponent(id)}`).catch(() => ({ doc: '' })),
|
|
308
|
+
get<{ files: Array<{ path: string; status: string; additions: number; deletions: number; preexisting?: boolean }>; artifacts: Array<{ path: string; kind: string }>; totals: { files: number; additions: number; deletions: number }; baseline: { kind: string } }>(`/runs/${encodeURIComponent(id)}/deck`).catch(() => null),
|
|
309
|
+
]);
|
|
310
|
+
// The director's last words: the final assistant text before the run ended.
|
|
311
|
+
let report = '';
|
|
312
|
+
for (const e of events) {
|
|
313
|
+
if (e.event !== 'message' || e.data?.agent !== 'director') continue;
|
|
314
|
+
const msg = e.data.msg as { type?: string; message?: { content?: Array<{ type: string; text?: string }> } } | undefined;
|
|
315
|
+
const text = msg?.type === 'assistant' ? msg.message?.content?.filter((c) => c.type === 'text' && c.text).map((c) => c.text).join('\n') : '';
|
|
316
|
+
if (text && text.trim().length > 40) report = text.trim();
|
|
317
|
+
}
|
|
318
|
+
const dw = doneWhen(docRes.doc);
|
|
319
|
+
const files = deck?.files ?? [];
|
|
320
|
+
const own = files.filter((x) => !x.preexisting);
|
|
321
|
+
const fileLines = own.slice(0, 40).map((x) => ` ${x.status.padEnd(8)} ${x.path} +${x.additions} −${x.deletions}`);
|
|
322
|
+
const images = (deck?.artifacts ?? []).filter((a) => a.kind === 'image').length;
|
|
323
|
+
const lines = [
|
|
324
|
+
runLine(r),
|
|
325
|
+
r.git ? `branch ${r.git.branch} from ${r.git.base}${r.git.commits ? ` · ${r.git.commits} commit${r.git.commits === 1 ? '' : 's'}` : ''}${r.git.pr ? ` · PR ${r.git.pr}${r.git.prState ? ` (${r.git.prState})` : ''}` : ' · no pull request yet (the human opens it from the run page)'}` : 'not a git repository',
|
|
326
|
+
`DONE WHEN ${dw.done.length}/${dw.done.length + dw.open.length}${dw.open.length ? ` · open: ${dw.open.join(' · ')}` : ''}`,
|
|
327
|
+
deck ? `changed: ${own.length} file${own.length === 1 ? '' : 's'} · +${deck.totals.additions} −${deck.totals.deletions}${images ? ` · ${images} screenshot${images === 1 ? '' : 's'}` : ''}${files.length > own.length ? ` · ${files.length - own.length} already dirty before the run` : ''}` : null,
|
|
328
|
+
fileLines.length ? fileLines.join('\n') + (own.length > 40 ? `\n … ${own.length - 40} more` : '') : null,
|
|
329
|
+
`crew: ${(r.workers ?? []).length} worker${(r.workers ?? []).length === 1 ? '' : 's'} · ${(r.workers ?? []).filter((w) => w.status === 'done').length} done`,
|
|
330
|
+
report ? `\nDirector's report:\n${report.slice(0, 4000)}` : '\nNo final report from the director yet.',
|
|
331
|
+
].filter(Boolean);
|
|
332
|
+
return { text: lines.join('\n'), data: { run: r, doneWhen: dw, files: own, totals: deck?.totals, report } };
|
|
333
|
+
},
|
|
334
|
+
};
|
|
335
|
+
|
|
336
|
+
const transcript: ToolDef = {
|
|
337
|
+
name: 'run_transcript',
|
|
338
|
+
description: 'The run\'s recent events, one line each, oldest first. since_ts (ms) narrows to what happened after a moment you already read.',
|
|
339
|
+
schema: { runId: z.string(), since_ts: z.number().optional(), limit: z.number().int().min(1).max(500).default(50) },
|
|
340
|
+
run: async ({ runId, since_ts, limit }) => {
|
|
341
|
+
const { events } = await get<{ events: Array<{ ts: number; event: string; data: Record<string, unknown> }> }>(`/runs/${encodeURIComponent(String(runId))}/events`);
|
|
342
|
+
const lines: Array<{ ts: number; line: string }> = [];
|
|
343
|
+
for (const e of events) {
|
|
344
|
+
if (since_ts && e.ts <= Number(since_ts)) continue;
|
|
345
|
+
const line = eventLine(e.event, e.data ?? {});
|
|
346
|
+
if (line) lines.push({ ts: e.ts, line });
|
|
347
|
+
}
|
|
348
|
+
const shown = lines.slice(-Number(limit ?? 50));
|
|
349
|
+
return {
|
|
350
|
+
text: shown.length ? shown.map((l) => `${new Date(l.ts).toISOString().slice(11, 19)} ${l.line}`).join('\n') : 'Nothing yet.',
|
|
351
|
+
data: { events: shown, lastTs: shown.at(-1)?.ts ?? null },
|
|
352
|
+
};
|
|
353
|
+
},
|
|
354
|
+
};
|
|
355
|
+
|
|
356
|
+
const missionDoc: ToolDef = {
|
|
357
|
+
name: 'mission_doc',
|
|
358
|
+
description: 'The run\'s MISSION.md: DONE WHEN criteria, log, state — what the director itself keeps.',
|
|
359
|
+
schema: { runId: z.string() },
|
|
360
|
+
run: async ({ runId }) => {
|
|
361
|
+
const { doc } = await get<{ doc: string }>(`/missiondoc?run=${encodeURIComponent(String(runId))}`);
|
|
362
|
+
return { text: doc || 'MISSION.md not written yet.', data: { doc } };
|
|
363
|
+
},
|
|
364
|
+
};
|
|
365
|
+
|
|
366
|
+
const memory: ToolDef = {
|
|
367
|
+
name: 'project_memory',
|
|
368
|
+
description: 'The project\'s memory (.foreman/MEMORY.md): what earlier crews learned — how to run and test it, ports, traps.',
|
|
369
|
+
schema: { projectId: z.string() },
|
|
370
|
+
run: async ({ projectId }) => {
|
|
371
|
+
const d = await get<{ text: string; updatedAt?: number }>(`/projects/${encodeURIComponent(String(projectId))}/memory`);
|
|
372
|
+
return { text: d.text || 'No memory yet.', data: d };
|
|
373
|
+
},
|
|
374
|
+
};
|
|
375
|
+
|
|
376
|
+
const search: ToolDef = {
|
|
377
|
+
name: 'search_runs',
|
|
378
|
+
description: 'Runs across the fleet whose title, brief, project or folder match.',
|
|
379
|
+
schema: { q: z.string().min(1) },
|
|
380
|
+
run: async ({ q }) => {
|
|
381
|
+
const d = await get<{ runs: RunSummary[] }>(`/search?q=${encodeURIComponent(String(q))}`);
|
|
382
|
+
return { text: d.runs.length ? d.runs.map(runLine).join('\n') : 'No match.', data: d };
|
|
383
|
+
},
|
|
384
|
+
};
|
|
385
|
+
|
|
386
|
+
const doctor: ToolDef = {
|
|
387
|
+
name: 'doctor',
|
|
388
|
+
description: 'What this machine has for missions: credentials, providers, browser, reach — the same checks as `foreman doctor`.',
|
|
389
|
+
schema: {},
|
|
390
|
+
run: async () => {
|
|
391
|
+
const d = await get<{ checks: Array<{ name: string; status: string; detail: string; fix?: string }> }>('/doctor');
|
|
392
|
+
return { text: d.checks.map((c) => `${c.status === 'ok' ? '✓' : c.status === 'warn' ? '!' : '✗'} ${c.name}: ${c.detail}${c.fix && c.status !== 'ok' ? `\n ${c.fix}` : ''}`).join('\n'), data: d };
|
|
393
|
+
},
|
|
394
|
+
};
|
|
395
|
+
|
|
396
|
+
const link: ToolDef = {
|
|
397
|
+
name: 'link_project',
|
|
398
|
+
description: 'Link a folder on this machine as a project (idempotent), or clone a Git URL under the projects root and link it. Returns the project id.',
|
|
399
|
+
schema: { folder: z.string().optional(), git_url: z.string().optional(), branch: z.string().optional() },
|
|
400
|
+
run: async ({ folder, git_url, branch }) => {
|
|
401
|
+
if (git_url) {
|
|
402
|
+
const started = await post<{ id?: string; dest?: string; error?: string }>('/projects/clone', { url: git_url, branch });
|
|
403
|
+
if (!started.ok) return { text: `Could not start the clone: ${started.json.error ?? started.status}` };
|
|
404
|
+
for (let i = 0; i < 600; i++) {
|
|
405
|
+
const job = await get<{ state: string; progress: string; projectId?: string; error?: string }>(`/projects/clone/${started.json.id}`);
|
|
406
|
+
if (job.state === 'done') return { text: `Cloned to ${started.json.dest} and linked as project ${job.projectId}.`, data: { projectId: job.projectId, folder: started.json.dest } };
|
|
407
|
+
if (job.state === 'error') return { text: `Clone failed: ${job.error}` };
|
|
408
|
+
await new Promise((r) => setTimeout(r, 1000));
|
|
409
|
+
}
|
|
410
|
+
return { text: 'The clone is still running; check fleet_status in a minute.' };
|
|
411
|
+
}
|
|
412
|
+
if (!folder) return { text: 'Give a folder path or a git_url.' };
|
|
413
|
+
const abs = path.resolve(String(folder));
|
|
414
|
+
const r = await post<{ project?: { id: string; name: string; folder: string }; error?: string }>('/projects', { folder: abs });
|
|
415
|
+
if (!r.ok || !r.json.project) return { text: `Could not link ${abs}: ${r.json.error ?? r.status}` };
|
|
416
|
+
return { text: `Linked ${r.json.project.name} (${r.json.project.id}) at ${r.json.project.folder}.`, data: r.json.project };
|
|
417
|
+
},
|
|
418
|
+
};
|
|
419
|
+
|
|
420
|
+
const start: ToolDef = {
|
|
421
|
+
name: 'start_mission',
|
|
422
|
+
description: 'Start a mission on a project: the brief, a dollar cap, optional director/worker models and a browser. The mission runs in Foreman under its own governance; approvals and questions go to the human on the dashboard or phone, never through this tool. Follow with run_status(wait_seconds).',
|
|
423
|
+
schema: {
|
|
424
|
+
projectId: z.string(), brief: z.string().min(10), budgetUsd: z.number().positive(),
|
|
425
|
+
director: z.string().optional(), worker: z.string().optional(), browser: z.boolean().optional(),
|
|
426
|
+
},
|
|
427
|
+
run: async ({ projectId, brief, budgetUsd, director, worker, browser }) => {
|
|
428
|
+
const pid = String(projectId);
|
|
429
|
+
const projects = (await get<{ projects: ProjectCard[] }>('/projects')).projects;
|
|
430
|
+
const p = projects.find((x) => x.id === pid);
|
|
431
|
+
if (!p) return { text: `No project ${pid}. fleet_status lists them; link_project adds one.` };
|
|
432
|
+
if (isForemanCheckout(p.folder)) return { text: 'Refused: this folder is Foreman itself, and Foreman never runs missions on its own oversight infrastructure.' };
|
|
433
|
+
const r = await post<{ ok?: boolean; error?: string }>('/run', {
|
|
434
|
+
projectId: pid, mission: brief, budgetUsd, directorModel: director, workerModel: worker, browserTools: browser === true ? true : undefined,
|
|
435
|
+
});
|
|
436
|
+
if (r.status === 409) return { text: `${p.name} already has an active mission; see fleet_status. One mission per project at a time.` };
|
|
437
|
+
if (!r.ok) return { text: `Could not start: ${r.json.error ?? r.status}` };
|
|
438
|
+
// The run id lands a moment later; read it back so the caller can watch it.
|
|
439
|
+
for (let i = 0; i < 20; i++) {
|
|
440
|
+
const { runs } = await get<{ runs: RunSummary[] }>(`/runs?projectId=${encodeURIComponent(pid)}`);
|
|
441
|
+
const live = runs.find((x) => x.status === 'running');
|
|
442
|
+
if (live) return { text: `Started ${live.id} on ${p.name}, cap ${usd(live.budgetUsd)}. Watch it with run_status(runId, wait_seconds).`, data: { runId: live.id, projectId: pid } };
|
|
443
|
+
await new Promise((res) => setTimeout(res, 250));
|
|
444
|
+
}
|
|
445
|
+
return { text: `Started on ${p.name}; the run id was not visible yet — list_runs will show it.`, data: { projectId: pid } };
|
|
446
|
+
},
|
|
447
|
+
};
|
|
448
|
+
|
|
449
|
+
const steer: ToolDef = {
|
|
450
|
+
name: 'steer',
|
|
451
|
+
description: 'Send an operator note to a running director (a hint, a priority, a correction). Relays the human; it does not approve anything.',
|
|
452
|
+
schema: { runId: z.string(), text: z.string().min(1) },
|
|
453
|
+
run: async ({ runId, text }) => {
|
|
454
|
+
const r = await post<{ ok?: boolean; error?: string }>('/steer', { runId, text });
|
|
455
|
+
return { text: r.ok ? 'Delivered to the director for its next turn.' : `Could not steer: ${r.json.error ?? r.status}` };
|
|
456
|
+
},
|
|
457
|
+
};
|
|
458
|
+
|
|
459
|
+
return [fleet, listRuns, runStatus, runReport, transcript, missionDoc, memory, search, doctor, link, start, steer];
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
/** Runs the MCP server over stdio until the client goes away. Nothing may be written to stdout but the protocol. */
|
|
463
|
+
// 127.0.0.1 rather than localhost: a stray process on the IPv6 wildcard
|
|
464
|
+
// (a worker's dev server, once) answers `localhost` first in most resolvers
|
|
465
|
+
// and would shadow Foreman for this client too. FOREMAN_URL overrides.
|
|
466
|
+
export async function serveMcp(base = process.env.FOREMAN_URL || 'http://127.0.0.1:4177'): Promise<void> {
|
|
467
|
+
const { McpServer } = await import('@modelcontextprotocol/sdk/server/mcp.js');
|
|
468
|
+
const { StdioServerTransport } = await import('@modelcontextprotocol/sdk/server/stdio.js');
|
|
469
|
+
const server = new McpServer({ name: 'foreman', version: '1' });
|
|
470
|
+
for (const t of foremanTools({ base })) {
|
|
471
|
+
server.registerTool(t.name, { description: t.description, inputSchema: t.schema }, async (args: Record<string, unknown>) => {
|
|
472
|
+
try {
|
|
473
|
+
const r = await t.run(args ?? {});
|
|
474
|
+
return { content: [{ type: 'text' as const, text: r.text }], ...(r.data !== undefined ? { structuredContent: r.data as Record<string, unknown> } : {}) };
|
|
475
|
+
} catch (err) {
|
|
476
|
+
return { content: [{ type: 'text' as const, text: err instanceof Error ? err.message : String(err) }], isError: true };
|
|
477
|
+
}
|
|
478
|
+
});
|
|
479
|
+
}
|
|
480
|
+
await server.connect(new StdioServerTransport());
|
|
481
|
+
}
|
package/src/orchestrator.ts
CHANGED
|
@@ -604,6 +604,8 @@ type AgentRole = 'director' | 'worker';
|
|
|
604
604
|
* and which role's rates price the tokens.
|
|
605
605
|
*/
|
|
606
606
|
interface WorkerOverrides {
|
|
607
|
+
/** Set on the one continuation a worker gets after the turn cap. */
|
|
608
|
+
continued?: boolean;
|
|
607
609
|
env?: AgentEnv;
|
|
608
610
|
model?: string;
|
|
609
611
|
priceRole?: AgentRole;
|
|
@@ -746,6 +748,19 @@ const USAGE_LIMIT_RE = /out of usage credits|usage limit reached|upgrade to incr
|
|
|
746
748
|
/** One worker's outcome, as runWorker hands it back. */
|
|
747
749
|
interface WorkerOutcome { report: string; isError: boolean }
|
|
748
750
|
|
|
751
|
+
/**
|
|
752
|
+
* Turns a worker gets before the SDK stops it. Raised from 60: five workers
|
|
753
|
+
* in four missions hit the cap mid-task on legitimate, brief-sized work
|
|
754
|
+
* (a hundred TypeScript errors in one package; a test-suite port), and each
|
|
755
|
+
* time the director paid to respawn one that re-read the same files.
|
|
756
|
+
*/
|
|
757
|
+
const WORKER_MAX_TURNS = 100;
|
|
758
|
+
/** What a worker stopped by the cap is told when its session is picked back up. */
|
|
759
|
+
const WORKER_CONTINUE_PROMPT =
|
|
760
|
+
'You were stopped by the turn cap, not by a failure. Your session and your files are as you left them; ' +
|
|
761
|
+
'`git status` and `git diff` show your uncommitted work. Continue the same task from where you were — do not ' +
|
|
762
|
+
'start over or re-read what you already know — finish it, report with report_progress, and end your turn.';
|
|
763
|
+
|
|
749
764
|
/**
|
|
750
765
|
* A worker record plus the in-process handles that must never be persisted:
|
|
751
766
|
* the live query (for interrupts), the run promise (so nothing is left
|
|
@@ -1983,7 +1998,7 @@ export class MissionRun {
|
|
|
1983
1998
|
permissionMode: 'default',
|
|
1984
1999
|
resume: resumeSessionId,
|
|
1985
2000
|
model: overrides?.model || this.meta.workerModel,
|
|
1986
|
-
maxTurns:
|
|
2001
|
+
maxTurns: WORKER_MAX_TURNS,
|
|
1987
2002
|
systemPrompt: { type: 'preset', preset: 'claude_code', append: WORKER_CHARTER },
|
|
1988
2003
|
...(overrides?.env ?? this.agentEnv.worker),
|
|
1989
2004
|
// Built per worker: the report_progress handler closes over this id,
|
|
@@ -1996,6 +2011,7 @@ export class MissionRun {
|
|
|
1996
2011
|
w.q = q;
|
|
1997
2012
|
|
|
1998
2013
|
let report = '';
|
|
2014
|
+
let hitTurnCap = false;
|
|
1999
2015
|
let isError = false;
|
|
2000
2016
|
|
|
2001
2017
|
// The stall watchdog. A silent worker no longer blocks the director, but
|
|
@@ -2041,6 +2057,7 @@ export class MissionRun {
|
|
|
2041
2057
|
if (m.type === 'result') {
|
|
2042
2058
|
report = String(m.result ?? '');
|
|
2043
2059
|
isError = Boolean(m.is_error);
|
|
2060
|
+
hitTurnCap = m.subtype === 'error_max_turns' || /maximum number of turns/i.test(report);
|
|
2044
2061
|
this.noteUsageLimit(report);
|
|
2045
2062
|
// Same ordering rule as the director loop: the cost event carries
|
|
2046
2063
|
// usage, so usage has to be current before it is emitted.
|
|
@@ -2060,6 +2077,16 @@ export class MissionRun {
|
|
|
2060
2077
|
w.q = undefined;
|
|
2061
2078
|
}
|
|
2062
2079
|
|
|
2080
|
+
// A worker stopped by the turn cap mid-task is not a failed worker. Its
|
|
2081
|
+
// session is resumed once with a continue turn — the same context, the
|
|
2082
|
+
// same files — instead of handing the director an error it can only
|
|
2083
|
+
// answer by spawning a replacement that re-reads everything. Once: a
|
|
2084
|
+
// worker that burns two allowances is stuck, and that IS the director's.
|
|
2085
|
+
if (hitTurnCap && !overrides?.continued && w.sessionId && !stalled && !looping) {
|
|
2086
|
+
this.emit('worker_progress', { id: workerId, status: `reached the ${WORKER_MAX_TURNS}-turn cap mid-task; continuing the same session once` });
|
|
2087
|
+
return this.runWorker(workerId, WORKER_CONTINUE_PROMPT, w.sessionId, { ...overrides, continued: true });
|
|
2088
|
+
}
|
|
2089
|
+
|
|
2063
2090
|
// Told to the director as a fact plus its options, not as an order: it is
|
|
2064
2091
|
// the agent with the context to know whether this needs a different
|
|
2065
2092
|
// approach, a different worker, or a human.
|