@amenophis1er/foreman 0.1.13 → 0.1.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -4
- package/package.json +1 -1
- package/src/deck.ts +4 -2
- package/src/gitwork.test.ts +16 -1
- package/src/gitwork.ts +27 -1
- package/src/mcp.test.ts +56 -6
- package/src/mcp.ts +91 -12
- package/src/notify.ts +5 -1
- package/src/orchestrator.test.ts +25 -1
- package/src/orchestrator.ts +92 -1
- package/src/server.ts +60 -5
- package/src/snapshot.test.ts +68 -0
- package/src/snapshot.ts +116 -0
- package/src/store.test.ts +16 -0
- package/src/store.ts +17 -0
- package/src/types.ts +14 -0
- package/ui/dist/assets/{index--dvfX3Mo.js → index-CnCv2QiH.js} +24 -24
- package/ui/dist/index.html +1 -1
package/README.md
CHANGED
|
@@ -230,7 +230,7 @@ result.
|
|
|
230
230
|
```sh
|
|
231
231
|
npm ci && npm run setup # dependencies, then the dashboard build
|
|
232
232
|
npm start # serves http://localhost:4177
|
|
233
|
-
npm test #
|
|
233
|
+
npm test # 379 tests, node:test
|
|
234
234
|
npm run typecheck # server and dashboard
|
|
235
235
|
npm run dev # API + Vite together
|
|
236
236
|
scripts/dev-restart.sh # restarts the server only when nothing would be lost
|
|
@@ -247,10 +247,12 @@ codex mcp add foreman -- foreman mcp
|
|
|
247
247
|
agy mcp add foreman -- foreman mcp
|
|
248
248
|
```
|
|
249
249
|
|
|
250
|
-
Tools: `fleet_status`, `list_runs`, `run_status` (with `wait_seconds
|
|
251
|
-
that
|
|
250
|
+
Tools: `fleet_status`, `list_runs`, `run_status` (with `wait_seconds` and
|
|
251
|
+
`until`: one call that blocks until the run changes, finishes, or needs you),
|
|
252
|
+
`run_report` (a finished run in one call: the director's report, DONE WHEN,
|
|
253
|
+
changed files, branch and pull request), `run_transcript`, `mission_doc`,
|
|
252
254
|
`project_memory`, `search_runs`, `doctor`, `link_project` (folder or Git URL),
|
|
253
|
-
`start_mission`, `steer`. It talks to the running server at `FOREMAN_URL`
|
|
255
|
+
`start_mission`, `steer`. Start, wait until finished, read the report: three calls. It talks to the running server at `FOREMAN_URL`
|
|
254
256
|
(default `http://localhost:4177`) and has no logic of its own.
|
|
255
257
|
|
|
256
258
|
Deliberately absent: approving or denying, answering the director's questions,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@amenophis1er/foreman",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.15",
|
|
4
4
|
"description": "Autonomous mission runner on the Claude Agent SDK: a director plans, delegates to workers, verifies, and reports — from one dashboard, your phone, or the CLI.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"claude",
|
package/src/deck.ts
CHANGED
|
@@ -847,7 +847,7 @@ function sendJson(res: ServerResponse, code: number, body: unknown): void {
|
|
|
847
847
|
*/
|
|
848
848
|
export async function handleDeckRoute(
|
|
849
849
|
req: IncomingMessage, res: ServerResponse, url: URL,
|
|
850
|
-
lookup: (scope: 'runs' | 'projects', id: string) => Promise<{ folder: string } | null>,
|
|
850
|
+
lookup: (scope: 'runs' | 'projects', id: string) => Promise<{ folder: string; deck?: Deck } | null>,
|
|
851
851
|
): Promise<boolean> {
|
|
852
852
|
// The same jail and the same viewer serve two scopes: a run (its deck,
|
|
853
853
|
// relative to a baseline) and a project (its tree as it stands, no baseline).
|
|
@@ -863,7 +863,9 @@ export async function handleDeckRoute(
|
|
|
863
863
|
if (!run) { sendJson(res, 404, { error: 'not found' }); return true; }
|
|
864
864
|
|
|
865
865
|
if (what === 'deck') {
|
|
866
|
-
|
|
866
|
+
// A finished run's deck is the one frozen when it ended (snapshot.ts);
|
|
867
|
+
// the caller hands it over. A running run is diffed live.
|
|
868
|
+
sendJson(res, 200, run.deck ?? await deckFor(run.folder, runId));
|
|
867
869
|
return true;
|
|
868
870
|
}
|
|
869
871
|
if (what === 'tree') {
|
package/src/gitwork.test.ts
CHANGED
|
@@ -4,7 +4,7 @@ import os from 'node:os';
|
|
|
4
4
|
import path from 'node:path';
|
|
5
5
|
import { execFileSync } from 'node:child_process';
|
|
6
6
|
import { mkdtemp, writeFile } from 'node:fs/promises';
|
|
7
|
-
import { closeMissionBranch, ensureMissionBranch, gitInfo, missionBranchName, startMissionBranch } from './gitwork.js';
|
|
7
|
+
import { closeMissionBranch, ensureMissionBranch, gitInfo, missionBranchName, startMissionBranch, renameMissionBranch } from './gitwork.js';
|
|
8
8
|
|
|
9
9
|
const sh = (cwd: string, ...args: string[]) => execFileSync('git', args, { cwd, stdio: 'pipe', env: { ...process.env, GIT_CONFIG_GLOBAL: '/dev/null' } }).toString();
|
|
10
10
|
|
|
@@ -93,3 +93,18 @@ test('prDraft: the run title, the brief, the boxes as the mission left them, and
|
|
|
93
93
|
assert.match(d.body, /## Done when\n\n- \[x\] footer\.html exists\n- \[ \] linked from index/);
|
|
94
94
|
assert.match(d.body, /branch `foreman\/add-a-footer-ab12` from `main` · spend \$0\.42/);
|
|
95
95
|
});
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
test('renameMissionBranch takes the run title once there is one, and leaves a branch that moved on', async () => {
|
|
99
|
+
const dir = await repo();
|
|
100
|
+
const g = await startMissionBranch(dir, 'Repo: this folder is a git worktree. Do the thing.', '1788713434983-226123af');
|
|
101
|
+
assert.ok(!('error' in g));
|
|
102
|
+
assert.equal(g.branch, 'foreman/repo-this-folder-is-a-git-23af');
|
|
103
|
+
const renamed = await renameMissionBranch(dir, g.branch, 'Studio data sanitization and null handling', '1788713434983-226123af');
|
|
104
|
+
assert.equal(renamed, 'foreman/studio-data-sanitization-and-null-23af');
|
|
105
|
+
assert.equal(sh(dir, 'rev-parse', '--abbrev-ref', 'HEAD').trim(), renamed);
|
|
106
|
+
// Same name again: nothing to do. Not on the branch any more: left alone.
|
|
107
|
+
assert.equal(await renameMissionBranch(dir, renamed!, 'Studio data sanitization and null handling', '1788713434983-226123af'), null);
|
|
108
|
+
sh(dir, 'checkout', '-q', 'main');
|
|
109
|
+
assert.equal(await renameMissionBranch(dir, renamed!, 'Another title', '1788713434983-226123af'), null);
|
|
110
|
+
});
|
package/src/gitwork.ts
CHANGED
|
@@ -70,7 +70,12 @@ export async function gitInfo(folder: string): Promise<GitInfo> {
|
|
|
70
70
|
/** `foreman/<first words of the brief>-<id tail>`: readable in `git branch`, unique per run. */
|
|
71
71
|
export function missionBranchName(mission: string, runId: string): string {
|
|
72
72
|
const first = mission.split('\n').find((l) => l.trim())?.trim() ?? 'mission';
|
|
73
|
-
|
|
73
|
+
// Whole words up to six and forty characters: a name cut mid-word
|
|
74
|
+
// ("null-handli") reads worse than a shorter one.
|
|
75
|
+
const words = first.toLowerCase().replace(/[^a-z0-9]+/g, '-').replace(/^-+|-+$/g, '').split('-').filter(Boolean).slice(0, 6);
|
|
76
|
+
let slug = '';
|
|
77
|
+
for (const w of words) { const next = slug ? `${slug}-${w}` : w; if (next.length > 40) break; slug = next; }
|
|
78
|
+
slug = slug || words[0]?.slice(0, 40) || 'mission';
|
|
74
79
|
const tail = runId.replace(/[^a-z0-9]/gi, '').slice(-4).toLowerCase();
|
|
75
80
|
return `foreman/${slug}-${tail}`;
|
|
76
81
|
}
|
|
@@ -92,6 +97,27 @@ export async function startMissionBranch(folder: string, mission: string, runId:
|
|
|
92
97
|
return { branch, base: info.branch ?? 'HEAD', baseHead: info.head ?? null };
|
|
93
98
|
}
|
|
94
99
|
|
|
100
|
+
/**
|
|
101
|
+
* The branch takes the run's title once there is one. Branches are created
|
|
102
|
+
* before the title exists (the title is a model call that lands seconds
|
|
103
|
+
* later), so they started from the brief's first words — and briefs that all
|
|
104
|
+
* open with the same boilerplate gave every run the same name. Renamed in
|
|
105
|
+
* place, only while nothing has been committed on it and it is still checked
|
|
106
|
+
* out. Returns the new name, or null when it was left as it was.
|
|
107
|
+
*/
|
|
108
|
+
export async function renameMissionBranch(folder: string, from: string, title: string, runId: string): Promise<string | null> {
|
|
109
|
+
const to = missionBranchName(title, runId);
|
|
110
|
+
if (to === from) return null;
|
|
111
|
+
const info = await gitInfo(folder);
|
|
112
|
+
if (!info.repo || info.branch !== from) return null;
|
|
113
|
+
try {
|
|
114
|
+
await git(['branch', '-m', from, to], folder);
|
|
115
|
+
return to;
|
|
116
|
+
} catch {
|
|
117
|
+
return null;
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
|
|
95
121
|
/**
|
|
96
122
|
* Back on the mission's branch for a resume. Returns null when already or
|
|
97
123
|
* now there, else why not — a dirty tree that would be clobbered, typically.
|
package/src/mcp.test.ts
CHANGED
|
@@ -54,12 +54,12 @@ test('run_status: crew, DONE WHEN from the mission doc, needs, and no wait on a
|
|
|
54
54
|
assert.match(r.text, /stopped at its budget cap/);
|
|
55
55
|
assert.match(r.text, /DONE WHEN 1\/2\n open: docs updated/);
|
|
56
56
|
assert.match(r.text, /worker-1 · done/);
|
|
57
|
-
assert.match(r.text,
|
|
57
|
+
assert.match(r.text, /^finished$/m);
|
|
58
58
|
assert.ok(!calls.some((c) => c.path.startsWith('/events')), 'a finished run is never waited on');
|
|
59
59
|
});
|
|
60
60
|
|
|
61
|
-
test('run_status with a wait
|
|
62
|
-
const sse = new ReadableStream<Uint8Array>({
|
|
61
|
+
test('run_status with a wait returns when the run\'s picture changes, not on a bare event', async () => {
|
|
62
|
+
const sse = () => new ReadableStream<Uint8Array>({
|
|
63
63
|
start(c) {
|
|
64
64
|
const enc = new TextEncoder();
|
|
65
65
|
c.enqueue(enc.encode(': connected\n\n'));
|
|
@@ -67,18 +67,68 @@ test('run_status with a wait subscribes to /events and returns on the first even
|
|
|
67
67
|
c.enqueue(enc.encode('event: worker_started\ndata: {"runId":"r1","projectId":"p1","data":{"id":"worker-2"}}\n\n'));
|
|
68
68
|
},
|
|
69
69
|
});
|
|
70
|
+
let rounds = 0;
|
|
70
71
|
const { fetchImpl } = fakeServer({
|
|
71
|
-
|
|
72
|
+
// First round: an event, but the same run — no change. Second: a new worker.
|
|
73
|
+
'GET /runs': () => ({ runs: [run(rounds >= 2 ? { workers: [{ id: 'worker-1', status: 'done', costUsd: 0.4, task: 'write tests' }, { id: 'worker-2', status: 'running', costUsd: 0, task: 'more' }] } : {})] }),
|
|
72
74
|
'GET /projects': { projects: [{ id: 'p1', name: 'app', folder: '/x/app', activeRun: run(), needs: [{ kind: 'perm', id: 'a1', runId: 'r1', text: 'director wants Bash — rm -rf dist' }] }] },
|
|
73
75
|
'GET /missiondoc': { doc: '' },
|
|
74
|
-
'GET /events': new Response(sse, { status: 200, headers: { 'content-type': 'text/event-stream' } }),
|
|
76
|
+
'GET /events': () => { rounds += 1; return new Response(sse(), { status: 200, headers: { 'content-type': 'text/event-stream' } }); },
|
|
75
77
|
});
|
|
76
78
|
const r = await tool(foremanTools({ base: 'http://f', fetchImpl }), 'run_status').run({ runId: 'r1', wait_seconds: 5 });
|
|
77
79
|
assert.match(r.text, /changed: yes/);
|
|
80
|
+
assert.ok(rounds >= 2, 'the first event changed nothing visible, so it kept waiting');
|
|
81
|
+
assert.match(r.text, /worker-2 · running/);
|
|
78
82
|
assert.match(r.text, /NEEDS YOU \(1\) — only a human can answer/);
|
|
79
83
|
assert.match(r.text, /\[perm\] director wants Bash/);
|
|
80
84
|
});
|
|
81
85
|
|
|
86
|
+
test('run_status until=finished waits through events that are not the end, and stops at the end', async () => {
|
|
87
|
+
// Two rounds: a cost tick (not finished) then the run turns done.
|
|
88
|
+
let rounds = 0;
|
|
89
|
+
const sseOnce = () => new ReadableStream<Uint8Array>({ start(c) { c.enqueue(new TextEncoder().encode('event: cost\ndata: {"runId":"r1","projectId":"p1","data":{}}\n\n')); } });
|
|
90
|
+
const { fetchImpl, calls } = fakeServer({
|
|
91
|
+
'GET /runs': () => ({ runs: [run({ status: rounds >= 2 ? 'done' : 'running' })] }),
|
|
92
|
+
'GET /projects': { projects: [{ id: 'p1', name: 'app', folder: '/x/app', activeRun: null, needs: [] }] },
|
|
93
|
+
'GET /missiondoc': { doc: '' },
|
|
94
|
+
'GET /events': () => { rounds += 1; return new Response(sseOnce(), { status: 200 }); },
|
|
95
|
+
});
|
|
96
|
+
const r = await tool(foremanTools({ base: 'http://f', fetchImpl }), 'run_status').run({ runId: 'r1', wait_seconds: 10, until: 'finished' });
|
|
97
|
+
assert.match(r.text, /r1 · done/);
|
|
98
|
+
assert.match(r.text, /^finished$/m);
|
|
99
|
+
assert.ok(calls.filter((c) => c.path.startsWith('/events')).length >= 2, 'kept waiting past the first event');
|
|
100
|
+
});
|
|
101
|
+
|
|
102
|
+
test('run_status until=needs_you returns when an approval appears', async () => {
|
|
103
|
+
let asked = false;
|
|
104
|
+
const { fetchImpl } = fakeServer({
|
|
105
|
+
'GET /runs': { runs: [run()] },
|
|
106
|
+
'GET /projects': () => ({ projects: [{ id: 'p1', name: 'app', folder: '/x/app', activeRun: run(), needs: asked ? [{ kind: 'perm', id: 'a1', runId: 'r1', text: 'director wants Bash' }] : [] }] }),
|
|
107
|
+
'GET /missiondoc': { doc: '' },
|
|
108
|
+
'GET /events': () => { asked = true; return new Response(new ReadableStream<Uint8Array>({ start(c) { c.enqueue(new TextEncoder().encode('event: permission_request\ndata: {"runId":"r1","projectId":"p1","data":{}}\n\n')); } }), { status: 200 }); },
|
|
109
|
+
});
|
|
110
|
+
const r = await tool(foremanTools({ base: 'http://f', fetchImpl }), 'run_status').run({ runId: 'r1', wait_seconds: 5, until: 'needs_you' });
|
|
111
|
+
assert.match(r.text, /^needs you$/m);
|
|
112
|
+
assert.match(r.text, /\[perm\] director wants Bash/);
|
|
113
|
+
});
|
|
114
|
+
|
|
115
|
+
test('run_report: the director\'s last words, DONE WHEN, changed files and the branch in one call', async () => {
|
|
116
|
+
const say = (agent: string, text: string) => ({ ts: 1, event: 'message', data: { agent, msg: { type: 'assistant', message: { content: [{ type: 'text', text }] } } } });
|
|
117
|
+
const { fetchImpl } = fakeServer({
|
|
118
|
+
'GET /runs': { runs: [run({ status: 'done', git: { branch: 'foreman/fix-r1', base: 'main', commits: 1, pr: 'https://github.com/o/r/pull/9', prState: 'merged' } })] },
|
|
119
|
+
'GET /runs/r1/events': { events: [say('director', 'Starting.'), say('worker-1', 'A long worker message that must not be mistaken for the report.'), say('director', 'All done: tests pass, docs updated, nothing left undone in this mission.')] },
|
|
120
|
+
'GET /missiondoc': { doc: '- [x] tests pass\n- [x] docs updated\n' },
|
|
121
|
+
'GET /runs/r1/deck': { baseline: { kind: 'git' }, files: [{ path: 'src/a.ts', status: 'modified', additions: 10, deletions: 2 }, { path: 'old.txt', status: 'modified', additions: 1, deletions: 1, preexisting: true }], artifacts: [{ path: 'shot.png', kind: 'image' }], totals: { files: 2, additions: 11, deletions: 3 } },
|
|
122
|
+
});
|
|
123
|
+
const r = await tool(foremanTools({ base: 'http://f', fetchImpl }), 'run_report').run({ runId: 'r1' });
|
|
124
|
+
assert.match(r.text, /branch foreman\/fix-r1 from main · 1 commit · PR https:\/\/github.com\/o\/r\/pull\/9 \(merged\)/);
|
|
125
|
+
assert.match(r.text, /DONE WHEN 2\/2/);
|
|
126
|
+
assert.match(r.text, /changed: 1 file · \+11 −3 · 1 screenshot · 1 already dirty before the run/);
|
|
127
|
+
assert.match(r.text, /modified src\/a\.ts \+10 −2/);
|
|
128
|
+
assert.match(r.text, /Director's report:\nAll done: tests pass/);
|
|
129
|
+
assert.doesNotMatch(r.text, /worker message/);
|
|
130
|
+
});
|
|
131
|
+
|
|
82
132
|
test('waitForRunEvent gives up at the timeout when nothing arrives for the run', async () => {
|
|
83
133
|
const quiet = new ReadableStream<Uint8Array>({ start(c) { c.enqueue(new TextEncoder().encode(': connected\n\n')); } });
|
|
84
134
|
const { fetchImpl } = fakeServer({ 'GET /events': new Response(quiet, { status: 200 }) });
|
|
@@ -114,7 +164,7 @@ test('the tool set has no human-only actions', () => {
|
|
|
114
164
|
for (const forbidden of ['approve', 'deny', 'permission', 'answer', 'interrupt', 'resume', 'budget', 'pull_request', 'open_pr', 'settings', 'key']) {
|
|
115
165
|
assert.ok(!names.some((n) => n.split('_').includes(forbidden) || n === forbidden), `${forbidden} must not be a tool`);
|
|
116
166
|
}
|
|
117
|
-
assert.deepEqual(names, ['fleet_status', 'list_runs', 'run_status', 'run_transcript', 'mission_doc', 'project_memory', 'search_runs', 'doctor', 'link_project', 'start_mission', 'steer']);
|
|
167
|
+
assert.deepEqual(names, ['fleet_status', 'list_runs', 'run_status', 'run_report', 'run_transcript', 'mission_doc', 'project_memory', 'search_runs', 'doctor', 'link_project', 'start_mission', 'steer']);
|
|
118
168
|
});
|
|
119
169
|
|
|
120
170
|
test('a server that is not there is said in one sentence with the start command', async () => {
|
package/src/mcp.ts
CHANGED
|
@@ -222,22 +222,63 @@ export function foremanTools(opts: ForemanClientOptions): ToolDef[] {
|
|
|
222
222
|
},
|
|
223
223
|
};
|
|
224
224
|
|
|
225
|
+
/**
|
|
226
|
+
* What a reader of run_status would notice changing. `until: 'any'` waits
|
|
227
|
+
* for THIS to move, not for the next event: a fresh run emits a cost frame
|
|
228
|
+
* per SDK message, and returning on those meant "changed: true" twice in a
|
|
229
|
+
* row with identical payloads — a client had to poll after all.
|
|
230
|
+
*/
|
|
231
|
+
async function digestOf(id: string): Promise<string> {
|
|
232
|
+
const r = await findRun(id);
|
|
233
|
+
const needs = await needsOf(id);
|
|
234
|
+
const doc = await get<{ doc: string }>(`/missiondoc?run=${encodeURIComponent(id)}`).then((d) => d.doc ?? '').catch(() => '');
|
|
235
|
+
const dw = doneWhen(doc);
|
|
236
|
+
return JSON.stringify([r.status, r.stopReason ?? null, (r.costUsd ?? 0).toFixed(2), (r.workers ?? []).map((w) => `${w.id}:${w.status}`), dw.done.length, dw.open.length, needs.map((n) => n.id), r.git?.commits ?? 0, r.git?.pr ?? null]);
|
|
237
|
+
}
|
|
238
|
+
|
|
239
|
+
/** What is pending on a run right now, from the fleet payload. */
|
|
240
|
+
async function needsOf(id: string): Promise<Need[]> {
|
|
241
|
+
const projects = (await get<{ projects: ProjectCard[] }>('/projects')).projects;
|
|
242
|
+
const card = projects.find((p) => p.activeRun?.id === id);
|
|
243
|
+
return (card?.needs ?? []).filter((n) => !n.runId || n.runId === id);
|
|
244
|
+
}
|
|
245
|
+
|
|
225
246
|
const runStatus: ToolDef = {
|
|
226
247
|
name: 'run_status',
|
|
227
|
-
description: 'One run: status, spend against cap, crew and their states, DONE WHEN ticks, what needs a human, branch and pull request. With wait_seconds > 0 it
|
|
228
|
-
schema: {
|
|
229
|
-
|
|
248
|
+
description: 'One run: status, spend against cap, crew and their states, DONE WHEN ticks, what needs a human, branch and pull request. With wait_seconds > 0 it blocks until `until` is met — "any" change on the run, the run "finished", or it "needs_you" (a pending approval or question, or finished) — or until the timeout. Use it instead of polling; call again if it timed out.',
|
|
249
|
+
schema: {
|
|
250
|
+
runId: z.string(),
|
|
251
|
+
wait_seconds: z.number().int().min(0).max(maxWait).default(0),
|
|
252
|
+
until: z.enum(['any', 'finished', 'needs_you']).default('any'),
|
|
253
|
+
},
|
|
254
|
+
run: async ({ runId, wait_seconds, until }) => {
|
|
230
255
|
const id = String(runId);
|
|
256
|
+
const cond = String(until ?? 'any');
|
|
231
257
|
let changed: boolean | undefined;
|
|
232
258
|
if (Number(wait_seconds) > 0) {
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
259
|
+
// Wait in rounds: each round ends on the first event for the run,
|
|
260
|
+
// then the condition is checked against fresh state; an event that
|
|
261
|
+
// does not satisfy it (a cost tick, under "finished") starts another
|
|
262
|
+
// round with the time that is left. The clock is the outer bound.
|
|
263
|
+
const deadline = Date.now() + Number(wait_seconds) * 1000;
|
|
264
|
+
changed = false;
|
|
265
|
+
const before = await digestOf(id);
|
|
266
|
+
for (;;) {
|
|
267
|
+
const now = await findRun(id);
|
|
268
|
+
const satisfied = now.status !== 'running'
|
|
269
|
+
|| (cond === 'needs_you' && (await needsOf(id)).length > 0)
|
|
270
|
+
|| (cond === 'any' && changed);
|
|
271
|
+
if (satisfied) break;
|
|
272
|
+
const left = Math.ceil((deadline - Date.now()) / 1000);
|
|
273
|
+
if (left <= 0) break;
|
|
274
|
+
const got = await waitForRunEvent(base, id, left, f);
|
|
275
|
+
if (!got) break;
|
|
276
|
+
// An event arrived; only count it when the picture it paints differs.
|
|
277
|
+
if ((await digestOf(id)) !== before) changed = true;
|
|
278
|
+
}
|
|
236
279
|
}
|
|
237
280
|
const r = await findRun(id);
|
|
238
|
-
const
|
|
239
|
-
const card = projects.find((p) => p.activeRun?.id === id);
|
|
240
|
-
const needs = (card?.needs ?? []).filter((n) => !n.runId || n.runId === id);
|
|
281
|
+
const needs = await needsOf(id);
|
|
241
282
|
const doc = await get<{ doc: string }>(`/missiondoc?run=${encodeURIComponent(id)}`).then((d) => d.doc).catch(() => '');
|
|
242
283
|
const dw = doneWhen(doc);
|
|
243
284
|
const workers = (r.workers ?? []).map((w) => ` ${w.id} · ${w.status} · ${usd(w.costUsd)} · ${w.task.slice(0, 80)}`);
|
|
@@ -248,9 +289,47 @@ export function foremanTools(opts: ForemanClientOptions): ToolDef[] {
|
|
|
248
289
|
dw.done.length + dw.open.length ? `DONE WHEN ${dw.done.length}/${dw.done.length + dw.open.length}${dw.open.length ? `\n open: ${dw.open.join('\n open: ')}` : ''}` : null,
|
|
249
290
|
workers.length ? `crew:\n${workers.join('\n')}` : null,
|
|
250
291
|
needs.length ? `NEEDS YOU (${needs.length}) — only a human can answer these, on the dashboard or the phone:\n${needs.map((n) => ` [${n.kind}] ${n.text}`).join('\n')}` : null,
|
|
251
|
-
changed !== undefined ? (changed ? 'changed: yes' : 'changed: no (timeout)') : null,
|
|
292
|
+
changed !== undefined ? (r.status !== 'running' ? 'finished' : needs.length && cond === 'needs_you' ? 'needs you' : changed ? 'changed: yes' : 'changed: no (timeout — call again)') : null,
|
|
293
|
+
].filter(Boolean);
|
|
294
|
+
return { text: lines.join('\n'), data: { run: r, doneWhen: dw, needs, changed, waitedFor: Number(wait_seconds) > 0 ? cond : undefined } };
|
|
295
|
+
},
|
|
296
|
+
};
|
|
297
|
+
|
|
298
|
+
const runReport: ToolDef = {
|
|
299
|
+
name: 'run_report',
|
|
300
|
+
description: 'What a finished run produced, in one call: the director\'s final report, DONE WHEN ticks, the files it changed with +/− counts, the branch, commit and pull request, spend and crew. For a running run it reports the state so far.',
|
|
301
|
+
schema: { runId: z.string() },
|
|
302
|
+
run: async ({ runId }) => {
|
|
303
|
+
const id = String(runId);
|
|
304
|
+
const r = await findRun(id);
|
|
305
|
+
const [events, docRes, deck] = await Promise.all([
|
|
306
|
+
get<{ events: Array<{ ts: number; event: string; data: Record<string, unknown> }> }>(`/runs/${encodeURIComponent(id)}/events`).then((d) => d.events).catch(() => []),
|
|
307
|
+
get<{ doc: string }>(`/missiondoc?run=${encodeURIComponent(id)}`).catch(() => ({ doc: '' })),
|
|
308
|
+
get<{ files: Array<{ path: string; status: string; additions: number; deletions: number; preexisting?: boolean }>; artifacts: Array<{ path: string; kind: string }>; totals: { files: number; additions: number; deletions: number }; baseline: { kind: string } }>(`/runs/${encodeURIComponent(id)}/deck`).catch(() => null),
|
|
309
|
+
]);
|
|
310
|
+
// The director's last words: the final assistant text before the run ended.
|
|
311
|
+
let report = '';
|
|
312
|
+
for (const e of events) {
|
|
313
|
+
if (e.event !== 'message' || e.data?.agent !== 'director') continue;
|
|
314
|
+
const msg = e.data.msg as { type?: string; message?: { content?: Array<{ type: string; text?: string }> } } | undefined;
|
|
315
|
+
const text = msg?.type === 'assistant' ? msg.message?.content?.filter((c) => c.type === 'text' && c.text).map((c) => c.text).join('\n') : '';
|
|
316
|
+
if (text && text.trim().length > 40) report = text.trim();
|
|
317
|
+
}
|
|
318
|
+
const dw = doneWhen(docRes.doc);
|
|
319
|
+
const files = deck?.files ?? [];
|
|
320
|
+
const own = files.filter((x) => !x.preexisting);
|
|
321
|
+
const fileLines = own.slice(0, 40).map((x) => ` ${x.status.padEnd(8)} ${x.path} +${x.additions} −${x.deletions}`);
|
|
322
|
+
const images = (deck?.artifacts ?? []).filter((a) => a.kind === 'image').length;
|
|
323
|
+
const lines = [
|
|
324
|
+
runLine(r),
|
|
325
|
+
r.git ? `branch ${r.git.branch} from ${r.git.base}${r.git.commits ? ` · ${r.git.commits} commit${r.git.commits === 1 ? '' : 's'}` : ''}${r.git.pr ? ` · PR ${r.git.pr}${r.git.prState ? ` (${r.git.prState})` : ''}` : ' · no pull request yet (the human opens it from the run page)'}` : 'not a git repository',
|
|
326
|
+
`DONE WHEN ${dw.done.length}/${dw.done.length + dw.open.length}${dw.open.length ? ` · open: ${dw.open.join(' · ')}` : ''}`,
|
|
327
|
+
deck ? `changed: ${own.length} file${own.length === 1 ? '' : 's'} · +${deck.totals.additions} −${deck.totals.deletions}${images ? ` · ${images} screenshot${images === 1 ? '' : 's'}` : ''}${files.length > own.length ? ` · ${files.length - own.length} already dirty before the run` : ''}` : null,
|
|
328
|
+
fileLines.length ? fileLines.join('\n') + (own.length > 40 ? `\n … ${own.length - 40} more` : '') : null,
|
|
329
|
+
`crew: ${(r.workers ?? []).length} worker${(r.workers ?? []).length === 1 ? '' : 's'} · ${(r.workers ?? []).filter((w) => w.status === 'done').length} done`,
|
|
330
|
+
report ? `\nDirector's report:\n${report.slice(0, 4000)}` : '\nNo final report from the director yet.',
|
|
252
331
|
].filter(Boolean);
|
|
253
|
-
return { text: lines.join('\n'), data: { run: r, doneWhen: dw,
|
|
332
|
+
return { text: lines.join('\n'), data: { run: r, doneWhen: dw, files: own, totals: deck?.totals, report } };
|
|
254
333
|
},
|
|
255
334
|
};
|
|
256
335
|
|
|
@@ -377,7 +456,7 @@ export function foremanTools(opts: ForemanClientOptions): ToolDef[] {
|
|
|
377
456
|
},
|
|
378
457
|
};
|
|
379
458
|
|
|
380
|
-
return [fleet, listRuns, runStatus, transcript, missionDoc, memory, search, doctor, link, start, steer];
|
|
459
|
+
return [fleet, listRuns, runStatus, runReport, transcript, missionDoc, memory, search, doctor, link, start, steer];
|
|
381
460
|
}
|
|
382
461
|
|
|
383
462
|
/** Runs the MCP server over stdio until the client goes away. Nothing may be written to stdout but the protocol. */
|
package/src/notify.ts
CHANGED
|
@@ -234,10 +234,14 @@ export function shape(env: Envelope, ctx: NotifyContext): Shaped | null {
|
|
|
234
234
|
// --- money and completion -----------------------------------------
|
|
235
235
|
case 'budget_alert':
|
|
236
236
|
return { key: `budget:${env.runId}:${d.level}`, gate: 'budget',
|
|
237
|
-
text: `${head(d.level === 'exceeded' ? 'Budget exceeded — run interrupted'
|
|
237
|
+
text: `${head(d.level === 'exceeded' ? 'Budget exceeded — run interrupted'
|
|
238
|
+
: d.level === 'warn' ? 'Budget warning — winding down' : 'Budget cap reached')}${runLine}\n${esc(clip(d.text))}${foot}` };
|
|
238
239
|
case 'budget_stop':
|
|
239
240
|
return { key: `stop:${env.runId}`, gate: 'budget',
|
|
240
241
|
text: `${head('Run winding down')}${runLine}\n${esc(clip(d.reason))}${foot}` };
|
|
242
|
+
case 'mission_done_at_cap':
|
|
243
|
+
return { key: `donecap:${env.runId}`, gate: 'done',
|
|
244
|
+
text: `${head('Done, at the cap')}${runLine}\n${esc(clip(d.text))}${foot}` };
|
|
241
245
|
case 'mission_incomplete':
|
|
242
246
|
return { key: `incomplete:${env.runId}`, gate: 'done',
|
|
243
247
|
text: `${head('Not done')}${runLine}\n${esc(clip(d.text))}${foot}` };
|
package/src/orchestrator.test.ts
CHANGED
|
@@ -9,7 +9,7 @@ import {
|
|
|
9
9
|
stalledWorkerReport, workerStatusBlock,
|
|
10
10
|
watchRepeats, watchSilence, REPEAT_EXEMPT, observeToolUse,
|
|
11
11
|
DEFAULT_ASK_TIMEOUT_MS, armAskTimeout, unattendedAnswer, unattendedDenyMessage,
|
|
12
|
-
tokenCapLabel, stopReasonOf,
|
|
12
|
+
tokenCapLabel, stopReasonOf, budgetWarnDue, doneAtCap,
|
|
13
13
|
} from './orchestrator.js';
|
|
14
14
|
import { makePolicy, type PendingPermission } from './policy.js';
|
|
15
15
|
import type { AgentEnv } from './provider.js';
|
|
@@ -1256,3 +1256,27 @@ test('stopReasonOf names the cap behind a capReached sentence', () => {
|
|
|
1256
1256
|
assert.equal(stopReasonOf('TIME CAP REACHED: 240 minutes.'), 'time');
|
|
1257
1257
|
assert.equal(stopReasonOf('TOKEN CAP REACHED: 20.1M tokens.'), 'tokens');
|
|
1258
1258
|
});
|
|
1259
|
+
|
|
1260
|
+
test('budgetWarnDue: once past the threshold, and only while still under the cap', () => {
|
|
1261
|
+
assert.equal(budgetWarnDue(7.9, 10, 80), false);
|
|
1262
|
+
assert.equal(budgetWarnDue(8, 10, 80), true);
|
|
1263
|
+
assert.equal(budgetWarnDue(9.99, 10, 80), true);
|
|
1264
|
+
// At and past the cap the wind-down order speaks instead.
|
|
1265
|
+
assert.equal(budgetWarnDue(10, 10, 80), false);
|
|
1266
|
+
assert.equal(budgetWarnDue(12, 10, 80), false);
|
|
1267
|
+
// No threshold, a nonsensical one, or no cap: nothing to warn about.
|
|
1268
|
+
assert.equal(budgetWarnDue(9, 10, undefined), false);
|
|
1269
|
+
assert.equal(budgetWarnDue(9, 10, 0), false);
|
|
1270
|
+
assert.equal(budgetWarnDue(9, 10, 100), false);
|
|
1271
|
+
assert.equal(budgetWarnDue(9, 0, 80), false);
|
|
1272
|
+
});
|
|
1273
|
+
|
|
1274
|
+
test('doneAtCap: budget-stopped with every criterion ticked is done, and nothing else is', () => {
|
|
1275
|
+
const budget = { budgetStopped: true, wasInterrupted: false, usageLimited: false };
|
|
1276
|
+
assert.equal(doneAtCap(budget, []), true);
|
|
1277
|
+
assert.equal(doneAtCap(budget, ['screenshots saved']), false, 'an unticked box is not done');
|
|
1278
|
+
assert.equal(doneAtCap(budget, null), false, 'no DONE WHEN section says nothing either way');
|
|
1279
|
+
assert.equal(doneAtCap({ ...budget, wasInterrupted: true }, []), false, 'a human stop is not a finish');
|
|
1280
|
+
assert.equal(doneAtCap({ ...budget, usageLimited: true }, []), false, 'a usage limit is not a finish');
|
|
1281
|
+
assert.equal(doneAtCap({ ...budget, budgetStopped: false }, []), false);
|
|
1282
|
+
});
|
package/src/orchestrator.ts
CHANGED
|
@@ -266,6 +266,31 @@ const DEFAULT_MAX_TURNS = 150;
|
|
|
266
266
|
export const DEFAULT_MAX_TOKENS = 20_000_000;
|
|
267
267
|
|
|
268
268
|
/** The cap behind a capReached() sentence, as the run record stores it. */
|
|
269
|
+
/**
|
|
270
|
+
* Is it time to tell the director to start winding down? True once spend
|
|
271
|
+
* crosses the warn threshold and while it is still under the cap — past the
|
|
272
|
+
* cap the wind-down order takes over, and a warning then would be noise.
|
|
273
|
+
*/
|
|
274
|
+
export function budgetWarnDue(costUsd: number, budgetUsd: number, warnAt: number | undefined): boolean {
|
|
275
|
+
if (!Number.isFinite(warnAt as number) || (warnAt as number) <= 0 || (warnAt as number) >= 100) return false;
|
|
276
|
+
if (!(budgetUsd > 0)) return false;
|
|
277
|
+
return costUsd >= budgetUsd * ((warnAt as number) / 100) && costUsd < budgetUsd;
|
|
278
|
+
}
|
|
279
|
+
|
|
280
|
+
/**
|
|
281
|
+
* A run the budget stopped, whose own DONE WHEN criteria are all ticked, is
|
|
282
|
+
* done: it reached its criteria and then reached its limit, in that order.
|
|
283
|
+
* Only for the budget — a human interrupt or a usage limit says nothing
|
|
284
|
+
* about the work — and only when the doc actually had criteria to tick.
|
|
285
|
+
*/
|
|
286
|
+
export function doneAtCap(
|
|
287
|
+
flags: { budgetStopped: boolean; wasInterrupted: boolean; usageLimited: boolean },
|
|
288
|
+
unmet: string[] | null,
|
|
289
|
+
): boolean {
|
|
290
|
+
if (!flags.budgetStopped || flags.wasInterrupted || flags.usageLimited) return false;
|
|
291
|
+
return Array.isArray(unmet) && unmet.length === 0;
|
|
292
|
+
}
|
|
293
|
+
|
|
269
294
|
export function stopReasonOf(cap: string): 'budget' | 'turns' | 'time' | 'tokens' {
|
|
270
295
|
if (cap.startsWith('TURN')) return 'turns';
|
|
271
296
|
if (cap.startsWith('TIME')) return 'time';
|
|
@@ -604,6 +629,8 @@ type AgentRole = 'director' | 'worker';
|
|
|
604
629
|
* and which role's rates price the tokens.
|
|
605
630
|
*/
|
|
606
631
|
interface WorkerOverrides {
|
|
632
|
+
/** Set on the one continuation a worker gets after the turn cap. */
|
|
633
|
+
continued?: boolean;
|
|
607
634
|
env?: AgentEnv;
|
|
608
635
|
model?: string;
|
|
609
636
|
priceRole?: AgentRole;
|
|
@@ -746,6 +773,19 @@ const USAGE_LIMIT_RE = /out of usage credits|usage limit reached|upgrade to incr
|
|
|
746
773
|
/** One worker's outcome, as runWorker hands it back. */
|
|
747
774
|
interface WorkerOutcome { report: string; isError: boolean }
|
|
748
775
|
|
|
776
|
+
/**
|
|
777
|
+
* Turns a worker gets before the SDK stops it. Raised from 60: five workers
|
|
778
|
+
* in four missions hit the cap mid-task on legitimate, brief-sized work
|
|
779
|
+
* (a hundred TypeScript errors in one package; a test-suite port), and each
|
|
780
|
+
* time the director paid to respawn one that re-read the same files.
|
|
781
|
+
*/
|
|
782
|
+
const WORKER_MAX_TURNS = 100;
|
|
783
|
+
/** What a worker stopped by the cap is told when its session is picked back up. */
|
|
784
|
+
const WORKER_CONTINUE_PROMPT =
|
|
785
|
+
'You were stopped by the turn cap, not by a failure. Your session and your files are as you left them; ' +
|
|
786
|
+
'`git status` and `git diff` show your uncommitted work. Continue the same task from where you were — do not ' +
|
|
787
|
+
'start over or re-read what you already know — finish it, report with report_progress, and end your turn.';
|
|
788
|
+
|
|
749
789
|
/**
|
|
750
790
|
* A worker record plus the in-process handles that must never be persisted:
|
|
751
791
|
* the live query (for interrupts), the run promise (so nothing is left
|
|
@@ -1065,6 +1105,7 @@ export class MissionRun {
|
|
|
1065
1105
|
// overrun quiet.
|
|
1066
1106
|
this.budgetNoticeSent = false;
|
|
1067
1107
|
this.budgetKillSent = false;
|
|
1108
|
+
this.budgetWarnSent = false;
|
|
1068
1109
|
this.budgetStopped = false;
|
|
1069
1110
|
}
|
|
1070
1111
|
}
|
|
@@ -1317,6 +1358,23 @@ export class MissionRun {
|
|
|
1317
1358
|
// the one lie a mission runner cannot afford. Downgrading to
|
|
1318
1359
|
// 'interrupted' is also the useful answer: it is what makes the run
|
|
1319
1360
|
// resumable rather than closed.
|
|
1361
|
+
// A mission whose own record says every criterion is verified is done,
|
|
1362
|
+
// even if the cap ended the turn it was writing its report in. Calling
|
|
1363
|
+
// that 'interrupted' told the fleet a finished mission had failed, and
|
|
1364
|
+
// invited a resume that spent more to rewrite a report already on disk.
|
|
1365
|
+
// The stop reason stays on the record, so nothing is hidden.
|
|
1366
|
+
if (this.meta.status === 'interrupted' && this.budgetStopped
|
|
1367
|
+
&& !this.wasInterrupted && !this.usageLimited) {
|
|
1368
|
+
const unmet = await this.unmetCriteria();
|
|
1369
|
+
if (doneAtCap({ budgetStopped: this.budgetStopped, wasInterrupted: this.wasInterrupted, usageLimited: this.usageLimited }, unmet)) {
|
|
1370
|
+
this.meta.status = 'done';
|
|
1371
|
+
this.emit('mission_done_at_cap', {
|
|
1372
|
+
costUsd: this.meta.costUsd, budgetUsd: this.meta.budgetUsd, reason: this.meta.stopReason,
|
|
1373
|
+
text: 'Every DONE WHEN criterion was verified before the cap ended the run, so this mission is done. ' +
|
|
1374
|
+
'It stopped at its limit rather than finishing under it — the report may be shorter than usual.',
|
|
1375
|
+
});
|
|
1376
|
+
}
|
|
1377
|
+
}
|
|
1320
1378
|
if (this.meta.status === 'done') {
|
|
1321
1379
|
const unmet = await this.unmetCriteria();
|
|
1322
1380
|
if (unmet?.length) {
|
|
@@ -1467,6 +1525,7 @@ export class MissionRun {
|
|
|
1467
1525
|
|
|
1468
1526
|
private budgetNoticeSent = false;
|
|
1469
1527
|
private budgetKillSent = false;
|
|
1528
|
+
private budgetWarnSent = false;
|
|
1470
1529
|
|
|
1471
1530
|
/**
|
|
1472
1531
|
* Keeps an "always allow" grant out of `git status`.
|
|
@@ -1637,6 +1696,26 @@ export class MissionRun {
|
|
|
1637
1696
|
void this.interrupt();
|
|
1638
1697
|
return;
|
|
1639
1698
|
}
|
|
1699
|
+
// Before the fence, a nudge. A director that only learns of the cap when
|
|
1700
|
+
// it hits it does its verification and its report inside the one turn it
|
|
1701
|
+
// has left — which is how two missions that had finished their work were
|
|
1702
|
+
// cut mid-report and read as failures. Warn while there is still room to
|
|
1703
|
+
// wind down deliberately.
|
|
1704
|
+
if (!this.budgetWarnSent && budgetWarnDue(costUsd, budgetUsd, this.meta.budgetWarnAt)) {
|
|
1705
|
+
this.budgetWarnSent = true;
|
|
1706
|
+
const pct = Math.round((costUsd / budgetUsd) * 100);
|
|
1707
|
+
this.emit('budget_alert', {
|
|
1708
|
+
level: 'warn', costUsd, budgetUsd,
|
|
1709
|
+
text: `${pct}% of the budget spent ($${costUsd.toFixed(2)} of $${budgetUsd.toFixed(2)}) — director told to start verifying.`,
|
|
1710
|
+
});
|
|
1711
|
+
this.directorInput?.push(
|
|
1712
|
+
'[BUDGET — automated notice]\n' +
|
|
1713
|
+
`You have spent $${costUsd.toFixed(2)} of the $${budgetUsd.toFixed(2)} cap (${pct}%). ` +
|
|
1714
|
+
'Start winding down now: finish or stop the work in flight, do not begin anything you ' +
|
|
1715
|
+
'cannot complete and verify within what is left, and get your verification and MISSION.md ' +
|
|
1716
|
+
'up to date. At the cap you get one final turn, and at 125% the run is interrupted.',
|
|
1717
|
+
);
|
|
1718
|
+
}
|
|
1640
1719
|
if (!this.budgetNoticeSent && costUsd >= budgetUsd) {
|
|
1641
1720
|
this.budgetNoticeSent = true;
|
|
1642
1721
|
this.emit('budget_alert', {
|
|
@@ -1983,7 +2062,7 @@ export class MissionRun {
|
|
|
1983
2062
|
permissionMode: 'default',
|
|
1984
2063
|
resume: resumeSessionId,
|
|
1985
2064
|
model: overrides?.model || this.meta.workerModel,
|
|
1986
|
-
maxTurns:
|
|
2065
|
+
maxTurns: WORKER_MAX_TURNS,
|
|
1987
2066
|
systemPrompt: { type: 'preset', preset: 'claude_code', append: WORKER_CHARTER },
|
|
1988
2067
|
...(overrides?.env ?? this.agentEnv.worker),
|
|
1989
2068
|
// Built per worker: the report_progress handler closes over this id,
|
|
@@ -1996,6 +2075,7 @@ export class MissionRun {
|
|
|
1996
2075
|
w.q = q;
|
|
1997
2076
|
|
|
1998
2077
|
let report = '';
|
|
2078
|
+
let hitTurnCap = false;
|
|
1999
2079
|
let isError = false;
|
|
2000
2080
|
|
|
2001
2081
|
// The stall watchdog. A silent worker no longer blocks the director, but
|
|
@@ -2041,6 +2121,7 @@ export class MissionRun {
|
|
|
2041
2121
|
if (m.type === 'result') {
|
|
2042
2122
|
report = String(m.result ?? '');
|
|
2043
2123
|
isError = Boolean(m.is_error);
|
|
2124
|
+
hitTurnCap = m.subtype === 'error_max_turns' || /maximum number of turns/i.test(report);
|
|
2044
2125
|
this.noteUsageLimit(report);
|
|
2045
2126
|
// Same ordering rule as the director loop: the cost event carries
|
|
2046
2127
|
// usage, so usage has to be current before it is emitted.
|
|
@@ -2060,6 +2141,16 @@ export class MissionRun {
|
|
|
2060
2141
|
w.q = undefined;
|
|
2061
2142
|
}
|
|
2062
2143
|
|
|
2144
|
+
// A worker stopped by the turn cap mid-task is not a failed worker. Its
|
|
2145
|
+
// session is resumed once with a continue turn — the same context, the
|
|
2146
|
+
// same files — instead of handing the director an error it can only
|
|
2147
|
+
// answer by spawning a replacement that re-reads everything. Once: a
|
|
2148
|
+
// worker that burns two allowances is stuck, and that IS the director's.
|
|
2149
|
+
if (hitTurnCap && !overrides?.continued && w.sessionId && !stalled && !looping) {
|
|
2150
|
+
this.emit('worker_progress', { id: workerId, status: `reached the ${WORKER_MAX_TURNS}-turn cap mid-task; continuing the same session once` });
|
|
2151
|
+
return this.runWorker(workerId, WORKER_CONTINUE_PROMPT, w.sessionId, { ...overrides, continued: true });
|
|
2152
|
+
}
|
|
2153
|
+
|
|
2063
2154
|
// Told to the director as a fact plus its options, not as an order: it is
|
|
2064
2155
|
// the agent with the context to know whether this needs a different
|
|
2065
2156
|
// approach, a different worker, or a human.
|