@amenophis1er/foreman 0.1.7 → 0.1.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -3
- package/package.json +1 -1
- package/src/browser.ts +188 -0
- package/src/cli.ts +6 -1
- package/src/clone.test.ts +35 -0
- package/src/clone.ts +129 -0
- package/src/deck.test.ts +2 -2
- package/src/deck.ts +32 -6
- package/src/fleet-planner.ts +5 -4
- package/src/gitwork.test.ts +95 -0
- package/src/gitwork.ts +239 -0
- package/src/guard.test.ts +104 -0
- package/src/guard.ts +131 -0
- package/src/http-body.test.ts +42 -0
- package/src/http-body.ts +57 -0
- package/src/memory.test.ts +46 -0
- package/src/memory.ts +107 -0
- package/src/notify/commands.ts +1 -1
- package/src/notify/telegram.ts +1 -1
- package/src/notify.ts +5 -0
- package/src/orchestrator.test.ts +104 -0
- package/src/orchestrator.ts +109 -47
- package/src/planner.ts +7 -2
- package/src/preflight.ts +29 -43
- package/src/role-provider.test.ts +55 -0
- package/src/role-provider.ts +65 -0
- package/src/server.ts +548 -63
- package/src/services.test.ts +36 -1
- package/src/services.ts +60 -3
- package/src/store.test.ts +18 -0
- package/src/store.ts +16 -1
- package/src/track-record.test.ts +46 -0
- package/src/track-record.ts +149 -0
- package/src/types.ts +17 -0
- package/ui/dist/assets/index-h0osI3Nn.js +68 -0
- package/ui/dist/index.html +1 -1
- package/ui/dist/assets/index-BPFc7O5V.js +0 -67
package/src/memory.ts
ADDED
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Project memory: the page of notes a good contractor keeps about a site.
|
|
3
|
+
*
|
|
4
|
+
* `.foreman/MEMORY.md` — one file per project, human-readable, a page or two
|
|
5
|
+
* at most. How to run the tests, which port is taken, where the CSS lives
|
|
6
|
+
* and why, what the client hates. Read by the director, the workers and the
|
|
7
|
+
* planner at the start of every turn; written by the director at the end of
|
|
8
|
+
* a mission through one tool; visible in the rail so the human can see what
|
|
9
|
+
* their project "believes", and edit it with any text editor.
|
|
10
|
+
*
|
|
11
|
+
* Memory makes the crew better informed, never more powerful: it is prose
|
|
12
|
+
* in a prompt, and everything the crew may do is still exactly what the
|
|
13
|
+
* tool policy and the budget allow. It also never carries a secret — the
|
|
14
|
+
* server strips anything that looks like one before writing.
|
|
15
|
+
*/
|
|
16
|
+
import path from 'node:path';
|
|
17
|
+
import { mkdir, readFile, stat, writeFile } from 'node:fs/promises';
|
|
18
|
+
|
|
19
|
+
export const MEMORY_FILE = '.foreman/MEMORY.md';
|
|
20
|
+
/** Bytes kept in the file. Past this the director is asked to prune, not append. */
|
|
21
|
+
export const MEMORY_CAP_BYTES = 8 * 1024;
|
|
22
|
+
/** Bytes injected into a prompt; the tail is dropped with a note if the file is larger. */
|
|
23
|
+
const INJECT_CAP_BYTES = 6 * 1024;
|
|
24
|
+
|
|
25
|
+
export function memoryPath(folder: string): string {
|
|
26
|
+
return path.join(folder, MEMORY_FILE);
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export async function readMemory(folder: string): Promise<{ text: string; updatedAt?: number }> {
|
|
30
|
+
const file = memoryPath(folder);
|
|
31
|
+
const text = await readFile(file, 'utf8').catch(() => '');
|
|
32
|
+
const st = text ? await stat(file).catch(() => null) : null;
|
|
33
|
+
return { text, updatedAt: st?.mtimeMs };
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* Things that must not live in a page of notes, whatever the model meant:
|
|
38
|
+
* API keys, tokens, private keys, and `KEY=value` lines for secret-looking
|
|
39
|
+
* names. Replaced, not dropped, so the line still reads as a line.
|
|
40
|
+
*/
|
|
41
|
+
const SECRET_PATTERNS: RegExp[] = [
|
|
42
|
+
/-----BEGIN [A-Z ]*PRIVATE KEY-----[\s\S]*?-----END [A-Z ]*PRIVATE KEY-----/g,
|
|
43
|
+
/\b(sk|rk|pk)-(?:live|test|proj|ant)?-?[A-Za-z0-9_-]{16,}\b/g,
|
|
44
|
+
/\b(?:ghp|gho|ghu|ghs|ghr|github_pat)_[A-Za-z0-9_]{20,}\b/g,
|
|
45
|
+
/\bxox[abprs]-[A-Za-z0-9-]{10,}\b/g,
|
|
46
|
+
/\bAKIA[0-9A-Z]{16}\b/g,
|
|
47
|
+
/\bAIza[0-9A-Za-z_-]{30,}\b/g,
|
|
48
|
+
/\beyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\b/g,
|
|
49
|
+
/\b([A-Z0-9_]*(?:SECRET|TOKEN|PASSWORD|PASSWD|API_KEY|APIKEY|PRIVATE_KEY)[A-Z0-9_]*)\s*[=:]\s*["']?[^\s"']{8,}["']?/gi,
|
|
50
|
+
/\b(?:bearer|token|password|secret|api[_-]?key)\s*[:=]\s*["']?[A-Za-z0-9_\-./+=]{16,}["']?/gi,
|
|
51
|
+
];
|
|
52
|
+
|
|
53
|
+
/** Strip what looks like a credential. Returns the text and how many replacements were made. */
|
|
54
|
+
export function scrubSecrets(text: string): { text: string; redacted: number } {
|
|
55
|
+
let redacted = 0;
|
|
56
|
+
let out = text;
|
|
57
|
+
for (const re of SECRET_PATTERNS) {
|
|
58
|
+
out = out.replace(re, (m, name?: string) => {
|
|
59
|
+
redacted += 1;
|
|
60
|
+
// Keep the variable name when there was one, so the line still explains itself.
|
|
61
|
+
return typeof name === 'string' && /^[A-Z0-9_]+$/.test(name) && m.startsWith(name) ? `${name}=[redacted]` : '[redacted]';
|
|
62
|
+
});
|
|
63
|
+
}
|
|
64
|
+
return { text: out, redacted };
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
/**
|
|
68
|
+
* Write the whole memory. The director hands over the full page, not a
|
|
69
|
+
* delta, because "rewrite the stale line" is the behaviour we want and
|
|
70
|
+
* append-only files only grow. Trimmed to the cap; the caller is told.
|
|
71
|
+
*/
|
|
72
|
+
export async function writeMemory(folder: string, text: string): Promise<{ bytes: number; redacted: number; trimmed: boolean }> {
|
|
73
|
+
const scrubbed = scrubSecrets(text.replace(/\r\n/g, '\n').trim());
|
|
74
|
+
let body = scrubbed.text;
|
|
75
|
+
let trimmed = false;
|
|
76
|
+
if (Buffer.byteLength(body, 'utf8') > MEMORY_CAP_BYTES) {
|
|
77
|
+
body = Buffer.from(body, 'utf8').subarray(0, MEMORY_CAP_BYTES).toString('utf8').replace(/[^\n]*$/, '').trimEnd();
|
|
78
|
+
trimmed = true;
|
|
79
|
+
}
|
|
80
|
+
await mkdir(path.dirname(memoryPath(folder)), { recursive: true });
|
|
81
|
+
await writeFile(memoryPath(folder), body ? `${body}\n` : '');
|
|
82
|
+
return { bytes: Buffer.byteLength(body, 'utf8'), redacted: scrubbed.redacted, trimmed };
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/**
|
|
86
|
+
* The memory as a prompt section, for the director, the workers and the
|
|
87
|
+
* planner. Empty memory says so in one line rather than vanishing, so the
|
|
88
|
+
* crew knows the file exists to be written.
|
|
89
|
+
*/
|
|
90
|
+
export function memorySection(text: string, role: 'director' | 'worker' | 'planner'): string {
|
|
91
|
+
const t = text.trim();
|
|
92
|
+
if (!t) {
|
|
93
|
+
return role === 'director'
|
|
94
|
+
? `\nPROJECT MEMORY (${MEMORY_FILE}): empty. Before you finish, write it with mcp__foreman__remember — a page at most of facts the next crew needs here.\n`
|
|
95
|
+
: '';
|
|
96
|
+
}
|
|
97
|
+
let body = t;
|
|
98
|
+
if (Buffer.byteLength(body, 'utf8') > INJECT_CAP_BYTES) {
|
|
99
|
+
body = Buffer.from(body, 'utf8').subarray(0, INJECT_CAP_BYTES).toString('utf8').replace(/[^\n]*$/, '') + '\n[… memory is longer than a page; the rest was not shown. Prune it.]';
|
|
100
|
+
}
|
|
101
|
+
const lead = role === 'director'
|
|
102
|
+
? `PROJECT MEMORY (${MEMORY_FILE}) — what earlier crews learned about this project. Trust it over guessing; correct it when it is wrong. Before you finish, rewrite it with mcp__foreman__remember so the next crew starts where you end.`
|
|
103
|
+
: role === 'worker'
|
|
104
|
+
? 'PROJECT MEMORY — what earlier crews learned about this project. Trust it over guessing; tell the director if it is wrong.'
|
|
105
|
+
: `PROJECT MEMORY (${MEMORY_FILE}) — what the crew learned about this project on earlier missions. Ground your advice in it, and say when a proposal would change something it records.`;
|
|
106
|
+
return `\n${lead}\n---\n${body}\n---\n`;
|
|
107
|
+
}
|
package/src/notify/commands.ts
CHANGED
|
@@ -66,7 +66,7 @@ export const HELP_TEXT = [
|
|
|
66
66
|
'',
|
|
67
67
|
'/projects — the fleet, with what is running',
|
|
68
68
|
'/status — the runs in flight and what they need',
|
|
69
|
-
'/new <name> — create a project under your projects root and link it',
|
|
69
|
+
'/new <name> — create a project under your projects root and link it; /new <git url> clones it there first',
|
|
70
70
|
'/plan <project> <what you want> — talk to that project\'s planner',
|
|
71
71
|
'/run <project> <brief> — skip the talk: start a mission at the project\'s default cap',
|
|
72
72
|
'/stop [project] — stop the planner reply in flight',
|
package/src/notify/telegram.ts
CHANGED
|
@@ -94,7 +94,7 @@ export function telegramTransport(token: string, chatId: string, apiBase = TELEG
|
|
|
94
94
|
export const BOT_COMMANDS: Array<{ command: string; description: string }> = [
|
|
95
95
|
{ command: 'projects', description: 'The fleet, with what is running' },
|
|
96
96
|
{ command: 'status', description: 'Runs in flight, spend, what needs you' },
|
|
97
|
-
{ command: 'new', description: 'Create a project: /new <name>' },
|
|
97
|
+
{ command: 'new', description: 'Create a project: /new <name>, or clone one: /new <git url>' },
|
|
98
98
|
{ command: 'plan', description: 'Talk to a planner: /plan <project> <what you want>' },
|
|
99
99
|
{ command: 'run', description: 'Skip the talk: /run <project> <brief>' },
|
|
100
100
|
{ command: 'stop', description: 'Stop the planner reply in flight' },
|
package/src/notify.ts
CHANGED
|
@@ -241,6 +241,11 @@ export function shape(env: Envelope, ctx: NotifyContext): Shaped | null {
|
|
|
241
241
|
case 'mission_incomplete':
|
|
242
242
|
return { key: `incomplete:${env.runId}`, gate: 'done',
|
|
243
243
|
text: `${head('Not done')}${runLine}\n${esc(clip(d.text))}${foot}` };
|
|
244
|
+
case 'pull_request': {
|
|
245
|
+
if (d.error || !d.url) return null;
|
|
246
|
+
return { key: `pr:${env.runId}`, gate: 'done',
|
|
247
|
+
text: `${head(d.method === 'gh' ? 'Pull request opened' : 'Branch pushed')}${runLine}\n<a href="${esc(String(d.url))}">${esc(String(d.url))}</a>` };
|
|
248
|
+
}
|
|
244
249
|
case 'service_exposed': {
|
|
245
250
|
// Informational, and worth a tap: the crew put something on the air.
|
|
246
251
|
const url = String(d.url ?? '');
|
package/src/orchestrator.test.ts
CHANGED
|
@@ -9,6 +9,7 @@ import {
|
|
|
9
9
|
stalledWorkerReport, workerStatusBlock,
|
|
10
10
|
watchRepeats, watchSilence, REPEAT_EXEMPT, observeToolUse,
|
|
11
11
|
DEFAULT_ASK_TIMEOUT_MS, armAskTimeout, unattendedAnswer, unattendedDenyMessage,
|
|
12
|
+
tokenCapLabel,
|
|
12
13
|
} from './orchestrator.js';
|
|
13
14
|
import { makePolicy, type PendingPermission } from './policy.js';
|
|
14
15
|
import type { AgentEnv } from './provider.js';
|
|
@@ -1145,3 +1146,106 @@ test('pendingAsks exposes an open question with its text and options, and forget
|
|
|
1145
1146
|
await p;
|
|
1146
1147
|
assert.deepEqual(run.pendingAsks(), []);
|
|
1147
1148
|
});
|
|
1149
|
+
|
|
1150
|
+
// ---------------------------------------------------------------------------
|
|
1151
|
+
// The token cap: the bound that holds where dollars cannot
|
|
1152
|
+
// ---------------------------------------------------------------------------
|
|
1153
|
+
|
|
1154
|
+
/**
|
|
1155
|
+
* A run whose usage is whatever the test says it is. Usage is read through
|
|
1156
|
+
* liveUsage(), so setting `meta.usage` is enough — the confirmed half of the
|
|
1157
|
+
* figure the cap actually consults.
|
|
1158
|
+
*/
|
|
1159
|
+
function cappedRun(over: Partial<RunMeta> = {}, usage?: Partial<{
|
|
1160
|
+
inputTokens: number; outputTokens: number; cacheReadTokens: number; cacheWriteTokens: number;
|
|
1161
|
+
}>) {
|
|
1162
|
+
const run = new MissionRun(
|
|
1163
|
+
meta({ costBasis: 'free', ...over }), () => {}, () => {}, noopAgentEnv,
|
|
1164
|
+
) as unknown as {
|
|
1165
|
+
meta: RunMeta; turns: number;
|
|
1166
|
+
capReached(): string | null;
|
|
1167
|
+
budgetLine(): string;
|
|
1168
|
+
};
|
|
1169
|
+
if (usage) {
|
|
1170
|
+
run.meta.usage = {
|
|
1171
|
+
inputTokens: 0, outputTokens: 0, cacheReadTokens: 0, cacheWriteTokens: 0, ...usage,
|
|
1172
|
+
};
|
|
1173
|
+
}
|
|
1174
|
+
return run;
|
|
1175
|
+
}
|
|
1176
|
+
|
|
1177
|
+
test('capReached says nothing while the token total is under the cap', () => {
|
|
1178
|
+
const run = cappedRun({}, {
|
|
1179
|
+
inputTokens: 1_000_000, outputTokens: 100_000,
|
|
1180
|
+
cacheReadTokens: 1_000_000, cacheWriteTokens: 500_000,
|
|
1181
|
+
});
|
|
1182
|
+
assert.equal(run.capReached(), null);
|
|
1183
|
+
});
|
|
1184
|
+
|
|
1185
|
+
test('capReached counts cache tokens too, and fires at the default 20M', () => {
|
|
1186
|
+
// Under on input and output alone; over once cache is counted — which is
|
|
1187
|
+
// exactly the run the cap exists for, since cache reads are most of the
|
|
1188
|
+
// traffic on a long director loop.
|
|
1189
|
+
const run = cappedRun({}, {
|
|
1190
|
+
inputTokens: 2_000_000, outputTokens: 400_000,
|
|
1191
|
+
cacheReadTokens: 16_000_000, cacheWriteTokens: 1_600_000,
|
|
1192
|
+
});
|
|
1193
|
+
const cap = run.capReached();
|
|
1194
|
+
assert.ok(cap?.startsWith('TOKEN CAP REACHED'), `got ${cap}`);
|
|
1195
|
+
assert.match(cap!, /20\.0M tokens/);
|
|
1196
|
+
});
|
|
1197
|
+
|
|
1198
|
+
test('capReached: the ledger\'s largest finished missions stay under the default', () => {
|
|
1199
|
+
// 15.9M tokens, done, $8.55 — a real Fable run from this machine. A cap
|
|
1200
|
+
// that would have ended it is a cap set wrong.
|
|
1201
|
+
const run = cappedRun({}, { inputTokens: 300_000, outputTokens: 200_000, cacheReadTokens: 15_000_000, cacheWriteTokens: 400_000 });
|
|
1202
|
+
assert.equal(run.capReached(), null);
|
|
1203
|
+
});
|
|
1204
|
+
|
|
1205
|
+
test('capReached honours an explicit maxTokens over the default', () => {
|
|
1206
|
+
const run = cappedRun({ maxTokens: 1_000_000 }, { inputTokens: 1_200_000 });
|
|
1207
|
+
assert.ok(run.capReached()?.startsWith('TOKEN CAP REACHED'));
|
|
1208
|
+
const roomy = cappedRun({ maxTokens: 40_000_000 }, { inputTokens: 25_000_000 });
|
|
1209
|
+
assert.equal(roomy.capReached(), null);
|
|
1210
|
+
});
|
|
1211
|
+
|
|
1212
|
+
test('the turn cap still wins when turns and tokens are both past their caps', () => {
|
|
1213
|
+
const run = cappedRun({ maxTurns: 10, maxTokens: 1_000 }, { inputTokens: 9_000_000 });
|
|
1214
|
+
run.turns = 50;
|
|
1215
|
+
assert.match(run.capReached()!, /^TURN CAP REACHED/);
|
|
1216
|
+
});
|
|
1217
|
+
|
|
1218
|
+
test('the token cap binds free and unpriced runs, and never a priced one', () => {
|
|
1219
|
+
for (const basis of ['free', 'unpriced'] as const) {
|
|
1220
|
+
const run = cappedRun({ costBasis: basis }, { inputTokens: 25_000_000 });
|
|
1221
|
+
assert.ok(run.capReached()?.startsWith('TOKEN CAP REACHED'), `basis ${basis}`);
|
|
1222
|
+
}
|
|
1223
|
+
// Priced: dollars are the cap. Under budget, tokens alone end nothing;
|
|
1224
|
+
// over budget, it is the budget that speaks.
|
|
1225
|
+
const priced = cappedRun({ costBasis: 'priced', budgetUsd: 5 }, { inputTokens: 25_000_000 });
|
|
1226
|
+
priced.meta.costUsd = 1;
|
|
1227
|
+
assert.equal(priced.capReached(), null);
|
|
1228
|
+
priced.meta.costUsd = 5;
|
|
1229
|
+
assert.match(priced.capReached()!, /^BUDGET CAP REACHED/);
|
|
1230
|
+
});
|
|
1231
|
+
|
|
1232
|
+
test('budgetLine tells an unmetered director its turn AND token bounds', () => {
|
|
1233
|
+
const free = cappedRun({ costBasis: 'free' }).budgetLine();
|
|
1234
|
+
assert.match(free, /150 director turns and 20M tokens/);
|
|
1235
|
+
const unpriced = cappedRun({ costBasis: 'unpriced' }).budgetLine();
|
|
1236
|
+
assert.match(unpriced, /150 director turns and 20M tokens/);
|
|
1237
|
+
// A custom cap is quoted as set, not as the default.
|
|
1238
|
+
assert.match(cappedRun({ maxTokens: 2_000_000 }).budgetLine(), /2M tokens/);
|
|
1239
|
+
// A priced run still speaks in dollars, and says nothing about tokens.
|
|
1240
|
+
const priced = cappedRun({ costBasis: 'priced' }).budgetLine();
|
|
1241
|
+
assert.match(priced, /^Budget: \$5\.00/);
|
|
1242
|
+
assert.doesNotMatch(priced, /tokens/);
|
|
1243
|
+
});
|
|
1244
|
+
|
|
1245
|
+
test('tokenCapLabel rounds to a figure a director can hold in mind', () => {
|
|
1246
|
+
assert.equal(tokenCapLabel(20_000_000), '20M tokens');
|
|
1247
|
+
assert.equal(tokenCapLabel(5_000_000), '5M tokens');
|
|
1248
|
+
assert.equal(tokenCapLabel(1_500_000), '1.5M tokens');
|
|
1249
|
+
assert.equal(tokenCapLabel(250_000), '250k tokens');
|
|
1250
|
+
assert.equal(tokenCapLabel(400), '400 tokens');
|
|
1251
|
+
});
|
package/src/orchestrator.ts
CHANGED
|
@@ -90,11 +90,13 @@ class MessageStream implements AsyncIterable<SDKUserMessage> {
|
|
|
90
90
|
}
|
|
91
91
|
}
|
|
92
92
|
import { WORK_DIR, makePolicy, type PendingPermission } from './policy.js';
|
|
93
|
+
import { PLAYWRIGHT_MCP_CLI } from './browser.js';
|
|
93
94
|
import type { AgentEnv } from './provider.js';
|
|
94
95
|
import { generateRunTitle } from './title.js';
|
|
95
96
|
import { combineBasis, costBasisOf, isPriced, type CostBasis } from './types.js';
|
|
96
97
|
import { priceUsage, type ModelPrice } from './prices.js';
|
|
97
98
|
import { captureBaseline } from './deck.js';
|
|
99
|
+
import { memorySection, readMemory, writeMemory } from './memory.js';
|
|
98
100
|
import type { RunMeta, TokenUsage, WorkerMeta, WorkerProgress } from './types.js';
|
|
99
101
|
|
|
100
102
|
/** A run's usage before its first `result` message. */
|
|
@@ -248,6 +250,32 @@ export async function ensureIgnoreLines(file: string, lines: readonly string[]):
|
|
|
248
250
|
// a metered one that dollars already stop.
|
|
249
251
|
const DEFAULT_MAX_TURNS = 150;
|
|
250
252
|
|
|
253
|
+
/**
|
|
254
|
+
* The same idea counted in tokens, for the same reason the turn cap exists.
|
|
255
|
+
*
|
|
256
|
+
* A turn cap bounds how many times the director speaks, not how much it says,
|
|
257
|
+
* and those come apart badly on a free or unpriced run: 150 turns over a large
|
|
258
|
+
* context is millions of tokens that nothing here was watching. It binds only
|
|
259
|
+
* where dollars cannot — a priced run already has a real cap, and this one
|
|
260
|
+
* must never end it first. Sized from the ledger, not from a hunch: cache
|
|
261
|
+
* reads are most of a director loop's traffic, and finished missions on this
|
|
262
|
+
* machine have run to 8–16M tokens total (a 5M figure was first proposed and
|
|
263
|
+
* would have cut several of them short). 20M is clear of every honest run
|
|
264
|
+
* seen so far and still well inside what a runaway reaches before the clock.
|
|
265
|
+
*/
|
|
266
|
+
export const DEFAULT_MAX_TOKENS = 20_000_000;
|
|
267
|
+
|
|
268
|
+
/**
|
|
269
|
+
* The token cap as a director should read it: "5M tokens", not "5000000".
|
|
270
|
+
* A budget is only useful if the agent it constrains can hold it in mind, and
|
|
271
|
+
* a raw seven-digit figure in a prompt is one more thing to misread.
|
|
272
|
+
*/
|
|
273
|
+
export function tokenCapLabel(n: number): string {
|
|
274
|
+
if (n >= 1e6) return `${Number((n / 1e6).toFixed(1))}M tokens`;
|
|
275
|
+
if (n >= 1000) return `${Number((n / 1000).toFixed(1))}k tokens`;
|
|
276
|
+
return `${n} tokens`;
|
|
277
|
+
}
|
|
278
|
+
|
|
251
279
|
/**
|
|
252
280
|
* How long a worker may say nothing at all before it is treated as stalled.
|
|
253
281
|
*
|
|
@@ -675,6 +703,12 @@ directing worker agents. Non-negotiable rules, in priority order:
|
|
|
675
703
|
failure is an escalation, never a self-repair.
|
|
676
704
|
7. When DONE WHEN is verified, update MISSION.md (all boxes ticked, final log
|
|
677
705
|
entry) and end with a short summary of what was built and how you verified it.
|
|
706
|
+
8. LEAVE NOTES FOR THE NEXT CREW. Before you finish, call mcp__foreman__remember
|
|
707
|
+
with the whole project memory as it should read now: how to run and test
|
|
708
|
+
the project, ports and paths that matter, conventions and the reasons
|
|
709
|
+
behind them, traps you fell into. Facts, one line each, a page at most.
|
|
710
|
+
Rewrite stale lines rather than appending; drop what no longer holds.
|
|
711
|
+
Never a secret, never anything outside this project.
|
|
678
712
|
`;
|
|
679
713
|
|
|
680
714
|
export const WORKER_CHARTER = `
|
|
@@ -697,9 +731,6 @@ When finished, end with a concise report of what you did and how you checked it.
|
|
|
697
731
|
`;
|
|
698
732
|
|
|
699
733
|
/** Foreman's bundled Playwright MCP server, resolved from this repo. */
|
|
700
|
-
const PLAYWRIGHT_MCP_CLI = fileURLToPath(
|
|
701
|
-
new URL('../node_modules/@playwright/mcp/cli.js', import.meta.url),
|
|
702
|
-
);
|
|
703
734
|
|
|
704
735
|
/** Claude subscription/quota exhaustion — an external pause, not a failure. */
|
|
705
736
|
const USAGE_LIMIT_RE = /out of usage credits|usage limit reached|upgrade to increase your usage/i;
|
|
@@ -756,6 +787,8 @@ export class MissionRun {
|
|
|
756
787
|
private readonly askTimers = new Map<string, { cancel(): void }>();
|
|
757
788
|
/** The director's streaming prompt; steering pushes into it. */
|
|
758
789
|
private directorInput?: MessageStream;
|
|
790
|
+
/** The project's notes as a worker-prompt section, read once per start; empty when there are none. */
|
|
791
|
+
private memoryForWorkers = '';
|
|
759
792
|
/** Director cost is cumulative per query; track the last figure for deltas. */
|
|
760
793
|
private directorCostSeen = 0;
|
|
761
794
|
private wasInterrupted = false;
|
|
@@ -834,6 +867,8 @@ export class MissionRun {
|
|
|
834
867
|
*/
|
|
835
868
|
private readonly host: {
|
|
836
869
|
exposeService?: (runId: string, port: number, label: string) => Promise<{ ok: true; url: string; path: string } | { ok: false; reason: string }>;
|
|
870
|
+
/** The browser channel detected at dispatch (src/browser.ts); Chrome when the host says nothing. */
|
|
871
|
+
browserChannel?: string;
|
|
837
872
|
} = {},
|
|
838
873
|
) {
|
|
839
874
|
this.meta = meta;
|
|
@@ -1108,9 +1143,12 @@ export class MissionRun {
|
|
|
1108
1143
|
'actually complete. Keep completed work; do not rewrite files that already ' +
|
|
1109
1144
|
'satisfy their milestone. Update the doc to match reality, then continue ' +
|
|
1110
1145
|
'the mission to DONE WHEN. ' +
|
|
1111
|
-
this.
|
|
1146
|
+
this.gitLine() +
|
|
1147
|
+
this.budgetNote() +
|
|
1148
|
+
memorySection((await readMemory(this.meta.folder)).text, 'director')
|
|
1112
1149
|
: `MISSION: ${this.meta.mission}\n\n${this.budgetLine()} ` +
|
|
1113
|
-
`Working directory: ${this.meta.folder}. Begin by writing .foreman/MISSION.md, then execute the plan
|
|
1150
|
+
`Working directory: ${this.meta.folder}. ${this.gitLine()}Begin by writing .foreman/MISSION.md, then execute the plan.` +
|
|
1151
|
+
memorySection((await readMemory(this.meta.folder)).text, 'director');
|
|
1114
1152
|
|
|
1115
1153
|
try {
|
|
1116
1154
|
// The mission doc directory ignores itself wholesale (`*`, which also
|
|
@@ -1135,6 +1173,7 @@ export class MissionRun {
|
|
|
1135
1173
|
void captureBaseline(this.meta.folder, this.meta.id).catch(() => {});
|
|
1136
1174
|
}
|
|
1137
1175
|
await mkdir(path.join(this.meta.folder, WORK_DIR), { recursive: true }).catch(() => {});
|
|
1176
|
+
this.memoryForWorkers = memorySection((await readMemory(this.meta.folder)).text, 'worker');
|
|
1138
1177
|
// Covers a .claude/ left by an earlier run; the one this run creates is
|
|
1139
1178
|
// handled again on the way out.
|
|
1140
1179
|
await this.ignoreLocalSettings();
|
|
@@ -1360,9 +1399,9 @@ export class MissionRun {
|
|
|
1360
1399
|
// Absolute path: the server runs with the mission folder as cwd,
|
|
1361
1400
|
// where npx cannot resolve Foreman's own dependency.
|
|
1362
1401
|
command: process.execPath,
|
|
1363
|
-
//
|
|
1364
|
-
//
|
|
1365
|
-
args: [PLAYWRIGHT_MCP_CLI, '--headless', '--isolated', '--browser', process.env.FOREMAN_BROWSER
|
|
1402
|
+
// The channel the server detected for this machine: FOREMAN_BROWSER,
|
|
1403
|
+
// else its Chrome, else Playwright's own Chromium (src/browser.ts).
|
|
1404
|
+
args: [PLAYWRIGHT_MCP_CLI, '--headless', '--isolated', '--browser', this.host.browserChannel ?? process.env.FOREMAN_BROWSER ?? 'chrome'],
|
|
1366
1405
|
},
|
|
1367
1406
|
};
|
|
1368
1407
|
}
|
|
@@ -1441,14 +1480,6 @@ export class MissionRun {
|
|
|
1441
1480
|
}
|
|
1442
1481
|
|
|
1443
1482
|
|
|
1444
|
-
/**
|
|
1445
|
-
* Fold in the SDK's own dollar figure for a role.
|
|
1446
|
-
*
|
|
1447
|
-
* Ignored outright where Foreman holds that role's real rates: the SDK
|
|
1448
|
-
* prices every response with Anthropic's table, so on a gateway role its
|
|
1449
|
-
* number is fiction — and adding fiction to a figure computed from the
|
|
1450
|
-
* endpoint's own published rates would corrupt the one honest total.
|
|
1451
|
-
*/
|
|
1452
1483
|
/**
|
|
1453
1484
|
* `costUsd` from its parts. The upstream's own figure, once it has given
|
|
1454
1485
|
* one, replaces the rated figure for gateway tokens rather than adding to
|
|
@@ -1475,12 +1506,6 @@ export class MissionRun {
|
|
|
1475
1506
|
this.enforceBudget();
|
|
1476
1507
|
}
|
|
1477
1508
|
|
|
1478
|
-
/**
|
|
1479
|
-
* Folds one result message's token usage into the run total and persists
|
|
1480
|
-
* it. Called alongside addCost() from the same two call sites (director
|
|
1481
|
-
* loop, runWorker) so usage and cost are always in step — the honest
|
|
1482
|
-
* counterpart to a dollar figure that is not honest on every provider.
|
|
1483
|
-
*/
|
|
1484
1509
|
/**
|
|
1485
1510
|
* The one event that carries a run's economics. Emitted whenever either half
|
|
1486
1511
|
* changes — dollars OR tokens — because through a gateway the SDK often
|
|
@@ -1621,6 +1646,15 @@ export class MissionRun {
|
|
|
1621
1646
|
}
|
|
1622
1647
|
}
|
|
1623
1648
|
|
|
1649
|
+
/** The mission's branch, when it has one: stay on it, and leave merging and pushing alone. */
|
|
1650
|
+
private gitLine(): string {
|
|
1651
|
+
const g = this.meta.git;
|
|
1652
|
+
if (!g) return '';
|
|
1653
|
+
return `This mission runs on git branch ${g.branch}, created for it from ${g.base}. Stay on it: do not switch branches, ` +
|
|
1654
|
+
'do not merge, do not push, do not rebase or reset. You may commit as you go; Foreman commits whatever ' +
|
|
1655
|
+
'is left uncommitted when the mission ends. ';
|
|
1656
|
+
}
|
|
1657
|
+
|
|
1624
1658
|
/**
|
|
1625
1659
|
* What the director is told about its budget, in the units that are true.
|
|
1626
1660
|
*
|
|
@@ -1633,18 +1667,19 @@ export class MissionRun {
|
|
|
1633
1667
|
*/
|
|
1634
1668
|
private budgetLine(): string {
|
|
1635
1669
|
const turns = this.meta.maxTurns ?? DEFAULT_MAX_TURNS;
|
|
1670
|
+
const tokens = tokenCapLabel(this.meta.maxTokens ?? DEFAULT_MAX_TOKENS);
|
|
1636
1671
|
switch (costBasisOf(this.meta)) {
|
|
1637
1672
|
case 'free':
|
|
1638
1673
|
return `This run costs nothing per token — it is served by hardware the ` +
|
|
1639
1674
|
`operator already owns — so there is no spend cap. It is bounded by ` +
|
|
1640
|
-
`${turns} director turns.`;
|
|
1675
|
+
`${turns} director turns and ${tokens}.`;
|
|
1641
1676
|
case 'unpriced':
|
|
1642
1677
|
// Deliberately still "no spend cap", and deliberately not silent about
|
|
1643
1678
|
// the spend. A director told only that money is being spent, with no
|
|
1644
1679
|
// figure and no cap, invents a limit and winds itself down early.
|
|
1645
1680
|
return `This run does draw on a paid account, but Foreman cannot price it, ` +
|
|
1646
1681
|
`so there is no dollar cap and no figure to reason about — do not ration ` +
|
|
1647
|
-
`yourself against one. It is bounded by ${turns} director turns.`;
|
|
1682
|
+
`yourself against one. It is bounded by ${turns} director turns and ${tokens}.`;
|
|
1648
1683
|
case 'priced':
|
|
1649
1684
|
return `Budget: $${this.meta.budgetUsd.toFixed(2)} total for this run.`;
|
|
1650
1685
|
}
|
|
@@ -1714,10 +1749,21 @@ export class MissionRun {
|
|
|
1714
1749
|
return `TIME CAP REACHED: ${Math.round(elapsed / 60)} minutes.`;
|
|
1715
1750
|
}
|
|
1716
1751
|
// Money only binds where the figure is real. Enforcing it through a
|
|
1717
|
-
// gateway ends working runs over spend that never happened.
|
|
1718
|
-
|
|
1719
|
-
|
|
1720
|
-
|
|
1752
|
+
// gateway ends working runs over spend that never happened. Where it is
|
|
1753
|
+
// real it is the cap, full stop: a priced run is never ended by tokens,
|
|
1754
|
+
// which would cut a mission the human funded to its dollar figure.
|
|
1755
|
+
if (isPriced(this.meta)) {
|
|
1756
|
+
if (this.meta.costUsd < this.meta.budgetUsd) return null;
|
|
1757
|
+
return `BUDGET CAP REACHED: $${this.meta.costUsd.toFixed(2)} of $${this.meta.budgetUsd.toFixed(2)}.`;
|
|
1758
|
+
}
|
|
1759
|
+
// Unpriced or free: tokens are the bound dollars cannot be. Live usage
|
|
1760
|
+
// rather than the persisted total, so a gateway's interim tokens count too.
|
|
1761
|
+
const u = this.liveUsage();
|
|
1762
|
+
const total = u.inputTokens + u.outputTokens + u.cacheReadTokens + u.cacheWriteTokens;
|
|
1763
|
+
if (total >= (this.meta.maxTokens ?? DEFAULT_MAX_TOKENS)) {
|
|
1764
|
+
return `TOKEN CAP REACHED: ${(total / 1e6).toFixed(1)}M tokens.`;
|
|
1765
|
+
}
|
|
1766
|
+
return null;
|
|
1721
1767
|
}
|
|
1722
1768
|
|
|
1723
1769
|
private overBudget(): string | null {
|
|
@@ -1768,22 +1814,6 @@ export class MissionRun {
|
|
|
1768
1814
|
);
|
|
1769
1815
|
}
|
|
1770
1816
|
|
|
1771
|
-
/**
|
|
1772
|
-
* Starts a worker and returns at once; the session runs in the background.
|
|
1773
|
-
*
|
|
1774
|
-
* The split between this and {@link runWorker} is the asynchronous design in
|
|
1775
|
-
* one place: everything the director can observe about a worker — the record
|
|
1776
|
-
* in the map, `worker_started`, the `done` promise wait_for_worker races, the
|
|
1777
|
-
* stored report and `worker_finished` at the end — is settled here, around a
|
|
1778
|
-
* runWorker that only drives the SDK session. The promise is kept on the
|
|
1779
|
-
* record rather than dropped, so a worker is never a floating promise, and
|
|
1780
|
-
* the outcome is written onto the record rather than returned once, because
|
|
1781
|
-
* the director now reads it back whenever it asks.
|
|
1782
|
-
*
|
|
1783
|
-
* Synchronous up to the point runWorker takes over: by the time this returns
|
|
1784
|
-
* the record exists and `worker_started` has been emitted, which is exactly
|
|
1785
|
-
* the guarantee spawn_worker's immediate reply relies on.
|
|
1786
|
-
*/
|
|
1787
1817
|
/**
|
|
1788
1818
|
* Item 6, decided as "ask": a worker on a gateway provider has stalled or
|
|
1789
1819
|
* looped. Rather than letting the director cope alone or retrying somewhere
|
|
@@ -1860,6 +1890,22 @@ export class MissionRun {
|
|
|
1860
1890
|
};
|
|
1861
1891
|
}
|
|
1862
1892
|
|
|
1893
|
+
/**
|
|
1894
|
+
* Starts a worker and returns at once; the session runs in the background.
|
|
1895
|
+
*
|
|
1896
|
+
* The split between this and {@link runWorker} is the asynchronous design in
|
|
1897
|
+
* one place: everything the director can observe about a worker — the record
|
|
1898
|
+
* in the map, `worker_started`, the `done` promise wait_for_worker races, the
|
|
1899
|
+
* stored report and `worker_finished` at the end — is settled here, around a
|
|
1900
|
+
* runWorker that only drives the SDK session. The promise is kept on the
|
|
1901
|
+
* record rather than dropped, so a worker is never a floating promise, and
|
|
1902
|
+
* the outcome is written onto the record rather than returned once, because
|
|
1903
|
+
* the director now reads it back whenever it asks.
|
|
1904
|
+
*
|
|
1905
|
+
* Synchronous up to the point runWorker takes over: by the time this returns
|
|
1906
|
+
* the record exists and `worker_started` has been emitted, which is exactly
|
|
1907
|
+
* the guarantee spawn_worker's immediate reply relies on.
|
|
1908
|
+
*/
|
|
1863
1909
|
private launchWorker(workerId: string, prompt: string, resumeSessionId?: string, overrides?: WorkerOverrides): WorkerRuntime {
|
|
1864
1910
|
const existing = this.workers.get(workerId);
|
|
1865
1911
|
const w: WorkerRuntime = existing ?? {
|
|
@@ -2137,7 +2183,9 @@ export class MissionRun {
|
|
|
2137
2183
|
const stop = this.overBudget();
|
|
2138
2184
|
if (stop) return stop;
|
|
2139
2185
|
const id = `worker-${++this.workerSeq}`;
|
|
2140
|
-
|
|
2186
|
+
// The worker gets the project's notes with its brief: what earlier crews
|
|
2187
|
+
// learned is exactly what a fresh session lacks.
|
|
2188
|
+
this.launchWorker(id, this.memoryForWorkers ? `${task}\n${this.memoryForWorkers}` : task);
|
|
2141
2189
|
return `[${id} started] status: running. It works in the background — use check_workers ` +
|
|
2142
2190
|
`to watch it, and wait_for_worker when you need its result.`;
|
|
2143
2191
|
}
|
|
@@ -2279,7 +2327,6 @@ export class MissionRun {
|
|
|
2279
2327
|
this.pendingQuestions.set(id, resolve);
|
|
2280
2328
|
this.armAsk(id, (afterMs) => {
|
|
2281
2329
|
if (!this.pendingQuestions.delete(id)) return;
|
|
2282
|
-
this.askMeta.delete(id);
|
|
2283
2330
|
this.askMeta.delete(id);
|
|
2284
2331
|
timedOut = true;
|
|
2285
2332
|
this.emit('question_timeout', { id, afterMs });
|
|
@@ -2317,9 +2364,24 @@ export class MissionRun {
|
|
|
2317
2364
|
'running while they may want to look, and put the URL in your report.' }] };
|
|
2318
2365
|
},
|
|
2319
2366
|
);
|
|
2367
|
+
const remember = tool(
|
|
2368
|
+
'remember',
|
|
2369
|
+
'Rewrite the project memory (.foreman/MEMORY.md): the whole page as it should read now, ' +
|
|
2370
|
+
'for the next crew. Facts about this project — how to run and test it, ports and paths, ' +
|
|
2371
|
+
'conventions and why, traps — one line each, a page at most. Replace stale lines; do not ' +
|
|
2372
|
+
'append forever. Never a secret; anything that looks like one is stripped.',
|
|
2373
|
+
{ text: z.string().max(20_000).describe('The complete memory file content, Markdown') },
|
|
2374
|
+
async ({ text: body }) => {
|
|
2375
|
+
const r = await writeMemory(this.meta.folder, body);
|
|
2376
|
+
this.emit('memory_updated', { bytes: r.bytes, redacted: r.redacted, trimmed: r.trimmed,
|
|
2377
|
+
text: `Project memory rewritten (${r.bytes} bytes${r.redacted ? `, ${r.redacted} secret-looking value${r.redacted === 1 ? '' : 's'} stripped` : ''}${r.trimmed ? ', trimmed to the cap' : ''}).` });
|
|
2378
|
+
return { content: [{ type: 'text' as const, text:
|
|
2379
|
+
`Memory written (${r.bytes} bytes).${r.redacted ? ` ${r.redacted} value(s) that looked like secrets were replaced with [redacted]; do not put credentials in memory.` : ''}${r.trimmed ? ' It was longer than the cap and has been cut at the end — prune it to a page.' : ''}` }] };
|
|
2380
|
+
},
|
|
2381
|
+
);
|
|
2320
2382
|
return createSdkMcpServer({
|
|
2321
2383
|
name: 'foreman',
|
|
2322
|
-
tools: [spawnWorker, checkWorkers, waitForWorker, messageWorker, askHuman, exposeService],
|
|
2384
|
+
tools: [spawnWorker, checkWorkers, waitForWorker, messageWorker, askHuman, exposeService, remember],
|
|
2323
2385
|
});
|
|
2324
2386
|
}
|
|
2325
2387
|
}
|
package/src/planner.ts
CHANGED
|
@@ -29,6 +29,7 @@ import {
|
|
|
29
29
|
} from '@anthropic-ai/claude-agent-sdk';
|
|
30
30
|
import type { AgentEnv } from './provider.js';
|
|
31
31
|
import type { MissionProposal } from './types.js';
|
|
32
|
+
import { memorySection, readMemory } from './memory.js';
|
|
32
33
|
import {
|
|
33
34
|
armAskTimeout, formatAnswers, normaliseQuestions,
|
|
34
35
|
type AskAnswers, type AskQuestion, type PendingAsk,
|
|
@@ -172,6 +173,8 @@ export interface PlannerModel {
|
|
|
172
173
|
providerLabel: string;
|
|
173
174
|
costBasis: 'priced' | 'free' | 'unpriced';
|
|
174
175
|
note?: string;
|
|
176
|
+
/** Its track record here, from the run ledger: "here: director 4/5 done (~$0.78, ~18 min)". */
|
|
177
|
+
record?: string;
|
|
175
178
|
}
|
|
176
179
|
|
|
177
180
|
/**
|
|
@@ -205,7 +208,7 @@ export function pickKnownModel(
|
|
|
205
208
|
export function modelsSection(models: PlannerModel[] | undefined): string {
|
|
206
209
|
if (!models?.length) return '';
|
|
207
210
|
const lines = models.map((m) =>
|
|
208
|
-
` - ${m.id} — ${m.providerLabel} · ${m.costBasis}${m.note ? ` · ${m.note}` : ''}`);
|
|
211
|
+
` - ${m.id} — ${m.providerLabel} · ${m.costBasis}${m.note ? ` · ${m.note}` : ''}${m.record ? ` · ${m.record}` : ''}`);
|
|
209
212
|
return `\nMODELS AVAILABLE ON THIS MACHINE (use these exact ids in propose_mission):\n${lines.join('\n')}\n`;
|
|
210
213
|
}
|
|
211
214
|
|
|
@@ -214,6 +217,8 @@ export interface PlanningTurn {
|
|
|
214
217
|
projectId: string;
|
|
215
218
|
/** What the machine can run, so recommendations are real ids, not guesses. */
|
|
216
219
|
models?: PlannerModel[];
|
|
220
|
+
/** What missions have cost here and across the fleet, from the ledger; '' when too little history. */
|
|
221
|
+
anchor?: string;
|
|
217
222
|
/** Session to resume; absent starts a fresh conversation. */
|
|
218
223
|
sessionId?: string;
|
|
219
224
|
folder: string;
|
|
@@ -407,7 +412,7 @@ export async function runPlanningTurn(turn: PlanningTurn): Promise<PlanningResul
|
|
|
407
412
|
// thinking about the proposal, not discover it by asking.
|
|
408
413
|
systemPrompt: {
|
|
409
414
|
type: 'preset', preset: 'claude_code',
|
|
410
|
-
append: PLANNER_CHARTER + modelsSection(turn.models),
|
|
415
|
+
append: PLANNER_CHARTER + modelsSection(turn.models) + (turn.anchor ?? '') + memorySection((await readMemory(turn.folder)).text, 'planner'),
|
|
411
416
|
},
|
|
412
417
|
mcpServers: { foreman: createSdkMcpServer({ name: 'foreman', tools: [proposeMission, askUser] }) },
|
|
413
418
|
canUseTool,
|