@a-t-h-i/bot-lobby 0.3.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +266 -35
- package/package.json +1 -1
- package/prompts/backend.md +46 -1
- package/prompts/designer.md +94 -15
- package/prompts/master.md +70 -1
- package/prompts/panel.md +39 -0
- package/prompts/planner.md +64 -0
- package/prompts/qa.md +35 -2
- package/prompts/quickfix.md +41 -0
- package/prompts/researcher.md +6 -0
- package/prompts/reviewer.md +16 -0
- package/prompts/scout.md +11 -2
- package/prompts/worker.md +35 -2
- package/src/desk/client-extension.ts +101 -0
- package/src/desk/desk.ts +249 -0
- package/src/desk/ipc.ts +178 -0
- package/src/desk/session.ts +214 -0
- package/src/execution/agent-runner.ts +165 -14
- package/src/execution/pi-runner.ts +376 -65
- package/src/index.ts +7 -0
- package/src/lobby/feed.ts +253 -0
- package/src/lobby/issues.ts +227 -0
- package/src/lobby/layout.ts +174 -0
- package/src/lobby/planner.ts +474 -0
- package/src/lobby/quickfix.ts +227 -0
- package/src/lobby/runtime.ts +440 -0
- package/src/lobby/tabs/home.ts +164 -0
- package/src/lobby/tabs/issues.ts +72 -0
- package/src/lobby/tabs/metrics.ts +162 -0
- package/src/lobby/tabs/plan.ts +160 -0
- package/src/lobby/tabs/quickfix.ts +101 -0
- package/src/lobby/tabs/tasks.ts +209 -0
- package/src/lobby/view.ts +855 -0
- package/src/master/master.ts +21 -17
- package/src/master/research.ts +8 -7
- package/src/pi/activity.ts +165 -0
- package/src/pi/commands.ts +46 -59
- package/src/pi/events.ts +5 -2
- package/src/pi/expressions.ts +43 -12
- package/src/pi/kaomoji.ts +227 -0
- package/src/pi/mascot-art.ts +5 -15
- package/src/pi/model-support.ts +135 -0
- package/src/pi/run-summary.ts +172 -0
- package/src/pi/settings-ui.ts +162 -60
- package/src/pi/start-task.ts +63 -0
- package/src/pi/tools.ts +51 -8
- package/src/pi/ui.ts +151 -49
- package/src/pi/zen-large.ts +41 -4
- package/src/pi/zen-metrics.ts +47 -7
- package/src/pi/zen.ts +29 -8
- package/src/roles/reviewer.ts +24 -4
- package/src/roles/worker.ts +7 -1
- package/src/schemas/configuration.ts +177 -18
- package/src/schemas/findings.ts +26 -0
- package/src/schemas/task.ts +27 -1
- package/src/state/backlog.ts +106 -0
- package/src/state/comments.ts +136 -0
- package/src/state/metrics.ts +305 -0
- package/src/state/project.ts +9 -0
- package/src/text.ts +9 -0
- package/src/workflow/workflow.ts +161 -12
|
@@ -0,0 +1,305 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Model performance records: one line per finished run of any agent — the
|
|
3
|
+
* Master's own turns, scouts, workers, the QA gate, researchers, quick fixes,
|
|
4
|
+
* planner turns and planning panel seats — so the lobby can show how long each model takes at each
|
|
5
|
+
* thinking level, how often it succeeds and what it costs. Append-only JSON
|
|
6
|
+
* lines per project; reads keep the newest `MAX_READ` records.
|
|
7
|
+
*/
|
|
8
|
+
import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
|
|
9
|
+
import { dirname, join } from "node:path";
|
|
10
|
+
import type { AgentRun } from "../schemas/findings.ts";
|
|
11
|
+
import type { RunLogEntry, Task } from "../schemas/task.ts";
|
|
12
|
+
import { dataRoot } from "./project.ts";
|
|
13
|
+
|
|
14
|
+
export const METRIC_KINDS = ["master", "scout", "worker", "reviewer", "researcher", "quickfix", "planner", "panel"] as const;
|
|
15
|
+
export type MetricKind = (typeof METRIC_KINDS)[number];
|
|
16
|
+
|
|
17
|
+
export type MetricStatus = "success" | "failed" | "cancelled" | "timeout";
|
|
18
|
+
|
|
19
|
+
export interface MetricRecord {
|
|
20
|
+
id: string;
|
|
21
|
+
kind: MetricKind;
|
|
22
|
+
/** Display name of the agent: MASTER, DEV, DESIGN, QA, RESEARCH, QUICK FIX, ORACLE (planning). */
|
|
23
|
+
agent: string;
|
|
24
|
+
model?: string;
|
|
25
|
+
thinking?: string;
|
|
26
|
+
status: MetricStatus;
|
|
27
|
+
startedAt: string;
|
|
28
|
+
durationMs: number;
|
|
29
|
+
turns?: number;
|
|
30
|
+
tools?: number;
|
|
31
|
+
input?: number;
|
|
32
|
+
output?: number;
|
|
33
|
+
cost?: number;
|
|
34
|
+
taskId?: string;
|
|
35
|
+
stalled?: boolean;
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
/** Newest records kept in memory for aggregation. */
|
|
39
|
+
export const MAX_READ = 5000;
|
|
40
|
+
|
|
41
|
+
export function metricsPath(root: string, configDir: string): string {
|
|
42
|
+
return join(dataRoot(root, configDir), "metrics.jsonl");
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export function appendMetrics(root: string, configDir: string, records: readonly MetricRecord[]): void {
|
|
46
|
+
if (records.length === 0) return;
|
|
47
|
+
const path = metricsPath(root, configDir);
|
|
48
|
+
try {
|
|
49
|
+
mkdirSync(dirname(path), { recursive: true });
|
|
50
|
+
appendFileSync(path, records.map((record) => `${JSON.stringify(record)}\n`).join(""), "utf8");
|
|
51
|
+
} catch {
|
|
52
|
+
// Metrics are best-effort; a read-only tree must never fail a workflow step.
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
function isRecord(value: unknown): value is MetricRecord {
|
|
57
|
+
const record = value as Partial<MetricRecord> | undefined;
|
|
58
|
+
return Boolean(record && typeof record.id === "string" && typeof record.kind === "string" && typeof record.durationMs === "number");
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
export function readMetrics(root: string, configDir: string, limit = MAX_READ): MetricRecord[] {
|
|
62
|
+
const path = metricsPath(root, configDir);
|
|
63
|
+
if (!existsSync(path)) return [];
|
|
64
|
+
let text: string;
|
|
65
|
+
try {
|
|
66
|
+
text = readFileSync(path, "utf8");
|
|
67
|
+
} catch {
|
|
68
|
+
return [];
|
|
69
|
+
}
|
|
70
|
+
const records: MetricRecord[] = [];
|
|
71
|
+
for (const line of text.split("\n").slice(-limit - 1)) {
|
|
72
|
+
if (!line.trim()) continue;
|
|
73
|
+
try {
|
|
74
|
+
const value = JSON.parse(line) as unknown;
|
|
75
|
+
if (isRecord(value)) records.push(value);
|
|
76
|
+
} catch {
|
|
77
|
+
// Torn lines are skipped.
|
|
78
|
+
}
|
|
79
|
+
}
|
|
80
|
+
return records.slice(-limit);
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
const AGENT_NAMES: Record<string, string> = { backend: "DEV", designer: "DESIGN", qa: "QA" };
|
|
84
|
+
|
|
85
|
+
function durationBetween(startedAt: string, finishedAt: string | undefined): number {
|
|
86
|
+
if (!finishedAt) return 0;
|
|
87
|
+
const ms = Date.parse(finishedAt) - Date.parse(startedAt);
|
|
88
|
+
return Number.isFinite(ms) && ms > 0 ? ms : 0;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
function finalStatus(status: AgentRun["status"]): MetricStatus {
|
|
92
|
+
return status === "running" ? "failed" : status;
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
/** A finished subagent run as a metric record. */
|
|
96
|
+
export function metricFromRun(run: AgentRun): MetricRecord {
|
|
97
|
+
const agent = run.role === "researcher" ? "RESEARCH" : AGENT_NAMES[run.domain] ?? run.domain.toUpperCase();
|
|
98
|
+
const turns = run.turns ?? run.usage?.turns;
|
|
99
|
+
return {
|
|
100
|
+
id: run.runId,
|
|
101
|
+
kind: run.role,
|
|
102
|
+
agent,
|
|
103
|
+
...(run.model ? { model: run.model } : {}),
|
|
104
|
+
...(run.thinking ? { thinking: run.thinking } : {}),
|
|
105
|
+
status: finalStatus(run.status),
|
|
106
|
+
startedAt: run.startedAt,
|
|
107
|
+
durationMs: durationBetween(run.startedAt, run.finishedAt),
|
|
108
|
+
...(turns ? { turns } : {}),
|
|
109
|
+
...(run.tools ? { tools: run.tools } : {}),
|
|
110
|
+
...(run.usage ? { input: run.usage.input, output: run.usage.output, cost: run.usage.cost } : {}),
|
|
111
|
+
taskId: run.taskId,
|
|
112
|
+
...(run.stalled ? { stalled: true } : {}),
|
|
113
|
+
};
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/** A persisted run-log entry as a metric record, so runs from before the metrics log still count. */
|
|
117
|
+
export function metricFromLog(entry: RunLogEntry, taskId: string): MetricRecord {
|
|
118
|
+
const agent = entry.role === "researcher" ? "RESEARCH" : AGENT_NAMES[entry.domain] ?? entry.domain.toUpperCase();
|
|
119
|
+
return {
|
|
120
|
+
id: entry.runId,
|
|
121
|
+
kind: entry.role,
|
|
122
|
+
agent,
|
|
123
|
+
...(entry.model ? { model: entry.model } : {}),
|
|
124
|
+
...(entry.thinking ? { thinking: entry.thinking } : {}),
|
|
125
|
+
status: finalStatus(entry.status),
|
|
126
|
+
startedAt: entry.startedAt,
|
|
127
|
+
durationMs: durationBetween(entry.startedAt, entry.finishedAt),
|
|
128
|
+
...(entry.turns ? { turns: entry.turns } : {}),
|
|
129
|
+
...(entry.tools ? { tools: entry.tools } : {}),
|
|
130
|
+
...(entry.input !== undefined ? { input: entry.input, output: entry.output ?? 0, cost: entry.cost ?? 0 } : {}),
|
|
131
|
+
taskId,
|
|
132
|
+
...(entry.stalled ? { stalled: true } : {}),
|
|
133
|
+
};
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
/** Records from the log plus every task's run log, deduplicated by id (the log wins). */
|
|
137
|
+
export function collectMetrics(logged: readonly MetricRecord[], tasks: readonly Task[]): MetricRecord[] {
|
|
138
|
+
const byId = new Map<string, MetricRecord>();
|
|
139
|
+
for (const task of tasks) for (const entry of task.runLog ?? []) byId.set(entry.runId, metricFromLog(entry, task.id));
|
|
140
|
+
for (const record of logged) byId.set(record.id, record);
|
|
141
|
+
return [...byId.values()].sort((a, b) => a.startedAt.localeCompare(b.startedAt));
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
export interface MetricGroup {
|
|
145
|
+
/** The model id without its provider, or "unknown" when a run never reported one. */
|
|
146
|
+
model: string;
|
|
147
|
+
thinking: string;
|
|
148
|
+
/** Agent kinds seen in the group, most frequent first. */
|
|
149
|
+
kinds: MetricKind[];
|
|
150
|
+
runs: number;
|
|
151
|
+
successes: number;
|
|
152
|
+
/** Runs that stalled or hit their time limit. */
|
|
153
|
+
timeouts: number;
|
|
154
|
+
avgMs: number;
|
|
155
|
+
p50Ms: number;
|
|
156
|
+
p90Ms: number;
|
|
157
|
+
avgTurns: number;
|
|
158
|
+
avgTools: number;
|
|
159
|
+
avgTokens: number;
|
|
160
|
+
avgCost: number;
|
|
161
|
+
totalCost: number;
|
|
162
|
+
/** Output tokens per second of wall time, a rough throughput signal. */
|
|
163
|
+
tokensPerSecond: number;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
export type GroupBy = "model" | "model-kind";
|
|
167
|
+
|
|
168
|
+
function percentile(sorted: readonly number[], fraction: number): number {
|
|
169
|
+
if (sorted.length === 0) return 0;
|
|
170
|
+
const index = Math.min(sorted.length - 1, Math.max(0, Math.ceil(fraction * sorted.length) - 1));
|
|
171
|
+
return sorted[index]!;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
function mean(values: readonly number[]): number {
|
|
175
|
+
return values.length === 0 ? 0 : values.reduce((sum, value) => sum + value, 0) / values.length;
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* The model a record ran on, without its provider: the Master reports
|
|
180
|
+
* `provider/id` while subagent streams report the bare id, and both are the
|
|
181
|
+
* same model.
|
|
182
|
+
*/
|
|
183
|
+
export function modelName(model: string | undefined): string {
|
|
184
|
+
if (!model) return "unknown";
|
|
185
|
+
const slash = model.indexOf("/");
|
|
186
|
+
return slash >= 0 ? model.slice(slash + 1) : model;
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
function groupKey(record: MetricRecord, by: GroupBy): string {
|
|
190
|
+
const base = `${modelName(record.model)}\0${record.thinking ?? "—"}`;
|
|
191
|
+
return by === "model-kind" ? `${base}\0${record.kind}` : base;
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/**
|
|
195
|
+
* Aggregate records per model and thinking level (optionally split by agent
|
|
196
|
+
* kind). Cancelled runs say nothing about a model's speed, so they count as
|
|
197
|
+
* runs but stay out of the timing and throughput figures.
|
|
198
|
+
*/
|
|
199
|
+
export function aggregateMetrics(records: readonly MetricRecord[], by: GroupBy = "model"): MetricGroup[] {
|
|
200
|
+
const groups = new Map<string, MetricRecord[]>();
|
|
201
|
+
for (const record of records) {
|
|
202
|
+
const key = groupKey(record, by);
|
|
203
|
+
const list = groups.get(key);
|
|
204
|
+
if (list) list.push(record);
|
|
205
|
+
else groups.set(key, [record]);
|
|
206
|
+
}
|
|
207
|
+
return [...groups.values()].map((list) => {
|
|
208
|
+
const timed = list.filter((record) => record.status !== "cancelled" && record.durationMs > 0);
|
|
209
|
+
const durations = timed.map((record) => record.durationMs).sort((a, b) => a - b);
|
|
210
|
+
const kindCounts = new Map<MetricKind, number>();
|
|
211
|
+
for (const record of list) kindCounts.set(record.kind, (kindCounts.get(record.kind) ?? 0) + 1);
|
|
212
|
+
const totalCost = list.reduce((sum, record) => sum + (record.cost ?? 0), 0);
|
|
213
|
+
const outputTokens = timed.reduce((sum, record) => sum + (record.output ?? 0), 0);
|
|
214
|
+
const seconds = durations.reduce((sum, ms) => sum + ms, 0) / 1000;
|
|
215
|
+
return {
|
|
216
|
+
model: modelName(list[0]!.model),
|
|
217
|
+
thinking: list[0]!.thinking ?? "—",
|
|
218
|
+
kinds: [...kindCounts.entries()].sort((a, b) => b[1] - a[1]).map(([kind]) => kind),
|
|
219
|
+
runs: list.length,
|
|
220
|
+
successes: list.filter((record) => record.status === "success").length,
|
|
221
|
+
timeouts: list.filter((record) => record.status === "timeout" || record.stalled).length,
|
|
222
|
+
avgMs: mean(durations),
|
|
223
|
+
p50Ms: percentile(durations, 0.5),
|
|
224
|
+
p90Ms: percentile(durations, 0.9),
|
|
225
|
+
avgTurns: mean(list.filter((record) => record.turns).map((record) => record.turns!)),
|
|
226
|
+
avgTools: mean(list.filter((record) => record.tools !== undefined).map((record) => record.tools!)),
|
|
227
|
+
avgTokens: mean(list.filter((record) => record.input !== undefined).map((record) => (record.input ?? 0) + (record.output ?? 0))),
|
|
228
|
+
avgCost: list.length > 0 ? totalCost / list.length : 0,
|
|
229
|
+
totalCost,
|
|
230
|
+
tokensPerSecond: seconds > 0 ? outputTokens / seconds : 0,
|
|
231
|
+
};
|
|
232
|
+
});
|
|
233
|
+
}
|
|
234
|
+
|
|
235
|
+
export type SortKey = "runs" | "avg" | "success" | "cost";
|
|
236
|
+
export const SORT_KEYS: readonly SortKey[] = ["runs", "avg", "success", "cost"];
|
|
237
|
+
|
|
238
|
+
export function sortGroups(groups: readonly MetricGroup[], key: SortKey): MetricGroup[] {
|
|
239
|
+
const score = (group: MetricGroup): number => {
|
|
240
|
+
if (key === "avg") return group.avgMs;
|
|
241
|
+
if (key === "success") return group.runs === 0 ? 0 : group.successes / group.runs;
|
|
242
|
+
if (key === "cost") return group.totalCost;
|
|
243
|
+
return group.runs;
|
|
244
|
+
};
|
|
245
|
+
// Fastest first for time; largest first otherwise.
|
|
246
|
+
const direction = key === "avg" ? 1 : -1;
|
|
247
|
+
return [...groups].sort((a, b) => direction * (score(a) - score(b)) || a.model.localeCompare(b.model));
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
export interface TaskStats {
|
|
251
|
+
completed: number;
|
|
252
|
+
abandoned: number;
|
|
253
|
+
active: number;
|
|
254
|
+
/** Mean wall time from creation to completion over completed tasks. */
|
|
255
|
+
avgCompleteMs: number;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
export function taskStats(tasks: readonly Task[]): TaskStats {
|
|
259
|
+
const completed = tasks.filter((task) => task.state === "completed");
|
|
260
|
+
const times = completed.map((task) => durationBetween(task.createdAt, task.updatedAt)).filter((ms) => ms > 0);
|
|
261
|
+
return {
|
|
262
|
+
completed: completed.length,
|
|
263
|
+
abandoned: tasks.filter((task) => task.state === "abandoned").length,
|
|
264
|
+
active: tasks.filter((task) => task.state !== "completed" && task.state !== "abandoned").length,
|
|
265
|
+
avgCompleteMs: mean(times),
|
|
266
|
+
};
|
|
267
|
+
}
|
|
268
|
+
|
|
269
|
+
export interface TaskTimeGroup {
|
|
270
|
+
model: string;
|
|
271
|
+
thinking: string;
|
|
272
|
+
tasks: number;
|
|
273
|
+
avgMs: number;
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
/**
|
|
277
|
+
* How long completed tasks took from request to done, grouped by the model and
|
|
278
|
+
* thinking level the oracle ran most of its turns on for that task. Tasks with
|
|
279
|
+
* no recorded Master turn are left out.
|
|
280
|
+
*/
|
|
281
|
+
export function taskTimesByModel(tasks: readonly Task[], records: readonly MetricRecord[]): TaskTimeGroup[] {
|
|
282
|
+
const masterTurns = new Map<string, Map<string, number>>();
|
|
283
|
+
for (const record of records) {
|
|
284
|
+
if (record.kind !== "master" || !record.taskId) continue;
|
|
285
|
+
const key = `${modelName(record.model)}\0${record.thinking ?? "—"}`;
|
|
286
|
+
const counts = masterTurns.get(record.taskId) ?? new Map<string, number>();
|
|
287
|
+
counts.set(key, (counts.get(key) ?? 0) + 1);
|
|
288
|
+
masterTurns.set(record.taskId, counts);
|
|
289
|
+
}
|
|
290
|
+
const groups = new Map<string, number[]>();
|
|
291
|
+
for (const task of tasks) {
|
|
292
|
+
if (task.state !== "completed") continue;
|
|
293
|
+
const counts = masterTurns.get(task.id);
|
|
294
|
+
const ms = durationBetween(task.createdAt, task.updatedAt);
|
|
295
|
+
if (!counts || ms <= 0) continue;
|
|
296
|
+
const key = [...counts.entries()].sort((a, b) => b[1] - a[1])[0]![0];
|
|
297
|
+
groups.set(key, [...(groups.get(key) ?? []), ms]);
|
|
298
|
+
}
|
|
299
|
+
return [...groups.entries()]
|
|
300
|
+
.map(([key, times]) => {
|
|
301
|
+
const [model, thinking] = key.split("\0");
|
|
302
|
+
return { model: model!, thinking: thinking!, tasks: times.length, avgMs: mean(times) };
|
|
303
|
+
})
|
|
304
|
+
.sort((a, b) => b.tasks - a.tasks || a.avgMs - b.avgMs);
|
|
305
|
+
}
|
package/src/state/project.ts
CHANGED
|
@@ -80,6 +80,15 @@ function configSourcePath(): string {
|
|
|
80
80
|
return candidates.find((path) => existsSync(path)) ?? candidates[0]!;
|
|
81
81
|
}
|
|
82
82
|
|
|
83
|
+
/** The config file as written, unresolved; undefined when missing or unreadable. */
|
|
84
|
+
export function readRawConfig(): unknown {
|
|
85
|
+
try {
|
|
86
|
+
return JSON.parse(readFileSync(configSourcePath(), "utf8"));
|
|
87
|
+
} catch {
|
|
88
|
+
return undefined;
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
|
|
83
92
|
/** Load the global config; fall back to defaults on any read/parse error. */
|
|
84
93
|
export function loadConfig(): BotLobbyConfig {
|
|
85
94
|
try {
|
package/src/text.ts
CHANGED
|
@@ -49,3 +49,12 @@ export function shortTitle(request: string, maxWords = 3): string {
|
|
|
49
49
|
const content = words.filter((word) => !FILLER_WORDS.has(fillerKey(word)));
|
|
50
50
|
return (content.length > 0 ? content : words).slice(0, maxWords).join(" ");
|
|
51
51
|
}
|
|
52
|
+
|
|
53
|
+
/** Compact duration such as "45s", "3m" or "2m 05s"; a non-finite input reads "0s". */
|
|
54
|
+
export function shortDuration(ms: number): string {
|
|
55
|
+
const seconds = Number.isFinite(ms) ? Math.max(0, Math.round(ms / 1000)) : 0;
|
|
56
|
+
const minutes = Math.floor(seconds / 60);
|
|
57
|
+
if (minutes === 0) return `${seconds}s`;
|
|
58
|
+
const rest = seconds % 60;
|
|
59
|
+
return rest === 0 ? `${minutes}m` : `${minutes}m ${String(rest).padStart(2, "0")}s`;
|
|
60
|
+
}
|
package/src/workflow/workflow.ts
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
import { join } from "node:path";
|
|
2
|
-
import type { BotLobbyConfig } from "../schemas/configuration.ts";
|
|
2
|
+
import type { BotLobbyConfig, ProfileResolver } from "../schemas/configuration.ts";
|
|
3
3
|
import type { AgentRun, Pushback, ResearchResult, ReviewResult } from "../schemas/findings.ts";
|
|
4
4
|
import {
|
|
5
|
+
MAX_RUN_LOG,
|
|
5
6
|
MAX_WORKER_RECORDS,
|
|
6
7
|
TASK_STATES,
|
|
7
8
|
TERMINAL_STATES,
|
|
@@ -21,6 +22,10 @@ import { compactKnowledgeFile, overThreshold } from "../knowledge/compactor.ts";
|
|
|
21
22
|
import { knowledgeDir, type KnowledgeAgent } from "../knowledge/paths.ts";
|
|
22
23
|
import { writeScratchpad } from "../state/persistence.ts";
|
|
23
24
|
import { spawnPiProcess, type ProcessRunner } from "../execution/pi-runner.ts";
|
|
25
|
+
import { mapConcurrent } from "../execution/agent-runner.ts";
|
|
26
|
+
import { parseWorkerResult } from "../roles/worker.ts";
|
|
27
|
+
import { autoNote, DESK_TOOLS, DeskSession } from "../desk/session.ts";
|
|
28
|
+
import type { Handover } from "../desk/desk.ts";
|
|
24
29
|
import { readRepositoryDiff } from "../execution/git.ts";
|
|
25
30
|
import {
|
|
26
31
|
loadScoutResults,
|
|
@@ -39,7 +44,10 @@ import { detectSharedFiles, summarizeOutcomes } from "../master/synthesis.ts";
|
|
|
39
44
|
import { truncate } from "../text.ts";
|
|
40
45
|
import { assertNoPendingApprovals, pendingApprovals, requestApproval, resolveApproval } from "./approvals.ts";
|
|
41
46
|
import { pingApproval } from "../pi/notify.ts";
|
|
47
|
+
import { describeRun, runLogEntry } from "../pi/run-summary.ts";
|
|
42
48
|
import { nextStates } from "./transitions.ts";
|
|
49
|
+
import { appendMetrics, metricFromRun } from "../state/metrics.ts";
|
|
50
|
+
import { markCommentsAddressed, pendingComments, readPlanComments } from "../state/comments.ts";
|
|
43
51
|
|
|
44
52
|
export const ORCHESTRATE_ACTIONS = [
|
|
45
53
|
"clarify",
|
|
@@ -79,6 +87,8 @@ export interface OrchestrateParams {
|
|
|
79
87
|
domain?: string;
|
|
80
88
|
/** implement: the concrete instruction for the worker. */
|
|
81
89
|
task?: string;
|
|
90
|
+
/** implement: several domains at once, run in parallel through the file desk. */
|
|
91
|
+
assignments?: Array<{ domain: string; task: string }>;
|
|
82
92
|
/** knowledge: which persistent file the text belongs to. */
|
|
83
93
|
kind?: KnowledgeKind;
|
|
84
94
|
/** compact: the knowledge file being rewritten. */
|
|
@@ -97,6 +107,8 @@ export interface WorkflowDeps {
|
|
|
97
107
|
/** The pi session driving this workflow; task ownership is skipped when absent (tests, headless use). */
|
|
98
108
|
sessionId?: string;
|
|
99
109
|
config: BotLobbyConfig;
|
|
110
|
+
/** Per-run model/thinking/time limit, clamped to each model; plain settings when absent. */
|
|
111
|
+
profile?: ProfileResolver;
|
|
100
112
|
signal?: AbortSignal;
|
|
101
113
|
onUpdate?: (run: AgentRun) => void;
|
|
102
114
|
ask: (question: string) => Promise<string | undefined>;
|
|
@@ -110,6 +122,8 @@ export interface WorkflowResult {
|
|
|
110
122
|
taskId: string;
|
|
111
123
|
state: TaskState;
|
|
112
124
|
message: string;
|
|
125
|
+
/** Subagent runs that finished during this action, oldest first. */
|
|
126
|
+
runs?: AgentRun[];
|
|
113
127
|
}
|
|
114
128
|
|
|
115
129
|
export type ApprovalChoice = "approve" | "amend" | "decline";
|
|
@@ -250,6 +264,7 @@ async function handleScout(task: Task, params: OrchestrateParams, deps: Workflow
|
|
|
250
264
|
dataRoots: readDataRoots(deps.root, deps.configDir),
|
|
251
265
|
taskDir: taskDirFor(deps.root, deps.configDir, task.id),
|
|
252
266
|
config: deps.config,
|
|
267
|
+
profile: deps.profile,
|
|
253
268
|
signal: deps.signal,
|
|
254
269
|
onUpdate: deps.onUpdate,
|
|
255
270
|
},
|
|
@@ -344,6 +359,7 @@ function researchRequestFor(
|
|
|
344
359
|
domain,
|
|
345
360
|
instruction,
|
|
346
361
|
config: deps.config,
|
|
362
|
+
profile: deps.profile,
|
|
347
363
|
cwd: deps.cwd,
|
|
348
364
|
taskDir,
|
|
349
365
|
signal: deps.signal,
|
|
@@ -373,6 +389,8 @@ async function handlePropose(task: Task, params: OrchestrateParams, deps: Workfl
|
|
|
373
389
|
if (task.state === "clarifying") transition(task, "awaiting_approval");
|
|
374
390
|
task.proposal = proposal;
|
|
375
391
|
writeFileEnsured(join(taskDirFor(deps.root, deps.configDir, task.id), "proposal.md"), proposal);
|
|
392
|
+
// A new proposal answers any lobby comments left on the previous one.
|
|
393
|
+
if (!task.plan) addressComments(task, deps);
|
|
376
394
|
for (const concern of params.concerns ?? []) recordDecision(task, `Concern: ${concern}`);
|
|
377
395
|
transition(task, "awaiting_approval");
|
|
378
396
|
if (!deps.config.workflow.requireApprovalForFeatures) return applyApprovalChoice(task, "approve");
|
|
@@ -384,15 +402,40 @@ async function handlePropose(task: Task, params: OrchestrateParams, deps: Workfl
|
|
|
384
402
|
return applyApprovalChoice(task, "amend", await deps.ask("What should change?"));
|
|
385
403
|
}
|
|
386
404
|
|
|
405
|
+
/** States in which `plan` replaces an approved plan instead of recording the first one. */
|
|
406
|
+
const AMEND_PLAN_STATES: readonly TaskState[] = ["implementing", "reviewing"];
|
|
407
|
+
|
|
408
|
+
/** Mark the user's outstanding lobby comments on this task as addressed; returns how many. */
|
|
409
|
+
function addressComments(task: Task, deps: WorkflowDeps): number {
|
|
410
|
+
const pending = pendingComments(readPlanComments(deps.root, deps.configDir, task.id));
|
|
411
|
+
if (pending.length > 0) markCommentsAddressed(deps.root, deps.configDir, task.id, pending.map((comment) => comment.id));
|
|
412
|
+
return pending.length;
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
function commentsNote(count: number): string {
|
|
416
|
+
return count > 0 ? ` ${count} lobby comment${count === 1 ? "" : "s"} marked addressed.` : "";
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
/**
|
|
420
|
+
* Record the internal plan while planning, or amend it later (for example when
|
|
421
|
+
* the user comments on it from the lobby): an amendment replaces the plan and
|
|
422
|
+
* keeps the task where it is, so finished steps stay done.
|
|
423
|
+
*/
|
|
387
424
|
function handlePlan(task: Task, params: OrchestrateParams, deps: WorkflowDeps): string {
|
|
388
|
-
requireState(task, ["planning"]);
|
|
425
|
+
requireState(task, ["planning", ...AMEND_PLAN_STATES]);
|
|
389
426
|
const plan = params.plan?.trim();
|
|
390
427
|
if (!plan) throw new Error("plan requires the plan text");
|
|
391
428
|
const missing = validatePlan(plan);
|
|
392
429
|
if (missing.length > 0) throw new Error(`plan is missing: ${missing.join(", ")}`);
|
|
430
|
+
const amending = AMEND_PLAN_STATES.includes(task.state);
|
|
393
431
|
task.plan = plan;
|
|
394
432
|
writeFileEnsured(join(taskDirFor(deps.root, deps.configDir, task.id), "plan.md"), plan);
|
|
395
|
-
|
|
433
|
+
const addressed = addressComments(task, deps);
|
|
434
|
+
if (amending) {
|
|
435
|
+
recordDecision(task, `Plan amended${addressed > 0 ? ` for ${addressed} user comment${addressed === 1 ? "" : "s"}` : ""}.`);
|
|
436
|
+
return `Plan amended.${commentsNote(addressed)} Next: continue with action=implement for the next open step (or action=qa when the work is complete).`;
|
|
437
|
+
}
|
|
438
|
+
return `Plan recorded.${commentsNote(addressed)} Next: call action=implement with domain and task for the first step.`;
|
|
396
439
|
}
|
|
397
440
|
|
|
398
441
|
/** Record approvals a worker asked for; auto-approve when config allows it. */
|
|
@@ -500,6 +543,7 @@ function workerRequest(deps: WorkflowDeps, task: Task, domain: Domain, instructi
|
|
|
500
543
|
cwd: deps.cwd,
|
|
501
544
|
dataRoots: readDataRoots(deps.root, deps.configDir),
|
|
502
545
|
config: deps.config,
|
|
546
|
+
profile: deps.profile,
|
|
503
547
|
signal: deps.signal,
|
|
504
548
|
onUpdate: deps.onUpdate,
|
|
505
549
|
};
|
|
@@ -518,15 +562,32 @@ function recordWorkerRun(task: Task, run: AgentRun): void {
|
|
|
518
562
|
task.workerRuns = [...(task.workerRuns ?? []), record].slice(-MAX_WORKER_RECORDS);
|
|
519
563
|
}
|
|
520
564
|
|
|
521
|
-
|
|
522
|
-
|
|
565
|
+
interface Assignment {
|
|
566
|
+
domain: Domain;
|
|
567
|
+
instruction: string;
|
|
568
|
+
}
|
|
569
|
+
|
|
570
|
+
/** The delegation as a list: `assignments` for a parallel batch, otherwise the single domain/task. */
|
|
571
|
+
function parseAssignments(params: OrchestrateParams): Assignment[] {
|
|
572
|
+
if (params.assignments && params.assignments.length > 0) {
|
|
573
|
+
const list = params.assignments.map((entry) => {
|
|
574
|
+
const instruction = entry.task?.trim();
|
|
575
|
+
if (!instruction) throw new Error("every assignment needs a task (what to implement)");
|
|
576
|
+
return { domain: parseDomain(entry.domain, "implement"), instruction };
|
|
577
|
+
});
|
|
578
|
+
const domains = list.map((entry) => entry.domain);
|
|
579
|
+
if (new Set(domains).size !== domains.length) throw new Error("parallel assignments need distinct domains (one worker per domain)");
|
|
580
|
+
return list;
|
|
581
|
+
}
|
|
523
582
|
const domain = parseDomain(params.domain, "implement");
|
|
524
583
|
const instruction = params.task?.trim();
|
|
525
584
|
if (!instruction) throw new Error("implement requires task (what to implement)");
|
|
526
|
-
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
|
|
585
|
+
return [{ domain, instruction }];
|
|
586
|
+
}
|
|
587
|
+
|
|
588
|
+
/** Record one worker's outcome on the task and return its report for the Master. */
|
|
589
|
+
function absorbWorkerOutcome(task: Task, deps: WorkflowDeps, outcome: WorkerOutcome): string {
|
|
590
|
+
const domain = outcome.result.domain;
|
|
530
591
|
recordWorkerRun(task, outcome.run);
|
|
531
592
|
const approvals = recordWorkerApprovals(task, outcome, deps.config);
|
|
532
593
|
const pushback = recordPushback(task, outcome);
|
|
@@ -535,6 +596,67 @@ async function handleImplement(task: Task, params: OrchestrateParams, deps: Work
|
|
|
535
596
|
return workerReport(outcome, approvals, pushback);
|
|
536
597
|
}
|
|
537
598
|
|
|
599
|
+
async function handleImplement(task: Task, params: OrchestrateParams, deps: WorkflowDeps): Promise<string> {
|
|
600
|
+
requireState(task, ["planning", "implementing", "reviewing"]);
|
|
601
|
+
const assignments = parseAssignments(params);
|
|
602
|
+
for (const { domain } of assignments) assertNoPendingApprovals(task, domain);
|
|
603
|
+
for (const { domain } of assignments) if (!task.domains.includes(domain)) task.domains.push(domain);
|
|
604
|
+
if (task.state !== "implementing") transition(task, "implementing");
|
|
605
|
+
if (assignments.length === 1) {
|
|
606
|
+
const { domain, instruction } = assignments[0]!;
|
|
607
|
+
const outcome = await runWorker(workerRequest(deps, task, domain, instruction), deps.runProcess ?? spawnPiProcess);
|
|
608
|
+
return absorbWorkerOutcome(task, deps, outcome);
|
|
609
|
+
}
|
|
610
|
+
return runParallelWorkers(task, deps, assignments);
|
|
611
|
+
}
|
|
612
|
+
|
|
613
|
+
/**
|
|
614
|
+
* Several domains at once. The workers share one file desk: each claims a file
|
|
615
|
+
* before editing it, queues for a busy one, and hands it over with a note; a
|
|
616
|
+
* worker that finishes hands over whatever it still holds automatically.
|
|
617
|
+
*/
|
|
618
|
+
async function runParallelWorkers(task: Task, deps: WorkflowDeps, assignments: Assignment[]): Promise<string> {
|
|
619
|
+
const session = new DeskSession({ cwd: deps.cwd });
|
|
620
|
+
await session.open();
|
|
621
|
+
let outcomes: WorkerOutcome[];
|
|
622
|
+
let unenforced: Domain[];
|
|
623
|
+
try {
|
|
624
|
+
outcomes = await mapConcurrent(assignments, deps.config.workflow.maxParallelWorkers, ({ domain, instruction }) => {
|
|
625
|
+
const request = workerRequest(deps, task, domain, instruction);
|
|
626
|
+
request.agent = {
|
|
627
|
+
env: session.env(domain),
|
|
628
|
+
extraTools: DESK_TOOLS,
|
|
629
|
+
onStart: (handle) => session.attach(domain, handle),
|
|
630
|
+
onAttemptEnd: (run) => {
|
|
631
|
+
const result = parseWorkerResult(domain, run.output);
|
|
632
|
+
session.release(domain, (path, next) => autoNote(domain, path, next, result.filesChanged, result.completed));
|
|
633
|
+
},
|
|
634
|
+
};
|
|
635
|
+
return runWorker(request, deps.runProcess ?? spawnPiProcess);
|
|
636
|
+
});
|
|
637
|
+
unenforced = assignments.map((entry) => entry.domain).filter((domain) => !session.greetedBy(domain));
|
|
638
|
+
} finally {
|
|
639
|
+
await session.close();
|
|
640
|
+
}
|
|
641
|
+
const reports = outcomes.map((outcome) => absorbWorkerOutcome(task, deps, outcome));
|
|
642
|
+
return [
|
|
643
|
+
`Parallel batch: ${assignments.map((entry) => entry.domain).join(", ")}.`,
|
|
644
|
+
...reports,
|
|
645
|
+
handoverLog(session.handovers()),
|
|
646
|
+
unenforced.length > 0
|
|
647
|
+
? `File checkout was not enforced for ${unenforced.join(", ")} (bot-lobby did not load in those workers); inspect the diff for overlapping edits.`
|
|
648
|
+
: "",
|
|
649
|
+
]
|
|
650
|
+
.filter((line) => line.length > 0)
|
|
651
|
+
.join("\n\n");
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
function handoverLog(handovers: readonly Handover[]): string {
|
|
655
|
+
if (handovers.length === 0) return "";
|
|
656
|
+
const lines = handovers.map((entry) => `- ${entry.path}: ${entry.from} → ${entry.to}${entry.auto ? " (on finish)" : ""} — ${truncate(entry.note, 200)}`);
|
|
657
|
+
return `File handovers:\n${lines.join("\n")}`;
|
|
658
|
+
}
|
|
659
|
+
|
|
538
660
|
function handleResolveApproval(task: Task, params: OrchestrateParams): string {
|
|
539
661
|
const id = params.approvalId?.trim();
|
|
540
662
|
const decision = params.decision;
|
|
@@ -597,6 +719,7 @@ function qaRequest(deps: WorkflowDeps, task: Task, diff: string, instruction?: s
|
|
|
597
719
|
cwd: deps.cwd,
|
|
598
720
|
dataRoots: readDataRoots(deps.root, deps.configDir),
|
|
599
721
|
config: deps.config,
|
|
722
|
+
profile: deps.profile,
|
|
600
723
|
signal: deps.signal,
|
|
601
724
|
onUpdate: deps.onUpdate,
|
|
602
725
|
};
|
|
@@ -784,12 +907,38 @@ export async function runWorkflowAction(params: OrchestrateParams, deps: Workflo
|
|
|
784
907
|
if (!handler) {
|
|
785
908
|
return { ok: false, taskId: task.id, state: task.state, message: `Unknown action "${params.action}".` };
|
|
786
909
|
}
|
|
910
|
+
const finished = new Map<string, AgentRun>();
|
|
911
|
+
const tracked: WorkflowDeps = {
|
|
912
|
+
...deps,
|
|
913
|
+
onUpdate: (run) => {
|
|
914
|
+
if (run.status !== "running") finished.set(run.runId, run);
|
|
915
|
+
deps.onUpdate?.(run);
|
|
916
|
+
},
|
|
917
|
+
};
|
|
787
918
|
try {
|
|
788
|
-
const message = await handler(task, params,
|
|
919
|
+
const message = await handler(task, params, tracked);
|
|
920
|
+
const runs = recordRunLog(task, finished, deps);
|
|
789
921
|
saveTask(deps.root, deps.configDir, task);
|
|
790
|
-
return { ok: true, taskId: task.id, state: task.state, message };
|
|
922
|
+
return { ok: true, taskId: task.id, state: task.state, message: `${message}${runsFooter(runs)}`, runs };
|
|
791
923
|
} catch (error) {
|
|
924
|
+
const runs = recordRunLog(task, finished, deps);
|
|
792
925
|
saveTask(deps.root, deps.configDir, task);
|
|
793
|
-
return { ok: false, taskId: task.id, state: task.state, message: `Rejected: ${(error as Error).message}
|
|
926
|
+
return { ok: false, taskId: task.id, state: task.state, message: `Rejected: ${(error as Error).message}`, runs };
|
|
794
927
|
}
|
|
795
928
|
}
|
|
929
|
+
|
|
930
|
+
/** Append this action's finished runs to the task's bounded run log and the project's metrics log. */
|
|
931
|
+
function recordRunLog(task: Task, finished: ReadonlyMap<string, AgentRun>, deps: WorkflowDeps): AgentRun[] {
|
|
932
|
+
const runs = [...finished.values()];
|
|
933
|
+
if (runs.length > 0) {
|
|
934
|
+
task.runLog = [...(task.runLog ?? []), ...runs.map(runLogEntry)].slice(-MAX_RUN_LOG);
|
|
935
|
+
appendMetrics(deps.root, deps.configDir, runs.map(metricFromRun));
|
|
936
|
+
}
|
|
937
|
+
return runs;
|
|
938
|
+
}
|
|
939
|
+
|
|
940
|
+
/** One line per run so the Master sees timing, model and any partial-report flag. */
|
|
941
|
+
function runsFooter(runs: readonly AgentRun[]): string {
|
|
942
|
+
if (runs.length === 0) return "";
|
|
943
|
+
return `\n\nRuns:\n${runs.map((run) => `- ${describeRun(run)}`).join("\n")}`;
|
|
944
|
+
}
|