@a-t-h-i/bot-lobby 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/README.md +266 -35
  2. package/package.json +1 -1
  3. package/prompts/backend.md +46 -1
  4. package/prompts/designer.md +94 -15
  5. package/prompts/master.md +70 -1
  6. package/prompts/panel.md +39 -0
  7. package/prompts/planner.md +64 -0
  8. package/prompts/qa.md +35 -2
  9. package/prompts/quickfix.md +41 -0
  10. package/prompts/researcher.md +6 -0
  11. package/prompts/reviewer.md +16 -0
  12. package/prompts/scout.md +11 -2
  13. package/prompts/worker.md +35 -2
  14. package/src/desk/client-extension.ts +101 -0
  15. package/src/desk/desk.ts +249 -0
  16. package/src/desk/ipc.ts +178 -0
  17. package/src/desk/session.ts +214 -0
  18. package/src/execution/agent-runner.ts +165 -14
  19. package/src/execution/pi-runner.ts +376 -65
  20. package/src/index.ts +7 -0
  21. package/src/lobby/feed.ts +253 -0
  22. package/src/lobby/issues.ts +227 -0
  23. package/src/lobby/layout.ts +174 -0
  24. package/src/lobby/planner.ts +474 -0
  25. package/src/lobby/quickfix.ts +227 -0
  26. package/src/lobby/runtime.ts +440 -0
  27. package/src/lobby/tabs/home.ts +164 -0
  28. package/src/lobby/tabs/issues.ts +72 -0
  29. package/src/lobby/tabs/metrics.ts +162 -0
  30. package/src/lobby/tabs/plan.ts +160 -0
  31. package/src/lobby/tabs/quickfix.ts +101 -0
  32. package/src/lobby/tabs/tasks.ts +209 -0
  33. package/src/lobby/view.ts +855 -0
  34. package/src/master/master.ts +21 -17
  35. package/src/master/research.ts +8 -7
  36. package/src/pi/activity.ts +165 -0
  37. package/src/pi/commands.ts +46 -59
  38. package/src/pi/events.ts +5 -2
  39. package/src/pi/expressions.ts +43 -12
  40. package/src/pi/kaomoji.ts +227 -0
  41. package/src/pi/mascot-art.ts +5 -15
  42. package/src/pi/model-support.ts +135 -0
  43. package/src/pi/run-summary.ts +172 -0
  44. package/src/pi/settings-ui.ts +162 -60
  45. package/src/pi/start-task.ts +63 -0
  46. package/src/pi/tools.ts +51 -8
  47. package/src/pi/ui.ts +151 -49
  48. package/src/pi/zen-large.ts +41 -4
  49. package/src/pi/zen-metrics.ts +47 -7
  50. package/src/pi/zen.ts +29 -8
  51. package/src/roles/reviewer.ts +24 -4
  52. package/src/roles/worker.ts +7 -1
  53. package/src/schemas/configuration.ts +177 -18
  54. package/src/schemas/findings.ts +26 -0
  55. package/src/schemas/task.ts +27 -1
  56. package/src/state/backlog.ts +106 -0
  57. package/src/state/comments.ts +136 -0
  58. package/src/state/metrics.ts +305 -0
  59. package/src/state/project.ts +9 -0
  60. package/src/text.ts +9 -0
  61. package/src/workflow/workflow.ts +161 -12
@@ -0,0 +1,305 @@
1
+ /**
2
+ * Model performance records: one line per finished run of any agent — the
3
+ * Master's own turns, scouts, workers, the QA gate, researchers, quick fixes,
4
+ * planner turns and planning panel seats — so the lobby can show how long each model takes at each
5
+ * thinking level, how often it succeeds and what it costs. Append-only JSON
6
+ * lines per project; reads keep the newest `MAX_READ` records.
7
+ */
8
+ import { appendFileSync, existsSync, mkdirSync, readFileSync } from "node:fs";
9
+ import { dirname, join } from "node:path";
10
+ import type { AgentRun } from "../schemas/findings.ts";
11
+ import type { RunLogEntry, Task } from "../schemas/task.ts";
12
+ import { dataRoot } from "./project.ts";
13
+
14
+ export const METRIC_KINDS = ["master", "scout", "worker", "reviewer", "researcher", "quickfix", "planner", "panel"] as const;
15
+ export type MetricKind = (typeof METRIC_KINDS)[number];
16
+
17
+ export type MetricStatus = "success" | "failed" | "cancelled" | "timeout";
18
+
19
+ export interface MetricRecord {
20
+ id: string;
21
+ kind: MetricKind;
22
+ /** Display name of the agent: MASTER, DEV, DESIGN, QA, RESEARCH, QUICK FIX, ORACLE (planning). */
23
+ agent: string;
24
+ model?: string;
25
+ thinking?: string;
26
+ status: MetricStatus;
27
+ startedAt: string;
28
+ durationMs: number;
29
+ turns?: number;
30
+ tools?: number;
31
+ input?: number;
32
+ output?: number;
33
+ cost?: number;
34
+ taskId?: string;
35
+ stalled?: boolean;
36
+ }
37
+
38
+ /** Newest records kept in memory for aggregation. */
39
+ export const MAX_READ = 5000;
40
+
41
+ export function metricsPath(root: string, configDir: string): string {
42
+ return join(dataRoot(root, configDir), "metrics.jsonl");
43
+ }
44
+
45
+ export function appendMetrics(root: string, configDir: string, records: readonly MetricRecord[]): void {
46
+ if (records.length === 0) return;
47
+ const path = metricsPath(root, configDir);
48
+ try {
49
+ mkdirSync(dirname(path), { recursive: true });
50
+ appendFileSync(path, records.map((record) => `${JSON.stringify(record)}\n`).join(""), "utf8");
51
+ } catch {
52
+ // Metrics are best-effort; a read-only tree must never fail a workflow step.
53
+ }
54
+ }
55
+
56
+ function isRecord(value: unknown): value is MetricRecord {
57
+ const record = value as Partial<MetricRecord> | undefined;
58
+ return Boolean(record && typeof record.id === "string" && typeof record.kind === "string" && typeof record.durationMs === "number");
59
+ }
60
+
61
+ export function readMetrics(root: string, configDir: string, limit = MAX_READ): MetricRecord[] {
62
+ const path = metricsPath(root, configDir);
63
+ if (!existsSync(path)) return [];
64
+ let text: string;
65
+ try {
66
+ text = readFileSync(path, "utf8");
67
+ } catch {
68
+ return [];
69
+ }
70
+ const records: MetricRecord[] = [];
71
+ for (const line of text.split("\n").slice(-limit - 1)) {
72
+ if (!line.trim()) continue;
73
+ try {
74
+ const value = JSON.parse(line) as unknown;
75
+ if (isRecord(value)) records.push(value);
76
+ } catch {
77
+ // Torn lines are skipped.
78
+ }
79
+ }
80
+ return records.slice(-limit);
81
+ }
82
+
83
+ const AGENT_NAMES: Record<string, string> = { backend: "DEV", designer: "DESIGN", qa: "QA" };
84
+
85
+ function durationBetween(startedAt: string, finishedAt: string | undefined): number {
86
+ if (!finishedAt) return 0;
87
+ const ms = Date.parse(finishedAt) - Date.parse(startedAt);
88
+ return Number.isFinite(ms) && ms > 0 ? ms : 0;
89
+ }
90
+
91
+ function finalStatus(status: AgentRun["status"]): MetricStatus {
92
+ return status === "running" ? "failed" : status;
93
+ }
94
+
95
+ /** A finished subagent run as a metric record. */
96
+ export function metricFromRun(run: AgentRun): MetricRecord {
97
+ const agent = run.role === "researcher" ? "RESEARCH" : AGENT_NAMES[run.domain] ?? run.domain.toUpperCase();
98
+ const turns = run.turns ?? run.usage?.turns;
99
+ return {
100
+ id: run.runId,
101
+ kind: run.role,
102
+ agent,
103
+ ...(run.model ? { model: run.model } : {}),
104
+ ...(run.thinking ? { thinking: run.thinking } : {}),
105
+ status: finalStatus(run.status),
106
+ startedAt: run.startedAt,
107
+ durationMs: durationBetween(run.startedAt, run.finishedAt),
108
+ ...(turns ? { turns } : {}),
109
+ ...(run.tools ? { tools: run.tools } : {}),
110
+ ...(run.usage ? { input: run.usage.input, output: run.usage.output, cost: run.usage.cost } : {}),
111
+ taskId: run.taskId,
112
+ ...(run.stalled ? { stalled: true } : {}),
113
+ };
114
+ }
115
+
116
+ /** A persisted run-log entry as a metric record, so runs from before the metrics log still count. */
117
+ export function metricFromLog(entry: RunLogEntry, taskId: string): MetricRecord {
118
+ const agent = entry.role === "researcher" ? "RESEARCH" : AGENT_NAMES[entry.domain] ?? entry.domain.toUpperCase();
119
+ return {
120
+ id: entry.runId,
121
+ kind: entry.role,
122
+ agent,
123
+ ...(entry.model ? { model: entry.model } : {}),
124
+ ...(entry.thinking ? { thinking: entry.thinking } : {}),
125
+ status: finalStatus(entry.status),
126
+ startedAt: entry.startedAt,
127
+ durationMs: durationBetween(entry.startedAt, entry.finishedAt),
128
+ ...(entry.turns ? { turns: entry.turns } : {}),
129
+ ...(entry.tools ? { tools: entry.tools } : {}),
130
+ ...(entry.input !== undefined ? { input: entry.input, output: entry.output ?? 0, cost: entry.cost ?? 0 } : {}),
131
+ taskId,
132
+ ...(entry.stalled ? { stalled: true } : {}),
133
+ };
134
+ }
135
+
136
+ /** Records from the log plus every task's run log, deduplicated by id (the log wins). */
137
+ export function collectMetrics(logged: readonly MetricRecord[], tasks: readonly Task[]): MetricRecord[] {
138
+ const byId = new Map<string, MetricRecord>();
139
+ for (const task of tasks) for (const entry of task.runLog ?? []) byId.set(entry.runId, metricFromLog(entry, task.id));
140
+ for (const record of logged) byId.set(record.id, record);
141
+ return [...byId.values()].sort((a, b) => a.startedAt.localeCompare(b.startedAt));
142
+ }
143
+
144
+ export interface MetricGroup {
145
+ /** The model id without its provider, or "unknown" when a run never reported one. */
146
+ model: string;
147
+ thinking: string;
148
+ /** Agent kinds seen in the group, most frequent first. */
149
+ kinds: MetricKind[];
150
+ runs: number;
151
+ successes: number;
152
+ /** Runs that stalled or hit their time limit. */
153
+ timeouts: number;
154
+ avgMs: number;
155
+ p50Ms: number;
156
+ p90Ms: number;
157
+ avgTurns: number;
158
+ avgTools: number;
159
+ avgTokens: number;
160
+ avgCost: number;
161
+ totalCost: number;
162
+ /** Output tokens per second of wall time, a rough throughput signal. */
163
+ tokensPerSecond: number;
164
+ }
165
+
166
+ export type GroupBy = "model" | "model-kind";
167
+
168
+ function percentile(sorted: readonly number[], fraction: number): number {
169
+ if (sorted.length === 0) return 0;
170
+ const index = Math.min(sorted.length - 1, Math.max(0, Math.ceil(fraction * sorted.length) - 1));
171
+ return sorted[index]!;
172
+ }
173
+
174
+ function mean(values: readonly number[]): number {
175
+ return values.length === 0 ? 0 : values.reduce((sum, value) => sum + value, 0) / values.length;
176
+ }
177
+
178
+ /**
179
+ * The model a record ran on, without its provider: the Master reports
180
+ * `provider/id` while subagent streams report the bare id, and both are the
181
+ * same model.
182
+ */
183
+ export function modelName(model: string | undefined): string {
184
+ if (!model) return "unknown";
185
+ const slash = model.indexOf("/");
186
+ return slash >= 0 ? model.slice(slash + 1) : model;
187
+ }
188
+
189
+ function groupKey(record: MetricRecord, by: GroupBy): string {
190
+ const base = `${modelName(record.model)}\0${record.thinking ?? "—"}`;
191
+ return by === "model-kind" ? `${base}\0${record.kind}` : base;
192
+ }
193
+
194
+ /**
195
+ * Aggregate records per model and thinking level (optionally split by agent
196
+ * kind). Cancelled runs say nothing about a model's speed, so they count as
197
+ * runs but stay out of the timing and throughput figures.
198
+ */
199
+ export function aggregateMetrics(records: readonly MetricRecord[], by: GroupBy = "model"): MetricGroup[] {
200
+ const groups = new Map<string, MetricRecord[]>();
201
+ for (const record of records) {
202
+ const key = groupKey(record, by);
203
+ const list = groups.get(key);
204
+ if (list) list.push(record);
205
+ else groups.set(key, [record]);
206
+ }
207
+ return [...groups.values()].map((list) => {
208
+ const timed = list.filter((record) => record.status !== "cancelled" && record.durationMs > 0);
209
+ const durations = timed.map((record) => record.durationMs).sort((a, b) => a - b);
210
+ const kindCounts = new Map<MetricKind, number>();
211
+ for (const record of list) kindCounts.set(record.kind, (kindCounts.get(record.kind) ?? 0) + 1);
212
+ const totalCost = list.reduce((sum, record) => sum + (record.cost ?? 0), 0);
213
+ const outputTokens = timed.reduce((sum, record) => sum + (record.output ?? 0), 0);
214
+ const seconds = durations.reduce((sum, ms) => sum + ms, 0) / 1000;
215
+ return {
216
+ model: modelName(list[0]!.model),
217
+ thinking: list[0]!.thinking ?? "—",
218
+ kinds: [...kindCounts.entries()].sort((a, b) => b[1] - a[1]).map(([kind]) => kind),
219
+ runs: list.length,
220
+ successes: list.filter((record) => record.status === "success").length,
221
+ timeouts: list.filter((record) => record.status === "timeout" || record.stalled).length,
222
+ avgMs: mean(durations),
223
+ p50Ms: percentile(durations, 0.5),
224
+ p90Ms: percentile(durations, 0.9),
225
+ avgTurns: mean(list.filter((record) => record.turns).map((record) => record.turns!)),
226
+ avgTools: mean(list.filter((record) => record.tools !== undefined).map((record) => record.tools!)),
227
+ avgTokens: mean(list.filter((record) => record.input !== undefined).map((record) => (record.input ?? 0) + (record.output ?? 0))),
228
+ avgCost: list.length > 0 ? totalCost / list.length : 0,
229
+ totalCost,
230
+ tokensPerSecond: seconds > 0 ? outputTokens / seconds : 0,
231
+ };
232
+ });
233
+ }
234
+
235
+ export type SortKey = "runs" | "avg" | "success" | "cost";
236
+ export const SORT_KEYS: readonly SortKey[] = ["runs", "avg", "success", "cost"];
237
+
238
+ export function sortGroups(groups: readonly MetricGroup[], key: SortKey): MetricGroup[] {
239
+ const score = (group: MetricGroup): number => {
240
+ if (key === "avg") return group.avgMs;
241
+ if (key === "success") return group.runs === 0 ? 0 : group.successes / group.runs;
242
+ if (key === "cost") return group.totalCost;
243
+ return group.runs;
244
+ };
245
+ // Fastest first for time; largest first otherwise.
246
+ const direction = key === "avg" ? 1 : -1;
247
+ return [...groups].sort((a, b) => direction * (score(a) - score(b)) || a.model.localeCompare(b.model));
248
+ }
249
+
250
+ export interface TaskStats {
251
+ completed: number;
252
+ abandoned: number;
253
+ active: number;
254
+ /** Mean wall time from creation to completion over completed tasks. */
255
+ avgCompleteMs: number;
256
+ }
257
+
258
+ export function taskStats(tasks: readonly Task[]): TaskStats {
259
+ const completed = tasks.filter((task) => task.state === "completed");
260
+ const times = completed.map((task) => durationBetween(task.createdAt, task.updatedAt)).filter((ms) => ms > 0);
261
+ return {
262
+ completed: completed.length,
263
+ abandoned: tasks.filter((task) => task.state === "abandoned").length,
264
+ active: tasks.filter((task) => task.state !== "completed" && task.state !== "abandoned").length,
265
+ avgCompleteMs: mean(times),
266
+ };
267
+ }
268
+
269
+ export interface TaskTimeGroup {
270
+ model: string;
271
+ thinking: string;
272
+ tasks: number;
273
+ avgMs: number;
274
+ }
275
+
276
+ /**
277
+ * How long completed tasks took from request to done, grouped by the model and
278
+ * thinking level the oracle ran most of its turns on for that task. Tasks with
279
+ * no recorded Master turn are left out.
280
+ */
281
+ export function taskTimesByModel(tasks: readonly Task[], records: readonly MetricRecord[]): TaskTimeGroup[] {
282
+ const masterTurns = new Map<string, Map<string, number>>();
283
+ for (const record of records) {
284
+ if (record.kind !== "master" || !record.taskId) continue;
285
+ const key = `${modelName(record.model)}\0${record.thinking ?? "—"}`;
286
+ const counts = masterTurns.get(record.taskId) ?? new Map<string, number>();
287
+ counts.set(key, (counts.get(key) ?? 0) + 1);
288
+ masterTurns.set(record.taskId, counts);
289
+ }
290
+ const groups = new Map<string, number[]>();
291
+ for (const task of tasks) {
292
+ if (task.state !== "completed") continue;
293
+ const counts = masterTurns.get(task.id);
294
+ const ms = durationBetween(task.createdAt, task.updatedAt);
295
+ if (!counts || ms <= 0) continue;
296
+ const key = [...counts.entries()].sort((a, b) => b[1] - a[1])[0]![0];
297
+ groups.set(key, [...(groups.get(key) ?? []), ms]);
298
+ }
299
+ return [...groups.entries()]
300
+ .map(([key, times]) => {
301
+ const [model, thinking] = key.split("\0");
302
+ return { model: model!, thinking: thinking!, tasks: times.length, avgMs: mean(times) };
303
+ })
304
+ .sort((a, b) => b.tasks - a.tasks || a.avgMs - b.avgMs);
305
+ }
@@ -80,6 +80,15 @@ function configSourcePath(): string {
80
80
  return candidates.find((path) => existsSync(path)) ?? candidates[0]!;
81
81
  }
82
82
 
83
+ /** The config file as written, unresolved; undefined when missing or unreadable. */
84
+ export function readRawConfig(): unknown {
85
+ try {
86
+ return JSON.parse(readFileSync(configSourcePath(), "utf8"));
87
+ } catch {
88
+ return undefined;
89
+ }
90
+ }
91
+
83
92
  /** Load the global config; fall back to defaults on any read/parse error. */
84
93
  export function loadConfig(): BotLobbyConfig {
85
94
  try {
package/src/text.ts CHANGED
@@ -49,3 +49,12 @@ export function shortTitle(request: string, maxWords = 3): string {
49
49
  const content = words.filter((word) => !FILLER_WORDS.has(fillerKey(word)));
50
50
  return (content.length > 0 ? content : words).slice(0, maxWords).join(" ");
51
51
  }
52
+
53
+ /** Compact duration such as "45s", "3m" or "2m 05s"; a non-finite input reads "0s". */
54
+ export function shortDuration(ms: number): string {
55
+ const seconds = Number.isFinite(ms) ? Math.max(0, Math.round(ms / 1000)) : 0;
56
+ const minutes = Math.floor(seconds / 60);
57
+ if (minutes === 0) return `${seconds}s`;
58
+ const rest = seconds % 60;
59
+ return rest === 0 ? `${minutes}m` : `${minutes}m ${String(rest).padStart(2, "0")}s`;
60
+ }
@@ -1,7 +1,8 @@
1
1
  import { join } from "node:path";
2
- import type { BotLobbyConfig } from "../schemas/configuration.ts";
2
+ import type { BotLobbyConfig, ProfileResolver } from "../schemas/configuration.ts";
3
3
  import type { AgentRun, Pushback, ResearchResult, ReviewResult } from "../schemas/findings.ts";
4
4
  import {
5
+ MAX_RUN_LOG,
5
6
  MAX_WORKER_RECORDS,
6
7
  TASK_STATES,
7
8
  TERMINAL_STATES,
@@ -21,6 +22,10 @@ import { compactKnowledgeFile, overThreshold } from "../knowledge/compactor.ts";
21
22
  import { knowledgeDir, type KnowledgeAgent } from "../knowledge/paths.ts";
22
23
  import { writeScratchpad } from "../state/persistence.ts";
23
24
  import { spawnPiProcess, type ProcessRunner } from "../execution/pi-runner.ts";
25
+ import { mapConcurrent } from "../execution/agent-runner.ts";
26
+ import { parseWorkerResult } from "../roles/worker.ts";
27
+ import { autoNote, DESK_TOOLS, DeskSession } from "../desk/session.ts";
28
+ import type { Handover } from "../desk/desk.ts";
24
29
  import { readRepositoryDiff } from "../execution/git.ts";
25
30
  import {
26
31
  loadScoutResults,
@@ -39,7 +44,10 @@ import { detectSharedFiles, summarizeOutcomes } from "../master/synthesis.ts";
39
44
  import { truncate } from "../text.ts";
40
45
  import { assertNoPendingApprovals, pendingApprovals, requestApproval, resolveApproval } from "./approvals.ts";
41
46
  import { pingApproval } from "../pi/notify.ts";
47
+ import { describeRun, runLogEntry } from "../pi/run-summary.ts";
42
48
  import { nextStates } from "./transitions.ts";
49
+ import { appendMetrics, metricFromRun } from "../state/metrics.ts";
50
+ import { markCommentsAddressed, pendingComments, readPlanComments } from "../state/comments.ts";
43
51
 
44
52
  export const ORCHESTRATE_ACTIONS = [
45
53
  "clarify",
@@ -79,6 +87,8 @@ export interface OrchestrateParams {
79
87
  domain?: string;
80
88
  /** implement: the concrete instruction for the worker. */
81
89
  task?: string;
90
+ /** implement: several domains at once, run in parallel through the file desk. */
91
+ assignments?: Array<{ domain: string; task: string }>;
82
92
  /** knowledge: which persistent file the text belongs to. */
83
93
  kind?: KnowledgeKind;
84
94
  /** compact: the knowledge file being rewritten. */
@@ -97,6 +107,8 @@ export interface WorkflowDeps {
97
107
  /** The pi session driving this workflow; task ownership is skipped when absent (tests, headless use). */
98
108
  sessionId?: string;
99
109
  config: BotLobbyConfig;
110
+ /** Per-run model/thinking/time limit, clamped to each model; plain settings when absent. */
111
+ profile?: ProfileResolver;
100
112
  signal?: AbortSignal;
101
113
  onUpdate?: (run: AgentRun) => void;
102
114
  ask: (question: string) => Promise<string | undefined>;
@@ -110,6 +122,8 @@ export interface WorkflowResult {
110
122
  taskId: string;
111
123
  state: TaskState;
112
124
  message: string;
125
+ /** Subagent runs that finished during this action, oldest first. */
126
+ runs?: AgentRun[];
113
127
  }
114
128
 
115
129
  export type ApprovalChoice = "approve" | "amend" | "decline";
@@ -250,6 +264,7 @@ async function handleScout(task: Task, params: OrchestrateParams, deps: Workflow
250
264
  dataRoots: readDataRoots(deps.root, deps.configDir),
251
265
  taskDir: taskDirFor(deps.root, deps.configDir, task.id),
252
266
  config: deps.config,
267
+ profile: deps.profile,
253
268
  signal: deps.signal,
254
269
  onUpdate: deps.onUpdate,
255
270
  },
@@ -344,6 +359,7 @@ function researchRequestFor(
344
359
  domain,
345
360
  instruction,
346
361
  config: deps.config,
362
+ profile: deps.profile,
347
363
  cwd: deps.cwd,
348
364
  taskDir,
349
365
  signal: deps.signal,
@@ -373,6 +389,8 @@ async function handlePropose(task: Task, params: OrchestrateParams, deps: Workfl
373
389
  if (task.state === "clarifying") transition(task, "awaiting_approval");
374
390
  task.proposal = proposal;
375
391
  writeFileEnsured(join(taskDirFor(deps.root, deps.configDir, task.id), "proposal.md"), proposal);
392
+ // A new proposal answers any lobby comments left on the previous one.
393
+ if (!task.plan) addressComments(task, deps);
376
394
  for (const concern of params.concerns ?? []) recordDecision(task, `Concern: ${concern}`);
377
395
  transition(task, "awaiting_approval");
378
396
  if (!deps.config.workflow.requireApprovalForFeatures) return applyApprovalChoice(task, "approve");
@@ -384,15 +402,40 @@ async function handlePropose(task: Task, params: OrchestrateParams, deps: Workfl
384
402
  return applyApprovalChoice(task, "amend", await deps.ask("What should change?"));
385
403
  }
386
404
 
405
+ /** States in which `plan` replaces an approved plan instead of recording the first one. */
406
+ const AMEND_PLAN_STATES: readonly TaskState[] = ["implementing", "reviewing"];
407
+
408
+ /** Mark the user's outstanding lobby comments on this task as addressed; returns how many. */
409
+ function addressComments(task: Task, deps: WorkflowDeps): number {
410
+ const pending = pendingComments(readPlanComments(deps.root, deps.configDir, task.id));
411
+ if (pending.length > 0) markCommentsAddressed(deps.root, deps.configDir, task.id, pending.map((comment) => comment.id));
412
+ return pending.length;
413
+ }
414
+
415
+ function commentsNote(count: number): string {
416
+ return count > 0 ? ` ${count} lobby comment${count === 1 ? "" : "s"} marked addressed.` : "";
417
+ }
418
+
419
+ /**
420
+ * Record the internal plan while planning, or amend it later (for example when
421
+ * the user comments on it from the lobby): an amendment replaces the plan and
422
+ * keeps the task where it is, so finished steps stay done.
423
+ */
387
424
  function handlePlan(task: Task, params: OrchestrateParams, deps: WorkflowDeps): string {
388
- requireState(task, ["planning"]);
425
+ requireState(task, ["planning", ...AMEND_PLAN_STATES]);
389
426
  const plan = params.plan?.trim();
390
427
  if (!plan) throw new Error("plan requires the plan text");
391
428
  const missing = validatePlan(plan);
392
429
  if (missing.length > 0) throw new Error(`plan is missing: ${missing.join(", ")}`);
430
+ const amending = AMEND_PLAN_STATES.includes(task.state);
393
431
  task.plan = plan;
394
432
  writeFileEnsured(join(taskDirFor(deps.root, deps.configDir, task.id), "plan.md"), plan);
395
- return "Plan recorded. Next: call action=implement with domain and task for the first step.";
433
+ const addressed = addressComments(task, deps);
434
+ if (amending) {
435
+ recordDecision(task, `Plan amended${addressed > 0 ? ` for ${addressed} user comment${addressed === 1 ? "" : "s"}` : ""}.`);
436
+ return `Plan amended.${commentsNote(addressed)} Next: continue with action=implement for the next open step (or action=qa when the work is complete).`;
437
+ }
438
+ return `Plan recorded.${commentsNote(addressed)} Next: call action=implement with domain and task for the first step.`;
396
439
  }
397
440
 
398
441
  /** Record approvals a worker asked for; auto-approve when config allows it. */
@@ -500,6 +543,7 @@ function workerRequest(deps: WorkflowDeps, task: Task, domain: Domain, instructi
500
543
  cwd: deps.cwd,
501
544
  dataRoots: readDataRoots(deps.root, deps.configDir),
502
545
  config: deps.config,
546
+ profile: deps.profile,
503
547
  signal: deps.signal,
504
548
  onUpdate: deps.onUpdate,
505
549
  };
@@ -518,15 +562,32 @@ function recordWorkerRun(task: Task, run: AgentRun): void {
518
562
  task.workerRuns = [...(task.workerRuns ?? []), record].slice(-MAX_WORKER_RECORDS);
519
563
  }
520
564
 
521
- async function handleImplement(task: Task, params: OrchestrateParams, deps: WorkflowDeps): Promise<string> {
522
- requireState(task, ["planning", "implementing", "reviewing"]);
565
+ interface Assignment {
566
+ domain: Domain;
567
+ instruction: string;
568
+ }
569
+
570
+ /** The delegation as a list: `assignments` for a parallel batch, otherwise the single domain/task. */
571
+ function parseAssignments(params: OrchestrateParams): Assignment[] {
572
+ if (params.assignments && params.assignments.length > 0) {
573
+ const list = params.assignments.map((entry) => {
574
+ const instruction = entry.task?.trim();
575
+ if (!instruction) throw new Error("every assignment needs a task (what to implement)");
576
+ return { domain: parseDomain(entry.domain, "implement"), instruction };
577
+ });
578
+ const domains = list.map((entry) => entry.domain);
579
+ if (new Set(domains).size !== domains.length) throw new Error("parallel assignments need distinct domains (one worker per domain)");
580
+ return list;
581
+ }
523
582
  const domain = parseDomain(params.domain, "implement");
524
583
  const instruction = params.task?.trim();
525
584
  if (!instruction) throw new Error("implement requires task (what to implement)");
526
- assertNoPendingApprovals(task, domain);
527
- if (!task.domains.includes(domain)) task.domains.push(domain);
528
- if (task.state !== "implementing") transition(task, "implementing");
529
- const outcome = await runWorker(workerRequest(deps, task, domain, instruction), deps.runProcess ?? spawnPiProcess);
585
+ return [{ domain, instruction }];
586
+ }
587
+
588
+ /** Record one worker's outcome on the task and return its report for the Master. */
589
+ function absorbWorkerOutcome(task: Task, deps: WorkflowDeps, outcome: WorkerOutcome): string {
590
+ const domain = outcome.result.domain;
530
591
  recordWorkerRun(task, outcome.run);
531
592
  const approvals = recordWorkerApprovals(task, outcome, deps.config);
532
593
  const pushback = recordPushback(task, outcome);
@@ -535,6 +596,67 @@ async function handleImplement(task: Task, params: OrchestrateParams, deps: Work
535
596
  return workerReport(outcome, approvals, pushback);
536
597
  }
537
598
 
599
+ async function handleImplement(task: Task, params: OrchestrateParams, deps: WorkflowDeps): Promise<string> {
600
+ requireState(task, ["planning", "implementing", "reviewing"]);
601
+ const assignments = parseAssignments(params);
602
+ for (const { domain } of assignments) assertNoPendingApprovals(task, domain);
603
+ for (const { domain } of assignments) if (!task.domains.includes(domain)) task.domains.push(domain);
604
+ if (task.state !== "implementing") transition(task, "implementing");
605
+ if (assignments.length === 1) {
606
+ const { domain, instruction } = assignments[0]!;
607
+ const outcome = await runWorker(workerRequest(deps, task, domain, instruction), deps.runProcess ?? spawnPiProcess);
608
+ return absorbWorkerOutcome(task, deps, outcome);
609
+ }
610
+ return runParallelWorkers(task, deps, assignments);
611
+ }
612
+
613
+ /**
614
+ * Several domains at once. The workers share one file desk: each claims a file
615
+ * before editing it, queues for a busy one, and hands it over with a note; a
616
+ * worker that finishes hands over whatever it still holds automatically.
617
+ */
618
+ async function runParallelWorkers(task: Task, deps: WorkflowDeps, assignments: Assignment[]): Promise<string> {
619
+ const session = new DeskSession({ cwd: deps.cwd });
620
+ await session.open();
621
+ let outcomes: WorkerOutcome[];
622
+ let unenforced: Domain[];
623
+ try {
624
+ outcomes = await mapConcurrent(assignments, deps.config.workflow.maxParallelWorkers, ({ domain, instruction }) => {
625
+ const request = workerRequest(deps, task, domain, instruction);
626
+ request.agent = {
627
+ env: session.env(domain),
628
+ extraTools: DESK_TOOLS,
629
+ onStart: (handle) => session.attach(domain, handle),
630
+ onAttemptEnd: (run) => {
631
+ const result = parseWorkerResult(domain, run.output);
632
+ session.release(domain, (path, next) => autoNote(domain, path, next, result.filesChanged, result.completed));
633
+ },
634
+ };
635
+ return runWorker(request, deps.runProcess ?? spawnPiProcess);
636
+ });
637
+ unenforced = assignments.map((entry) => entry.domain).filter((domain) => !session.greetedBy(domain));
638
+ } finally {
639
+ await session.close();
640
+ }
641
+ const reports = outcomes.map((outcome) => absorbWorkerOutcome(task, deps, outcome));
642
+ return [
643
+ `Parallel batch: ${assignments.map((entry) => entry.domain).join(", ")}.`,
644
+ ...reports,
645
+ handoverLog(session.handovers()),
646
+ unenforced.length > 0
647
+ ? `File checkout was not enforced for ${unenforced.join(", ")} (bot-lobby did not load in those workers); inspect the diff for overlapping edits.`
648
+ : "",
649
+ ]
650
+ .filter((line) => line.length > 0)
651
+ .join("\n\n");
652
+ }
653
+
654
+ function handoverLog(handovers: readonly Handover[]): string {
655
+ if (handovers.length === 0) return "";
656
+ const lines = handovers.map((entry) => `- ${entry.path}: ${entry.from} → ${entry.to}${entry.auto ? " (on finish)" : ""} — ${truncate(entry.note, 200)}`);
657
+ return `File handovers:\n${lines.join("\n")}`;
658
+ }
659
+
538
660
  function handleResolveApproval(task: Task, params: OrchestrateParams): string {
539
661
  const id = params.approvalId?.trim();
540
662
  const decision = params.decision;
@@ -597,6 +719,7 @@ function qaRequest(deps: WorkflowDeps, task: Task, diff: string, instruction?: s
597
719
  cwd: deps.cwd,
598
720
  dataRoots: readDataRoots(deps.root, deps.configDir),
599
721
  config: deps.config,
722
+ profile: deps.profile,
600
723
  signal: deps.signal,
601
724
  onUpdate: deps.onUpdate,
602
725
  };
@@ -784,12 +907,38 @@ export async function runWorkflowAction(params: OrchestrateParams, deps: Workflo
784
907
  if (!handler) {
785
908
  return { ok: false, taskId: task.id, state: task.state, message: `Unknown action "${params.action}".` };
786
909
  }
910
+ const finished = new Map<string, AgentRun>();
911
+ const tracked: WorkflowDeps = {
912
+ ...deps,
913
+ onUpdate: (run) => {
914
+ if (run.status !== "running") finished.set(run.runId, run);
915
+ deps.onUpdate?.(run);
916
+ },
917
+ };
787
918
  try {
788
- const message = await handler(task, params, deps);
919
+ const message = await handler(task, params, tracked);
920
+ const runs = recordRunLog(task, finished, deps);
789
921
  saveTask(deps.root, deps.configDir, task);
790
- return { ok: true, taskId: task.id, state: task.state, message };
922
+ return { ok: true, taskId: task.id, state: task.state, message: `${message}${runsFooter(runs)}`, runs };
791
923
  } catch (error) {
924
+ const runs = recordRunLog(task, finished, deps);
792
925
  saveTask(deps.root, deps.configDir, task);
793
- return { ok: false, taskId: task.id, state: task.state, message: `Rejected: ${(error as Error).message}` };
926
+ return { ok: false, taskId: task.id, state: task.state, message: `Rejected: ${(error as Error).message}`, runs };
794
927
  }
795
928
  }
929
+
930
+ /** Append this action's finished runs to the task's bounded run log and the project's metrics log. */
931
+ function recordRunLog(task: Task, finished: ReadonlyMap<string, AgentRun>, deps: WorkflowDeps): AgentRun[] {
932
+ const runs = [...finished.values()];
933
+ if (runs.length > 0) {
934
+ task.runLog = [...(task.runLog ?? []), ...runs.map(runLogEntry)].slice(-MAX_RUN_LOG);
935
+ appendMetrics(deps.root, deps.configDir, runs.map(metricFromRun));
936
+ }
937
+ return runs;
938
+ }
939
+
940
+ /** One line per run so the Master sees timing, model and any partial-report flag. */
941
+ function runsFooter(runs: readonly AgentRun[]): string {
942
+ if (runs.length === 0) return "";
943
+ return `\n\nRuns:\n${runs.map((run) => `- ${describeRun(run)}`).join("\n")}`;
944
+ }