@a-t-h-i/bot-lobby 0.6.4 → 0.6.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +150 -12
  2. package/package.json +12 -3
  3. package/prompts/backend.md +3 -1
  4. package/prompts/designer.md +24 -1
  5. package/prompts/global.md +18 -0
  6. package/prompts/master.md +122 -20
  7. package/prompts/quickfix.md +16 -3
  8. package/prompts/worker.md +7 -2
  9. package/src/agents/backend.ts +2 -2
  10. package/src/agents/designer.ts +2 -2
  11. package/src/ask/dialog.ts +11 -1
  12. package/src/classifier/triage.ts +24 -7
  13. package/src/index.ts +4 -1
  14. package/src/lobby/feed.ts +7 -2
  15. package/src/lobby/keys.ts +0 -1
  16. package/src/lobby/quickfix.ts +29 -3
  17. package/src/lobby/runtime.ts +29 -26
  18. package/src/lobby/tabs/home.ts +6 -42
  19. package/src/lobby/tabs/quickfix.ts +1 -1
  20. package/src/lobby/tabs/tasks.ts +3 -2
  21. package/src/lobby/view.ts +18 -23
  22. package/src/master/decisions.ts +10 -2
  23. package/src/pi/activity.ts +1 -33
  24. package/src/pi/commands.ts +22 -11
  25. package/src/pi/events.ts +3 -17
  26. package/src/pi/plan-checklist.ts +305 -0
  27. package/src/pi/route.ts +180 -0
  28. package/src/pi/run-summary.ts +12 -3
  29. package/src/pi/settings-ui.ts +2 -2
  30. package/src/pi/start-task.ts +51 -5
  31. package/src/pi/tools.ts +13 -7
  32. package/src/pi/ui.ts +18 -284
  33. package/src/roles/worker.ts +11 -3
  34. package/src/schemas/configuration.ts +19 -6
  35. package/src/schemas/task.ts +31 -0
  36. package/src/workflow/brief.ts +59 -0
  37. package/src/workflow/track.ts +436 -0
  38. package/src/workflow/workflow.ts +151 -10
  39. package/src/pi/expressions.ts +0 -169
  40. package/src/pi/kaomoji.ts +0 -227
  41. package/src/pi/mascot-art.ts +0 -359
  42. package/src/pi/zen-large.ts +0 -699
  43. package/src/pi/zen-metrics.ts +0 -130
  44. package/src/pi/zen.ts +0 -659
@@ -1,8 +1,6 @@
1
1
  /**
2
2
  * One-word descriptions of the tool an agent is running: subagents show it as
3
- * `[spinner] word - elapsed`, and the oracle names the master's `orchestrate`
4
- * action. Pure lookup:
5
- * no clock, environment or I/O reads.
3
+ * `[spinner] word - elapsed`. Pure lookup: no clock, environment or I/O reads.
6
4
  */
7
5
  const ACTIVITY_WORDS: Record<string, string> = {
8
6
  read: "reading",
@@ -22,46 +20,16 @@ const ACTIVITY_WORDS: Record<string, string> = {
22
20
 
23
21
  const FALLBACK_WORD = "working";
24
22
 
25
- /** What the oracle says between its own tool calls while a master turn runs. */
26
- export const ORACLE_THINKING = "thinking";
27
-
28
23
  export function activityWord(toolName: string): string {
29
24
  return ACTIVITY_WORDS[toolName.trim().toLowerCase()] ?? FALLBACK_WORD;
30
25
  }
31
26
 
32
- /** Master `orchestrate` actions -> the word the oracle shows instead of the generic "orchestrating". */
33
- const ORACLE_ACTION_WORDS: Record<string, string> = {
34
- clarify: "asking",
35
- scout: "scouting",
36
- research: "researching",
37
- propose: "proposing",
38
- plan: "planning",
39
- implement: "delegating",
40
- qa: "reviewing",
41
- knowledge: "recording",
42
- compact: "recording",
43
- decide: "deciding",
44
- resolve_approval: "deciding",
45
- complete: "wrapping up",
46
- status: "checking",
47
- block: "blocking",
48
- resume: "resuming",
49
- cancel: "cancelling",
50
- };
51
-
52
27
  function orchestrateAction(args: unknown): string | undefined {
53
28
  if (!args || typeof args !== "object" || !("action" in args)) return undefined;
54
29
  const action = (args as { action?: unknown }).action;
55
30
  return typeof action === "string" ? action.trim().toLowerCase() : undefined;
56
31
  }
57
32
 
58
- /** One word for what the oracle (master) is doing; `orchestrate` maps by its action. */
59
- export function oracleActivityWord(toolName: string, args?: unknown): string {
60
- if (toolName.trim().toLowerCase() !== "orchestrate") return activityWord(toolName);
61
- const action = orchestrateAction(args);
62
- return (action && ORACLE_ACTION_WORDS[action]) || FALLBACK_WORD;
63
- }
64
-
65
33
  const DETAIL_CHARS = 28;
66
34
 
67
35
  function clip(text: string): string {
@@ -21,19 +21,22 @@ import { applyApprovalChoice, describeTask, describeOversizedKnowledge, lastQaAs
21
21
  import { applyStatus, registerRevealShortcut, setMinimized } from "./ui.ts";
22
22
  import { classifierSummary, openSettings } from "./settings-ui.ts";
23
23
  import { keyStatus } from "../classifier/instance.ts";
24
- import { kickoff, startPlannedTask, startTask } from "./start-task.ts";
24
+ import { kickoff, startPlannedTask } from "./start-task.ts";
25
+ import { startRequest } from "./route.ts";
25
26
  import { setAuto, toggleOwnAuto } from "./owner.ts";
26
27
  import { autoOpenLobby, showLobby } from "../lobby/runtime.ts";
27
28
  import { modelRef, thinkingMismatches } from "./model-support.ts";
28
29
  import { describeRun, runFromLog } from "./run-summary.ts";
29
30
  import { modelLookup } from "./tools.ts";
30
31
  import { budgetLine, budgetState, parseMinutes, readBudget, setBudget, startClock } from "../state/budget.ts";
32
+ import { qaStillDue } from "../workflow/track.ts";
31
33
 
32
34
  const HELP = [
33
35
  "/bot-lobby Open the lobby: tasks, plan, quick fix, metrics (alt+l)",
34
- "/bot-lobby <request> Start a task through the workflow",
35
- "/bot-lobby --task [--auto] <request> Start a task even when the request begins with a subcommand word",
36
+ "/bot-lobby <request> Start a request: a quick fix when one agent can do it alone (the oracle confirms), else a task",
37
+ "/bot-lobby --task [--auto] <request> Always a task (also when the request begins with a subcommand word)",
36
38
  "/bot-lobby --budget 90m <request> Start a task with a time budget the oracle divides between its agents",
39
+ "/bot-lobby --fast|--full <request> Start a task on the fast track (straight to the agents it needs) or the full workflow, whatever it reads as",
37
40
  "/bot-lobby budget [90m|off] Show or set this session's task time budget",
38
41
  "/bot-lobby status [taskId] Show the active task",
39
42
  "/bot-lobby tasks List tasks",
@@ -60,8 +63,8 @@ function isTaskId(value: string | undefined): boolean {
60
63
  return Boolean(value && /^TASK-/.test(value));
61
64
  }
62
65
 
63
- /** A task start's leading flags: `--task` (always a task), `--auto`, `--budget <time>` (or `--budget=<time>`). */
64
- const START_FLAG = /^--(task|auto)(?=\s|$)\s*|^--budget(?:=|\s+)(\S+)\s*/;
66
+ /** A task start's leading flags: `--task` (always a task), `--auto`, `--fast`, `--full`, `--budget <time>` (or `--budget=<time>`). */
67
+ const START_FLAG = /^--(task|auto|fast|full)(?=\s|$)\s*|^--budget(?:=|\s+)(\S+)\s*/;
65
68
 
66
69
  export interface ParsedCommand {
67
70
  sub: string | undefined;
@@ -72,6 +75,10 @@ export interface ParsedCommand {
72
75
  budget?: number;
73
76
  /** `--budget` with something that is not a time. */
74
77
  budgetError?: string;
78
+ /** `--fast` or `--full`: the task's path, whatever its request reads as. */
79
+ track?: "fast" | "full";
80
+ /** `--task`: a task, never routed to the quick-fix agent. */
81
+ task?: boolean;
75
82
  }
76
83
 
77
84
  export function parseCommand(args: string): ParsedCommand {
@@ -79,10 +86,12 @@ export function parseCommand(args: string): ParsedCommand {
79
86
  // Leading flags start a task: what follows is its request, even when it begins with a subcommand word
80
87
  // (a background session started from the lobby sends `--task [--auto] <request>`).
81
88
  let flagged = false;
82
- const extras: Pick<ParsedCommand, "auto" | "budget" | "budgetError"> = {};
89
+ const extras: Pick<ParsedCommand, "auto" | "budget" | "budgetError" | "track" | "task"> = {};
83
90
  for (let match = START_FLAG.exec(trimmed); match; match = START_FLAG.exec(trimmed)) {
84
91
  flagged = true;
85
92
  if (match[1] === "auto") extras.auto = true;
93
+ else if (match[1] === "task") extras.task = true;
94
+ else if (match[1] === "fast" || match[1] === "full") extras.track = match[1];
86
95
  else if (match[2] !== undefined) {
87
96
  const minutes = parseMinutes(match[2]);
88
97
  if (minutes) extras.budget = minutes;
@@ -119,7 +128,7 @@ function showStatus(ctx: ExtensionCommandContext, configDir: string, taskId?: st
119
128
  ].filter(Boolean).join("\n");
120
129
  const footer = knowledge ? `\n${knowledge}` : "";
121
130
  const budget = task ? readBudget(root, configDir, task.id) : undefined;
122
- const time = task && budget ? `\n${budgetLine(budget, budgetState(task.id, budget, task.qaVerdict === "pass"))}` : "";
131
+ const time = task && budget ? `\n${budgetLine(budget, budgetState(task.id, budget, !qaStillDue(task)))}` : "";
123
132
  ctx.ui.notify(task ? `${describeTask(task)}${time}${footer}` : `No bot-lobby task found in ${root}.${footer}`, task ? "info" : "warning");
124
133
  }
125
134
 
@@ -243,7 +252,7 @@ function budgetCommand(ctx: ExtensionCommandContext, configDir: string, value: s
243
252
  if (!task) return ctx.ui.notify("bot-lobby: no active task in this session — a time budget applies to a task (start one with /bot-lobby --budget 90m <request>)", "warning");
244
253
  if (!value) {
245
254
  const budget = readBudget(root, configDir, task.id);
246
- return ctx.ui.notify(budget ? `${task.id} — ${budgetLine(budget, budgetState(task.id, budget, task.qaVerdict === "pass"))}` : `${task.id} has no time budget. Set one with /bot-lobby budget 90m.`, "info");
255
+ return ctx.ui.notify(budget ? `${task.id} — ${budgetLine(budget, budgetState(task.id, budget, !qaStillDue(task)))}` : `${task.id} has no time budget. Set one with /bot-lobby budget 90m.`, "info");
247
256
  }
248
257
  const minutes = /^off$/i.test(value) ? 0 : parseMinutes(value);
249
258
  if (minutes === undefined) return ctx.ui.notify(`bot-lobby: "${value}" is not a time budget (try 90m, 1h or 1h30m)`, "warning");
@@ -251,7 +260,7 @@ function budgetCommand(ctx: ExtensionCommandContext, configDir: string, value: s
251
260
  // Set while the oracle works: its clock runs from now, not from its next turn.
252
261
  if (budget && !ctx.isIdle()) startClock(root, configDir, task.id);
253
262
  applyStatus(ctx, root, configDir);
254
- ctx.ui.notify(budget ? `bot-lobby: ${task.id} — ${budgetLine(budget, budgetState(task.id, budget, task.qaVerdict === "pass"))}` : `bot-lobby: ${task.id} has no time budget now.`, "info");
263
+ ctx.ui.notify(budget ? `bot-lobby: ${task.id} — ${budgetLine(budget, budgetState(task.id, budget, !qaStillDue(task)))}` : `bot-lobby: ${task.id} has no time budget now.`, "info");
255
264
  }
256
265
 
257
266
  function showKnowledge(ctx: ExtensionCommandContext, configDir: string): void {
@@ -315,14 +324,16 @@ export function registerCommands(pi: ExtensionAPI, configDir: string): void {
315
324
  return filtered.length > 0 ? filtered : null;
316
325
  },
317
326
  handler: async (args, ctx) => {
318
- const { sub, rest, restText, auto, budget, budgetError } = parseCommand(args ?? "");
327
+ const { sub, rest, restText, auto, budget, budgetError, track, task } = parseCommand(args ?? "");
319
328
  if (budgetError) return ctx.ui.notify(`bot-lobby: ${budgetError}`, "warning");
320
329
  if (!sub) {
321
330
  if (!restText) {
322
331
  if (!showLobby()) ctx.ui.notify(HELP, "info");
323
332
  return;
324
333
  }
325
- if (await startTask(pi, ctx, configDir, restText, { ...(auto ? { auto } : {}), ...(budget ? { budget } : {}) })) autoOpenLobby();
334
+ // A request one agent can do alone may go to the quick-fix agent once the oracle confirms; the rest start as tasks.
335
+ const started = await startRequest(pi, ctx, configDir, restText, { ...(auto ? { auto } : {}), ...(budget ? { budget } : {}), ...(track ? { track } : {}), ...(task ? { task } : {}) });
336
+ if (started === "task") autoOpenLobby();
326
337
  return;
327
338
  }
328
339
  switch (sub) {
package/src/pi/events.ts CHANGED
@@ -7,14 +7,14 @@ import { selectKnowledge } from "../knowledge/selector.ts";
7
7
  import { cancelAllRuns } from "../execution/agent-runner.ts";
8
8
  import { describeTask } from "../workflow/workflow.ts";
9
9
  import { truncate } from "../text.ts";
10
- import { applyStatus, clearStatus, isMinimized, setMinimized, setOracleActivity } from "./ui.ts";
11
- import { ORACLE_THINKING, oracleActivityWord } from "./activity.ts";
10
+ import { applyStatus, clearStatus, isMinimized, setMinimized } from "./ui.ts";
12
11
  import { isSubagentProcess, webToolsFor } from "./quiet.ts";
13
12
  import { registerQuietTools } from "./tool-renderers.ts";
14
13
  import { taskRequest, type Task, type TaskState } from "../schemas/task.ts";
15
14
  import { pendingComments, readPlanComments, type PlanComment } from "../state/comments.ts";
16
15
  import { isAutoMode } from "../state/auto.ts";
17
16
  import { triageContext } from "../classifier/triage.ts";
17
+ import { qaStillDue } from "../workflow/track.ts";
18
18
  import { previousTaskNote } from "./fresh-context.ts";
19
19
  import { budgetLine, budgetState, pauseClocks, readBudget, resumeClocks, startClock, stopClocks } from "../state/budget.ts";
20
20
 
@@ -33,7 +33,7 @@ export function masterWorkflowContext(task: Task, time = ""): string {
33
33
  export function budgetContext(root: string, configDir: string, task: Task): string {
34
34
  const budget = readBudget(root, configDir, task.id);
35
35
  if (!budget) return "";
36
- const state = budgetState(task.id, budget, task.qaVerdict === "pass");
36
+ const state = budgetState(task.id, budget, !qaStillDue(task));
37
37
  const running = budget.allotments.filter((entry) => !entry.endedAt).map((entry) => `${entry.who} (${entry.minutes}m)`);
38
38
  return [
39
39
  budgetLine(budget, state),
@@ -75,26 +75,16 @@ export function registerLifecycle(pi: ExtensionAPI, configDir: string): void {
75
75
  });
76
76
 
77
77
 
78
- // The oracle's speech bubble mirrors the master's own turn: thinking between
79
- // tool calls, the tool's word during one, and silent (your turn) once it ends.
80
- // Subagents report into runs instead.
81
- // Parallel tool calls: the newest still-running call keeps the word.
82
- const inFlight = new Map<string, string>();
83
- const showOracle = () => setOracleActivity([...inFlight.values()].at(-1) ?? ORACLE_THINKING);
84
78
  // A task's time budget counts while the oracle works on it, not while it waits on the user.
85
79
  const asking = new Set<string>();
86
80
  pi.on("agent_start", (_event, ctx) => {
87
81
  if (isSubagentProcess()) return;
88
- inFlight.clear();
89
- showOracle();
90
82
  const root = detectProjectRoot(ctx.cwd, configDir);
91
83
  const task = activeTask(root, configDir, ctx.sessionManager.getSessionId());
92
84
  if (task) startClock(root, configDir, task.id);
93
85
  });
94
86
  pi.on("tool_execution_start", (event) => {
95
87
  if (isSubagentProcess()) return;
96
- inFlight.set(event.toolCallId, oracleActivityWord(event.toolName, event.args));
97
- showOracle();
98
88
  if (ASKING_TOOLS.has(event.toolName) && !asking.has(event.toolCallId)) {
99
89
  asking.add(event.toolCallId);
100
90
  pauseClocks();
@@ -102,14 +92,10 @@ export function registerLifecycle(pi: ExtensionAPI, configDir: string): void {
102
92
  });
103
93
  pi.on("tool_execution_end", (event) => {
104
94
  if (isSubagentProcess()) return;
105
- inFlight.delete(event.toolCallId);
106
- showOracle();
107
95
  if (asking.delete(event.toolCallId)) resumeClocks();
108
96
  });
109
97
  pi.on("agent_end", () => {
110
98
  if (isSubagentProcess()) return;
111
- inFlight.clear();
112
- setOracleActivity(undefined);
113
99
  for (const _call of asking) resumeClocks();
114
100
  asking.clear();
115
101
  stopClocks();
@@ -0,0 +1,305 @@
1
+ /**
2
+ * The plan as a checklist: its steps parsed from the plan text, and each one
3
+ * done, current or pending from the worker runs so far. Pure.
4
+ */
5
+ import type { AgentRun } from "../schemas/findings.ts";
6
+
7
+ /** Upper bound on parsed plan steps so the checklist stays bounded. */
8
+ export const MAX_PLAN_STEPS = 50;
9
+
10
+ const PREFIX_CHARS = 32;
11
+ const MIN_PREFIX = 8;
12
+ const HEADER_LINE = /^\s*(?:#{1,6}\s+\S.*|\*\*[^*]+\*\*)\s*$/;
13
+ const STEP_SECTION = /sequence|steps|order/i;
14
+ const NUMBERED_STEP_LINE = /^(\s*)(\d+[.)])\s+(.*\S)\s*$/;
15
+ const BULLET_LINE = /^(\s*)([*-])\s+(.*\S)\s*$/;
16
+ const TOP_LEVEL_BULLET = /^()([*-])\s+(.*\S)\s*$/;
17
+ /** `### Step 2: Wire the API`, `**Step 2 — Wire the API**`, `Phase 3) Tests`: one plan step per heading. */
18
+ const STEP_HEADING = /^\s*(?:#{1,6}\s+)?(?:\*\*)?\s*(?:step|phase|stage)\s+#?\d+\s*(?:\*\*)?\s*[:.)\u2014\u2013-]\s*(?:\*\*)?\s*(.*?)\s*(?:\*\*)?\s*$/i;
19
+ const MIN_STEP_HEADINGS = 2;
20
+
21
+ export type PlanStepStatus = "done" | "current" | "pending";
22
+
23
+ export interface PlanStep {
24
+ text: string;
25
+ status: PlanStepStatus;
26
+ }
27
+
28
+ interface ListItem {
29
+ indent: number;
30
+ /** Column where the item's text starts; deeper items are nested under it. */
31
+ content: number;
32
+ text: string;
33
+ }
34
+
35
+ type ItemMatcher = (line: string) => ListItem | undefined;
36
+
37
+ function indentOf(line: string): number {
38
+ return /^\s*/.exec(line)![0].replace(/\t/g, " ").length;
39
+ }
40
+
41
+ function itemMatcher(pattern: RegExp): ItemMatcher {
42
+ return (line) => {
43
+ const match = pattern.exec(line);
44
+ if (!match) return undefined;
45
+ const indent = indentOf(match[1]!);
46
+ return { indent, content: indent + match[2]!.length + 1, text: match[3]! };
47
+ };
48
+ }
49
+
50
+ const numberedItem = itemMatcher(NUMBERED_STEP_LINE);
51
+ const bulletItem = itemMatcher(BULLET_LINE);
52
+ const topBulletItem = itemMatcher(TOP_LEVEL_BULLET);
53
+
54
+ /** Numbered `1.`/`1)` text, or a bullet inside a step section; undefined otherwise. */
55
+ function sectionItem(line: string): ListItem | undefined {
56
+ return numberedItem(line) ?? bulletItem(line);
57
+ }
58
+
59
+ /**
60
+ * Top-level list items only. As in CommonMark, an item indented to its parent's
61
+ * text column is a sub-point of that step, so nested bullets or a nested `1.`
62
+ * list never inflate the checklist. Prose at a shallower indent ends the parent.
63
+ */
64
+ function collectSteps(lines: readonly string[], accept: ItemMatcher): string[] {
65
+ const found: string[] = [];
66
+ let parent: number | undefined;
67
+ for (const line of lines) {
68
+ const item = accept(line);
69
+ if (!item) {
70
+ if (parent !== undefined && line.trim() && indentOf(line) < parent) parent = undefined;
71
+ continue;
72
+ }
73
+ if (parent !== undefined && item.indent >= parent) continue;
74
+ found.push(item.text);
75
+ parent = item.content;
76
+ }
77
+ return found;
78
+ }
79
+
80
+ /** Body of the first `sequence|steps|order` header, up to the next header; undefined when absent. */
81
+ function stepSection(lines: readonly string[]): string[] | undefined {
82
+ const start = lines.findIndex((line) => HEADER_LINE.test(line) && STEP_SECTION.test(line));
83
+ if (start < 0) return undefined;
84
+ const body: string[] = [];
85
+ for (const line of lines.slice(start + 1)) {
86
+ if (HEADER_LINE.test(line)) break;
87
+ body.push(line);
88
+ }
89
+ return body;
90
+ }
91
+
92
+ /** `Step N` headings, when the plan is structured as one heading per step. */
93
+ function headingSteps(lines: readonly string[]): string[] {
94
+ const found: string[] = [];
95
+ for (const line of lines) {
96
+ const match = STEP_HEADING.exec(line);
97
+ if (match) found.push(match[1] || line.replace(/[#*]/g, "").trim());
98
+ }
99
+ return found.length >= MIN_STEP_HEADINGS ? found : [];
100
+ }
101
+
102
+ /**
103
+ * Step texts from a free-form plan, capped. `Step N` headings win; then a
104
+ * `sequence|steps|order` section; then numbered lines; when none exists,
105
+ * top-level bullets are the last resort, so unrelated bullet lists under other
106
+ * headers never leak into the checklist. Only the shallowest items count.
107
+ */
108
+ export function planSteps(plan: string): string[] {
109
+ const lines = plan.split("\n");
110
+ const headings = headingSteps(lines);
111
+ if (headings.length > 0) return headings.slice(0, MAX_PLAN_STEPS);
112
+ const section = stepSection(lines);
113
+ if (section) {
114
+ const sectioned = collectSteps(section, sectionItem);
115
+ if (sectioned.length > 0) return sectioned.slice(0, MAX_PLAN_STEPS);
116
+ }
117
+ const numbered = collectSteps(lines, numberedItem);
118
+ if (numbered.length > 0) return numbered.slice(0, MAX_PLAN_STEPS);
119
+ return collectSteps(lines, topBulletItem).slice(0, MAX_PLAN_STEPS);
120
+ }
121
+
122
+ /* -------------------------------------------------------------------------
123
+ * Step matching. A worker instruction names its step explicitly ("Step 3: ...")
124
+ * or is scored against every step by shared paths, a shared opening phrase and
125
+ * word overlap. Plans reuse file paths across steps, so near-ties go to the
126
+ * earliest step still open instead of the first step that ever mentioned the
127
+ * path -- otherwise every later instruction re-matches step 1 and the tracker
128
+ * never moves.
129
+ * ---------------------------------------------------------------------- */
130
+
131
+ const STEP_REFERENCE = /\bsteps?\s*#?\s*(\d+)(?:\s*(?:-|\u2013|\u2014|to|through|thru|and|&)\s*#?\s*(\d+))?/gi;
132
+ /** Text allowed before a step reference that labels the instruction itself ("Now implement step 3:"). */
133
+ const LEADING_LABEL = /^[\s\W]*(?:(?:now|next|then|please|implement|do|complete|execute|start|begin|finish|work on|continue with|proceed with|plan)\s+)*(?:the\s+)?$/i;
134
+ const MIN_SCORE = 0.5;
135
+ const NEAR_TIE = 0.35;
136
+ const PATH_WEIGHT = 0.75;
137
+ const PREFIX_WEIGHT = 1;
138
+ const WORD = /[a-z0-9_][a-z0-9_./-]*[a-z0-9_]/g;
139
+ const MIN_WORD = 3;
140
+ const STOP_WORDS = new Set([
141
+ "the", "and", "for", "with", "that", "this", "from", "into", "then", "when", "each", "use", "make",
142
+ "sure", "new", "all", "any", "its", "are", "not", "but", "now", "via", "per", "our", "your", "you",
143
+ "has", "have", "will", "should", "must", "also", "only", "step", "steps", "plan", "approved",
144
+ "implement", "please", "add", "update", "change", "changes", "file", "files", "code",
145
+ ]);
146
+
147
+ function normalize(text: string): string {
148
+ return text.replace(/[`*_]/g, "").replace(/\s+/g, " ").trim().toLowerCase();
149
+ }
150
+
151
+ function significantWords(text: string): Set<string> {
152
+ const words = new Set<string>();
153
+ for (const word of normalize(text).match(WORD) ?? []) {
154
+ if (word.length >= MIN_WORD && !STOP_WORDS.has(word)) words.add(word);
155
+ }
156
+ return words;
157
+ }
158
+
159
+ /** Scoring context for one instruction, built once per run and reused for every step. */
160
+ interface InstructionIndex {
161
+ text: string;
162
+ words: Set<string>;
163
+ }
164
+
165
+ function indexInstruction(instruction: string): InstructionIndex {
166
+ return { text: normalize(instruction), words: significantWords(instruction) };
167
+ }
168
+
169
+ function prefixMatches(step: string, hay: string): boolean {
170
+ const tail = normalize(step.replace(/`[^`]+`/g, " ").replace(/^[\s:;,.\u2014\u2013-]+/, ""));
171
+ return tail.length >= MIN_PREFIX && hay.includes(tail.slice(0, PREFIX_CHARS));
172
+ }
173
+
174
+ function pathShare(step: string, hay: string): number {
175
+ const paths = [...step.matchAll(/`([^`]+)`/g)].map((match) => normalize(match[1]!)).filter((path) => path.length > 0);
176
+ if (paths.length === 0) return 0;
177
+ return paths.filter((path) => hay.includes(path)).length / paths.length;
178
+ }
179
+
180
+ function wordShare(step: string, words: ReadonlySet<string>): number {
181
+ const own = significantWords(step);
182
+ if (own.size === 0) return 0;
183
+ let shared = 0;
184
+ for (const word of own) if (words.has(word)) shared += 1;
185
+ return shared / own.size;
186
+ }
187
+
188
+ /** How strongly an instruction targets one step; 0 when it shares nothing. */
189
+ function stepScore(step: string, instruction: InstructionIndex): number {
190
+ const prefix = prefixMatches(step, instruction.text) ? PREFIX_WEIGHT : 0;
191
+ return prefix + PATH_WEIGHT * pathShare(step, instruction.text) + wordShare(step, instruction.words);
192
+ }
193
+
194
+ /**
195
+ * Highest 0-based step an instruction labels itself with ("Step 3: ...", "steps 2-4"),
196
+ * or -1. A reference counts when it opens the instruction or is the only one
197
+ * named, so "building on step 1, now do step 3" does not jump back to step 1.
198
+ */
199
+ export function explicitStepIndex(instruction: string, count: number): number {
200
+ const refs = [...instruction.matchAll(STEP_REFERENCE)];
201
+ if (refs.length === 0) return -1;
202
+ const first = refs[0]!;
203
+ const leading = LEADING_LABEL.test(instruction.slice(0, first.index ?? 0)) ? first : undefined;
204
+ const distinct = new Set(refs.map((ref) => ref[0].toLowerCase().replace(/\s+/g, "")));
205
+ const chosen = leading ?? (distinct.size === 1 ? refs[0] : undefined);
206
+ if (!chosen) return -1;
207
+ const last = Math.max(Number(chosen[1]), Number(chosen[2] ?? chosen[1]));
208
+ return last >= 1 && last <= count ? last - 1 : -1;
209
+ }
210
+
211
+ /**
212
+ * The step an instruction targets, given the steps already completed; -1 when
213
+ * nothing matches. Among near-tied candidates the earliest open step wins.
214
+ */
215
+ export function targetStep(steps: readonly string[], instruction: string | undefined, completed: ReadonlySet<number> = new Set()): number {
216
+ if (!instruction || steps.length === 0) return -1;
217
+ const explicit = explicitStepIndex(instruction, steps.length);
218
+ if (explicit >= 0) return explicit;
219
+ const index = indexInstruction(instruction);
220
+ const scores = steps.map((step) => stepScore(step, index));
221
+ const best = Math.max(...scores);
222
+ if (best < MIN_SCORE) return -1;
223
+ const near = scores.findIndex((score, at) => score >= best - NEAR_TIE && score >= MIN_SCORE && !completed.has(at));
224
+ return near >= 0 ? near : scores.indexOf(best);
225
+ }
226
+
227
+ /** The latest worker run carrying an instruction; its step is the current one. */
228
+ export function latestWorkerRun(runs: readonly AgentRun[]): AgentRun | undefined {
229
+ let latest: AgentRun | undefined;
230
+ for (const run of runs) {
231
+ if (run.role !== "worker" || !run.instruction) continue;
232
+ if (!latest || Date.parse(run.startedAt) >= Date.parse(latest.startedAt)) latest = run;
233
+ }
234
+ return latest;
235
+ }
236
+
237
+ function stepStatus(index: number, current: number): PlanStepStatus {
238
+ if (current < 0) return index === 0 ? "current" : "pending";
239
+ if (index < current) return "done";
240
+ return index === current ? "current" : "pending";
241
+ }
242
+
243
+ /** Lowest index not yet in `completed`, or -1 when every step is. */
244
+ function nextOpenStep(steps: readonly string[], completed: ReadonlySet<number>): number {
245
+ return steps.findIndex((_text, index) => !completed.has(index));
246
+ }
247
+
248
+ /** Marks every step up to and including `index` as completed. */
249
+ function markThrough(completed: Set<number>, index: number): void {
250
+ for (let step = 0; step <= index; step += 1) completed.add(step);
251
+ }
252
+
253
+ function byStart(runs: readonly AgentRun[]): AgentRun[] {
254
+ const time = (run: AgentRun) => {
255
+ const at = Date.parse(run.startedAt);
256
+ return Number.isFinite(at) ? at : 0;
257
+ };
258
+ return runs
259
+ .map((run, order) => ({ run, order }))
260
+ .sort((a, b) => time(a.run) - time(b.run) || a.order - b.order)
261
+ .map((entry) => entry.run);
262
+ }
263
+
264
+ /**
265
+ * Replays worker runs in start order. A successful run completes everything up
266
+ * to its target step; a run that matches nothing -- the common case for a
267
+ * reworded instruction -- completes the next still-open step, so the count only
268
+ * grows and a failure can never tick one off. The latest worker run's own target
269
+ * is reported separately so a running or failed step reads current.
270
+ */
271
+ function replaySteps(steps: readonly string[], runs: readonly AgentRun[]): { completed: number; latest: number } {
272
+ const completed = new Set<number>();
273
+ const latestRun = latestWorkerRun(runs);
274
+ let latest = -1;
275
+ for (const run of byStart(runs)) {
276
+ if (run.role !== "worker") continue;
277
+ const matched = targetStep(steps, run.instruction, completed);
278
+ if (run === latestRun) latest = matched >= 0 && run.status === "success" ? matched + 1 : matched;
279
+ if (run.status !== "success") continue;
280
+ const target = matched >= 0 ? matched : nextOpenStep(steps, completed);
281
+ if (target >= 0) markThrough(completed, target);
282
+ }
283
+ return { completed: completed.size, latest };
284
+ }
285
+
286
+ /** One-entry memo: the panel asks for the same checklist several times per frame. */
287
+ let checklistMemo: { plan: string; runs: readonly AgentRun[]; steps: PlanStep[] } | undefined;
288
+
289
+ /**
290
+ * Done/current/pending per plan step. A succeeded run completes its step, so the
291
+ * next step becomes current (and the last step reads done); running, failed,
292
+ * cancelled and timeout runs keep it current. Progress is monotonic: steps
293
+ * already completed by successful worker runs stay done, and a later call that
294
+ * names an earlier step can never tick it back. Memoized on the exact plan text
295
+ * and runs array, so callers must not mutate either.
296
+ */
297
+ export function planChecklist(plan: string, runs: readonly AgentRun[]): PlanStep[] {
298
+ if (checklistMemo && checklistMemo.plan === plan && checklistMemo.runs === runs) return checklistMemo.steps;
299
+ const texts = planSteps(plan);
300
+ const replay = replaySteps(texts, runs);
301
+ const current = Math.max(replay.completed, replay.latest);
302
+ const steps = texts.map((text, index) => ({ text, status: stepStatus(index, current) }));
303
+ checklistMemo = { plan, runs, steps };
304
+ return steps;
305
+ }