@a-t-h-i/bot-lobby 0.6.5 → 0.6.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +127 -6
  2. package/package.json +1 -1
  3. package/prompts/global.md +18 -0
  4. package/prompts/master.md +77 -2
  5. package/prompts/worker.md +7 -2
  6. package/src/ask/dialog.ts +15 -3
  7. package/src/ask/state.ts +10 -1
  8. package/src/ask/tool.ts +6 -1
  9. package/src/ask/view.ts +2 -1
  10. package/src/execution/agent-runner.ts +23 -1
  11. package/src/execution/fallback.ts +75 -0
  12. package/src/index.ts +1 -1
  13. package/src/lobby/feed.ts +8 -1
  14. package/src/lobby/keys.ts +0 -1
  15. package/src/lobby/layout.ts +22 -1
  16. package/src/lobby/mini.ts +159 -0
  17. package/src/lobby/planner.ts +10 -3
  18. package/src/lobby/quickfix.ts +17 -3
  19. package/src/lobby/runtime.ts +29 -26
  20. package/src/lobby/tabs/home.ts +6 -42
  21. package/src/lobby/tabs/tasks.ts +1 -1
  22. package/src/lobby/view.ts +31 -30
  23. package/src/master/master.ts +1 -1
  24. package/src/master/research.ts +1 -0
  25. package/src/pi/activity.ts +1 -33
  26. package/src/pi/events.ts +3 -16
  27. package/src/pi/master-fallback.ts +61 -0
  28. package/src/pi/model-support.ts +18 -3
  29. package/src/pi/plan-checklist.ts +305 -0
  30. package/src/pi/run-summary.ts +12 -3
  31. package/src/pi/settings-ui.ts +65 -12
  32. package/src/pi/tools.ts +5 -4
  33. package/src/pi/ui.ts +18 -284
  34. package/src/roles/worker.ts +1 -1
  35. package/src/schemas/configuration.ts +51 -15
  36. package/src/schemas/findings.ts +2 -0
  37. package/src/workflow/brief.ts +59 -0
  38. package/src/workflow/workflow.ts +15 -2
  39. package/src/pi/expressions.ts +0 -169
  40. package/src/pi/kaomoji.ts +0 -227
  41. package/src/pi/mascot-art.ts +0 -359
  42. package/src/pi/zen-large.ts +0 -699
  43. package/src/pi/zen-metrics.ts +0 -130
  44. package/src/pi/zen.ts +0 -659
@@ -0,0 +1,305 @@
1
+ /**
2
+ * The plan as a checklist: its steps parsed from the plan text, and each one
3
+ * done, current or pending from the worker runs so far. Pure.
4
+ */
5
+ import type { AgentRun } from "../schemas/findings.ts";
6
+
7
+ /** Upper bound on parsed plan steps so the checklist stays bounded. */
8
+ export const MAX_PLAN_STEPS = 50;
9
+
10
+ const PREFIX_CHARS = 32;
11
+ const MIN_PREFIX = 8;
12
+ const HEADER_LINE = /^\s*(?:#{1,6}\s+\S.*|\*\*[^*]+\*\*)\s*$/;
13
+ const STEP_SECTION = /sequence|steps|order/i;
14
+ const NUMBERED_STEP_LINE = /^(\s*)(\d+[.)])\s+(.*\S)\s*$/;
15
+ const BULLET_LINE = /^(\s*)([*-])\s+(.*\S)\s*$/;
16
+ const TOP_LEVEL_BULLET = /^()([*-])\s+(.*\S)\s*$/;
17
+ /** `### Step 2: Wire the API`, `**Step 2 — Wire the API**`, `Phase 3) Tests`: one plan step per heading. */
18
+ const STEP_HEADING = /^\s*(?:#{1,6}\s+)?(?:\*\*)?\s*(?:step|phase|stage)\s+#?\d+\s*(?:\*\*)?\s*[:.)\u2014\u2013-]\s*(?:\*\*)?\s*(.*?)\s*(?:\*\*)?\s*$/i;
19
+ const MIN_STEP_HEADINGS = 2;
20
+
21
+ export type PlanStepStatus = "done" | "current" | "pending";
22
+
23
+ export interface PlanStep {
24
+ text: string;
25
+ status: PlanStepStatus;
26
+ }
27
+
28
+ interface ListItem {
29
+ indent: number;
30
+ /** Column where the item's text starts; deeper items are nested under it. */
31
+ content: number;
32
+ text: string;
33
+ }
34
+
35
+ type ItemMatcher = (line: string) => ListItem | undefined;
36
+
37
+ function indentOf(line: string): number {
38
+ return /^\s*/.exec(line)![0].replace(/\t/g, " ").length;
39
+ }
40
+
41
+ function itemMatcher(pattern: RegExp): ItemMatcher {
42
+ return (line) => {
43
+ const match = pattern.exec(line);
44
+ if (!match) return undefined;
45
+ const indent = indentOf(match[1]!);
46
+ return { indent, content: indent + match[2]!.length + 1, text: match[3]! };
47
+ };
48
+ }
49
+
50
+ const numberedItem = itemMatcher(NUMBERED_STEP_LINE);
51
+ const bulletItem = itemMatcher(BULLET_LINE);
52
+ const topBulletItem = itemMatcher(TOP_LEVEL_BULLET);
53
+
54
+ /** Numbered `1.`/`1)` text, or a bullet inside a step section; undefined otherwise. */
55
+ function sectionItem(line: string): ListItem | undefined {
56
+ return numberedItem(line) ?? bulletItem(line);
57
+ }
58
+
59
+ /**
60
+ * Top-level list items only. As in CommonMark, an item indented to its parent's
61
+ * text column is a sub-point of that step, so nested bullets or a nested `1.`
62
+ * list never inflate the checklist. Prose at a shallower indent ends the parent.
63
+ */
64
+ function collectSteps(lines: readonly string[], accept: ItemMatcher): string[] {
65
+ const found: string[] = [];
66
+ let parent: number | undefined;
67
+ for (const line of lines) {
68
+ const item = accept(line);
69
+ if (!item) {
70
+ if (parent !== undefined && line.trim() && indentOf(line) < parent) parent = undefined;
71
+ continue;
72
+ }
73
+ if (parent !== undefined && item.indent >= parent) continue;
74
+ found.push(item.text);
75
+ parent = item.content;
76
+ }
77
+ return found;
78
+ }
79
+
80
+ /** Body of the first `sequence|steps|order` header, up to the next header; undefined when absent. */
81
+ function stepSection(lines: readonly string[]): string[] | undefined {
82
+ const start = lines.findIndex((line) => HEADER_LINE.test(line) && STEP_SECTION.test(line));
83
+ if (start < 0) return undefined;
84
+ const body: string[] = [];
85
+ for (const line of lines.slice(start + 1)) {
86
+ if (HEADER_LINE.test(line)) break;
87
+ body.push(line);
88
+ }
89
+ return body;
90
+ }
91
+
92
+ /** `Step N` headings, when the plan is structured as one heading per step. */
93
+ function headingSteps(lines: readonly string[]): string[] {
94
+ const found: string[] = [];
95
+ for (const line of lines) {
96
+ const match = STEP_HEADING.exec(line);
97
+ if (match) found.push(match[1] || line.replace(/[#*]/g, "").trim());
98
+ }
99
+ return found.length >= MIN_STEP_HEADINGS ? found : [];
100
+ }
101
+
102
+ /**
103
+ * Step texts from a free-form plan, capped. `Step N` headings win; then a
104
+ * `sequence|steps|order` section; then numbered lines; when none exists,
105
+ * top-level bullets are the last resort, so unrelated bullet lists under other
106
+ * headers never leak into the checklist. Only the shallowest items count.
107
+ */
108
+ export function planSteps(plan: string): string[] {
109
+ const lines = plan.split("\n");
110
+ const headings = headingSteps(lines);
111
+ if (headings.length > 0) return headings.slice(0, MAX_PLAN_STEPS);
112
+ const section = stepSection(lines);
113
+ if (section) {
114
+ const sectioned = collectSteps(section, sectionItem);
115
+ if (sectioned.length > 0) return sectioned.slice(0, MAX_PLAN_STEPS);
116
+ }
117
+ const numbered = collectSteps(lines, numberedItem);
118
+ if (numbered.length > 0) return numbered.slice(0, MAX_PLAN_STEPS);
119
+ return collectSteps(lines, topBulletItem).slice(0, MAX_PLAN_STEPS);
120
+ }
121
+
122
+ /* -------------------------------------------------------------------------
123
+ * Step matching. A worker instruction names its step explicitly ("Step 3: ...")
124
+ * or is scored against every step by shared paths, a shared opening phrase and
125
+ * word overlap. Plans reuse file paths across steps, so near-ties go to the
126
+ * earliest step still open instead of the first step that ever mentioned the
127
+ * path -- otherwise every later instruction re-matches step 1 and the tracker
128
+ * never moves.
129
+ * ---------------------------------------------------------------------- */
130
+
131
+ const STEP_REFERENCE = /\bsteps?\s*#?\s*(\d+)(?:\s*(?:-|\u2013|\u2014|to|through|thru|and|&)\s*#?\s*(\d+))?/gi;
132
+ /** Text allowed before a step reference that labels the instruction itself ("Now implement step 3:"). */
133
+ const LEADING_LABEL = /^[\s\W]*(?:(?:now|next|then|please|implement|do|complete|execute|start|begin|finish|work on|continue with|proceed with|plan)\s+)*(?:the\s+)?$/i;
134
+ const MIN_SCORE = 0.5;
135
+ const NEAR_TIE = 0.35;
136
+ const PATH_WEIGHT = 0.75;
137
+ const PREFIX_WEIGHT = 1;
138
+ const WORD = /[a-z0-9_][a-z0-9_./-]*[a-z0-9_]/g;
139
+ const MIN_WORD = 3;
140
+ const STOP_WORDS = new Set([
141
+ "the", "and", "for", "with", "that", "this", "from", "into", "then", "when", "each", "use", "make",
142
+ "sure", "new", "all", "any", "its", "are", "not", "but", "now", "via", "per", "our", "your", "you",
143
+ "has", "have", "will", "should", "must", "also", "only", "step", "steps", "plan", "approved",
144
+ "implement", "please", "add", "update", "change", "changes", "file", "files", "code",
145
+ ]);
146
+
147
+ function normalize(text: string): string {
148
+ return text.replace(/[`*_]/g, "").replace(/\s+/g, " ").trim().toLowerCase();
149
+ }
150
+
151
+ function significantWords(text: string): Set<string> {
152
+ const words = new Set<string>();
153
+ for (const word of normalize(text).match(WORD) ?? []) {
154
+ if (word.length >= MIN_WORD && !STOP_WORDS.has(word)) words.add(word);
155
+ }
156
+ return words;
157
+ }
158
+
159
+ /** Scoring context for one instruction, built once per run and reused for every step. */
160
+ interface InstructionIndex {
161
+ text: string;
162
+ words: Set<string>;
163
+ }
164
+
165
+ function indexInstruction(instruction: string): InstructionIndex {
166
+ return { text: normalize(instruction), words: significantWords(instruction) };
167
+ }
168
+
169
+ function prefixMatches(step: string, hay: string): boolean {
170
+ const tail = normalize(step.replace(/`[^`]+`/g, " ").replace(/^[\s:;,.\u2014\u2013-]+/, ""));
171
+ return tail.length >= MIN_PREFIX && hay.includes(tail.slice(0, PREFIX_CHARS));
172
+ }
173
+
174
+ function pathShare(step: string, hay: string): number {
175
+ const paths = [...step.matchAll(/`([^`]+)`/g)].map((match) => normalize(match[1]!)).filter((path) => path.length > 0);
176
+ if (paths.length === 0) return 0;
177
+ return paths.filter((path) => hay.includes(path)).length / paths.length;
178
+ }
179
+
180
+ function wordShare(step: string, words: ReadonlySet<string>): number {
181
+ const own = significantWords(step);
182
+ if (own.size === 0) return 0;
183
+ let shared = 0;
184
+ for (const word of own) if (words.has(word)) shared += 1;
185
+ return shared / own.size;
186
+ }
187
+
188
+ /** How strongly an instruction targets one step; 0 when it shares nothing. */
189
+ function stepScore(step: string, instruction: InstructionIndex): number {
190
+ const prefix = prefixMatches(step, instruction.text) ? PREFIX_WEIGHT : 0;
191
+ return prefix + PATH_WEIGHT * pathShare(step, instruction.text) + wordShare(step, instruction.words);
192
+ }
193
+
194
+ /**
195
+ * Highest 0-based step an instruction labels itself with ("Step 3: ...", "steps 2-4"),
196
+ * or -1. A reference counts when it opens the instruction or is the only one
197
+ * named, so "building on step 1, now do step 3" does not jump back to step 1.
198
+ */
199
+ export function explicitStepIndex(instruction: string, count: number): number {
200
+ const refs = [...instruction.matchAll(STEP_REFERENCE)];
201
+ if (refs.length === 0) return -1;
202
+ const first = refs[0]!;
203
+ const leading = LEADING_LABEL.test(instruction.slice(0, first.index ?? 0)) ? first : undefined;
204
+ const distinct = new Set(refs.map((ref) => ref[0].toLowerCase().replace(/\s+/g, "")));
205
+ const chosen = leading ?? (distinct.size === 1 ? refs[0] : undefined);
206
+ if (!chosen) return -1;
207
+ const last = Math.max(Number(chosen[1]), Number(chosen[2] ?? chosen[1]));
208
+ return last >= 1 && last <= count ? last - 1 : -1;
209
+ }
210
+
211
+ /**
212
+ * The step an instruction targets, given the steps already completed; -1 when
213
+ * nothing matches. Among near-tied candidates the earliest open step wins.
214
+ */
215
+ export function targetStep(steps: readonly string[], instruction: string | undefined, completed: ReadonlySet<number> = new Set()): number {
216
+ if (!instruction || steps.length === 0) return -1;
217
+ const explicit = explicitStepIndex(instruction, steps.length);
218
+ if (explicit >= 0) return explicit;
219
+ const index = indexInstruction(instruction);
220
+ const scores = steps.map((step) => stepScore(step, index));
221
+ const best = Math.max(...scores);
222
+ if (best < MIN_SCORE) return -1;
223
+ const near = scores.findIndex((score, at) => score >= best - NEAR_TIE && score >= MIN_SCORE && !completed.has(at));
224
+ return near >= 0 ? near : scores.indexOf(best);
225
+ }
226
+
227
+ /** The latest worker run carrying an instruction; its step is the current one. */
228
+ export function latestWorkerRun(runs: readonly AgentRun[]): AgentRun | undefined {
229
+ let latest: AgentRun | undefined;
230
+ for (const run of runs) {
231
+ if (run.role !== "worker" || !run.instruction) continue;
232
+ if (!latest || Date.parse(run.startedAt) >= Date.parse(latest.startedAt)) latest = run;
233
+ }
234
+ return latest;
235
+ }
236
+
237
+ function stepStatus(index: number, current: number): PlanStepStatus {
238
+ if (current < 0) return index === 0 ? "current" : "pending";
239
+ if (index < current) return "done";
240
+ return index === current ? "current" : "pending";
241
+ }
242
+
243
+ /** Lowest index not yet in `completed`, or -1 when every step is. */
244
+ function nextOpenStep(steps: readonly string[], completed: ReadonlySet<number>): number {
245
+ return steps.findIndex((_text, index) => !completed.has(index));
246
+ }
247
+
248
+ /** Marks every step up to and including `index` as completed. */
249
+ function markThrough(completed: Set<number>, index: number): void {
250
+ for (let step = 0; step <= index; step += 1) completed.add(step);
251
+ }
252
+
253
+ function byStart(runs: readonly AgentRun[]): AgentRun[] {
254
+ const time = (run: AgentRun) => {
255
+ const at = Date.parse(run.startedAt);
256
+ return Number.isFinite(at) ? at : 0;
257
+ };
258
+ return runs
259
+ .map((run, order) => ({ run, order }))
260
+ .sort((a, b) => time(a.run) - time(b.run) || a.order - b.order)
261
+ .map((entry) => entry.run);
262
+ }
263
+
264
+ /**
265
+ * Replays worker runs in start order. A successful run completes everything up
266
+ * to its target step; a run that matches nothing -- the common case for a
267
+ * reworded instruction -- completes the next still-open step, so the count only
268
+ * grows and a failure can never tick one off. The latest worker run's own target
269
+ * is reported separately so a running or failed step reads current.
270
+ */
271
+ function replaySteps(steps: readonly string[], runs: readonly AgentRun[]): { completed: number; latest: number } {
272
+ const completed = new Set<number>();
273
+ const latestRun = latestWorkerRun(runs);
274
+ let latest = -1;
275
+ for (const run of byStart(runs)) {
276
+ if (run.role !== "worker") continue;
277
+ const matched = targetStep(steps, run.instruction, completed);
278
+ if (run === latestRun) latest = matched >= 0 && run.status === "success" ? matched + 1 : matched;
279
+ if (run.status !== "success") continue;
280
+ const target = matched >= 0 ? matched : nextOpenStep(steps, completed);
281
+ if (target >= 0) markThrough(completed, target);
282
+ }
283
+ return { completed: completed.size, latest };
284
+ }
285
+
286
+ /** One-entry memo: the panel asks for the same checklist several times per frame. */
287
+ let checklistMemo: { plan: string; runs: readonly AgentRun[]; steps: PlanStep[] } | undefined;
288
+
289
+ /**
290
+ * Done/current/pending per plan step. A succeeded run completes its step, so the
291
+ * next step becomes current (and the last step reads done); running, failed,
292
+ * cancelled and timeout runs keep it current. Progress is monotonic: steps
293
+ * already completed by successful worker runs stay done, and a later call that
294
+ * names an earlier step can never tick it back. Memoized on the exact plan text
295
+ * and runs array, so callers must not mutate either.
296
+ */
297
+ export function planChecklist(plan: string, runs: readonly AgentRun[]): PlanStep[] {
298
+ if (checklistMemo && checklistMemo.plan === plan && checklistMemo.runs === runs) return checklistMemo.steps;
299
+ const texts = planSteps(plan);
300
+ const replay = replaySteps(texts, runs);
301
+ const current = Math.max(replay.completed, replay.latest);
302
+ const steps = texts.map((text, index) => ({ text, status: stepStatus(index, current) }));
303
+ checklistMemo = { plan, runs, steps };
304
+ return steps;
305
+ }
@@ -6,16 +6,25 @@
6
6
  import type { AgentRun } from "../schemas/findings.ts";
7
7
  import type { RunLogEntry } from "../schemas/task.ts";
8
8
  import { shortDuration } from "../text.ts";
9
- import { SLOT_LABELS, type SlotId } from "./mascot-art.ts";
10
9
 
11
- /** The zen column a run belongs to; a researcher reports under RESEARCH from any domain. */
10
+ export type SlotId = "dev" | "design" | "research" | "qa";
11
+
12
+ /** Uppercase agent names. */
13
+ const SLOT_LABELS: Record<SlotId, string> = {
14
+ dev: "DEV",
15
+ design: "DESIGN",
16
+ research: "RESEARCH",
17
+ qa: "QA",
18
+ };
19
+
20
+ /** The agent slot a run belongs to; a researcher reports under RESEARCH from any domain. */
12
21
  export function slotOf(run: Pick<AgentRun, "domain" | "role">): SlotId {
13
22
  if (run.role === "researcher") return "research";
14
23
  if (run.domain === "backend") return "dev";
15
24
  return run.domain === "designer" ? "design" : "qa";
16
25
  }
17
26
 
18
- /** Upper-case agent name as the scene shows it (DEV, DESIGN, RESEARCH, QA). */
27
+ /** Upper-case agent name (DEV, DESIGN, RESEARCH, QA). */
19
28
  export function agentName(run: Pick<AgentRun, "domain" | "role">): string {
20
29
  return SLOT_LABELS[slotOf(run)];
21
30
  }
@@ -40,11 +40,14 @@ interface EntryView {
40
40
  thinking?: string;
41
41
  instructions?: string;
42
42
  timeoutMs?: number;
43
+ /** The model this agent switches to when its own runs out of usage; unset = none. */
44
+ fallbackModel?: string;
45
+ fallbackThinking?: string;
43
46
  }
44
47
 
45
48
  function entryView(config: BotLobbyConfig, kind: SettingsKind): EntryView {
46
49
  if (kind === "master") return config.master;
47
- if (kind === "scout") return { model: config.scout.model, timeoutMs: config.scout.timeoutMs };
50
+ if (kind === "scout") return { model: config.scout.model, timeoutMs: config.scout.timeoutMs, ...(config.scout.fallbackModel ? { fallbackModel: config.scout.fallbackModel } : {}) };
48
51
  if (kind === "researcher") return config.researcher;
49
52
  if (kind === "quickfix") return config.quickFix;
50
53
  if (kind === "planner") return config.planner;
@@ -56,21 +59,33 @@ interface EntryPatch {
56
59
  thinking?: string;
57
60
  instructions?: string;
58
61
  timeoutMs?: number;
62
+ /** `inherit` clears the fallback (its thinking level with it). */
63
+ fallbackModel?: string;
64
+ fallbackThinking?: string;
65
+ }
66
+
67
+ /** A settings entry with the fallback patch applied: `inherit` removes the fallback model and its thinking. */
68
+ function withFallbackPatch<T extends { fallbackModel?: string; fallbackThinking?: string }>(entry: T): T {
69
+ if (entry.fallbackModel !== INHERIT_MODEL) return entry;
70
+ const { fallbackModel: _model, fallbackThinking: _thinking, ...rest } = entry;
71
+ return rest as T;
59
72
  }
60
73
 
61
74
  /** Apply a patch to one entry in a config copy; scouts ignore thinking and instructions. */
62
75
  export function patchEntry(config: BotLobbyConfig, kind: SettingsKind, patch: EntryPatch): BotLobbyConfig {
63
76
  const next: BotLobbyConfig = { ...config, agents: { ...config.agents } };
64
- if (kind === "master") next.master = { ...config.master, ...patch } as AgentModelConfig;
77
+ if (kind === "master") next.master = withFallbackPatch({ ...config.master, ...patch } as AgentModelConfig);
65
78
  else if (kind === "scout") {
66
- next.scout = {
79
+ const fallbackModel = patch.fallbackModel ?? config.scout.fallbackModel;
80
+ next.scout = withFallbackPatch({
67
81
  model: patch.model ?? config.scout.model,
68
82
  timeoutMs: patch.timeoutMs ?? config.scout.timeoutMs,
69
- };
70
- } else if (kind === "researcher") next.researcher = { ...config.researcher, ...patch } as AgentModelConfig;
71
- else if (kind === "quickfix") next.quickFix = { ...config.quickFix, ...patch } as AgentModelConfig;
72
- else if (kind === "planner") next.planner = { ...config.planner, ...patch } as AgentModelConfig;
73
- else next.agents[kind] = { ...config.agents[kind], ...patch } as AgentModelConfig;
83
+ ...(fallbackModel ? { fallbackModel } : {}),
84
+ });
85
+ } else if (kind === "researcher") next.researcher = withFallbackPatch({ ...config.researcher, ...patch } as AgentModelConfig);
86
+ else if (kind === "quickfix") next.quickFix = withFallbackPatch({ ...config.quickFix, ...patch } as AgentModelConfig);
87
+ else if (kind === "planner") next.planner = withFallbackPatch({ ...config.planner, ...patch } as AgentModelConfig);
88
+ else next.agents[kind] = withFallbackPatch({ ...config.agents[kind], ...patch } as AgentModelConfig);
74
89
  return next;
75
90
  }
76
91
 
@@ -275,6 +290,37 @@ async function editThinking(pi: ExtensionAPI, ctx: ExtensionContext, kind: Setti
275
290
  await commit(pi, ctx, kind, { thinking: level }, `thinking → ${level}`);
276
291
  }
277
292
 
293
+ /** The model an entry falls back to when its own runs out of usage; `none` removes it. */
294
+ async function editFallbackModel(pi: ExtensionAPI, ctx: ExtensionContext, kind: SettingsKind): Promise<void> {
295
+ const view = entryView(loadConfig(), kind);
296
+ const current = view.fallbackModel ?? INHERIT_MODEL;
297
+ const none: SelectItem = { value: INHERIT_MODEL, label: current === INHERIT_MODEL ? "none ✓" : "none", description: "No fallback: when its model runs out of usage the run fails" };
298
+ const choice = await pick(ctx, `Fallback model — ${kindLabel(kind)}`, [none, ...modelItems(ctx, current, false)], { search: true });
299
+ if (choice === undefined) return;
300
+ const typed = choice === CUSTOM_MODEL ? (await ctx.ui.input("Model id", "provider/model"))?.trim() : choice;
301
+ if (!typed) return;
302
+ await commit(pi, ctx, kind, { fallbackModel: typed }, typed === INHERIT_MODEL ? "fallback removed" : `fallback model → ${typed}`);
303
+ if (typed === INHERIT_MODEL || kind === "scout") return;
304
+ // Keep the fallback's thinking level one its model can run.
305
+ const level = entryView(loadConfig(), kind).fallbackThinking;
306
+ const check = level ? checkThinking(modelLookup(ctx)(typed), level) : undefined;
307
+ if (check?.warning) {
308
+ updateEntry(kind, { fallbackThinking: check.level });
309
+ ctx.ui.notify(`bot-lobby: ${kindLabel(kind)} fallback — ${check.warning}.`, "warning");
310
+ }
311
+ }
312
+
313
+ /** The thinking level on the fallback model, limited to what that model supports. */
314
+ async function editFallbackThinking(pi: ExtensionAPI, ctx: ExtensionContext, kind: SettingsKind): Promise<void> {
315
+ const view = entryView(loadConfig(), kind);
316
+ if (!view.fallbackModel) return;
317
+ const model = modelLookup(ctx)(view.fallbackModel);
318
+ const title = `Fallback thinking — ${kindLabel(kind)} (${view.fallbackModel})`;
319
+ const level = await pick(ctx, title, thinkingItems(supportedThinking(model), view.fallbackThinking ?? view.thinking));
320
+ if (!level || !isThinkingLevel(level)) return;
321
+ await commit(pi, ctx, kind, { fallbackThinking: level }, `fallback thinking → ${level}`);
322
+ }
323
+
278
324
  async function editTimeout(pi: ExtensionAPI, ctx: ExtensionContext, kind: SettingsKind): Promise<void> {
279
325
  const current = entryView(loadConfig(), kind).timeoutMs ?? loadConfig().workflow.agentTimeoutMs;
280
326
  const typed = (await ctx.ui.input(`Time limit (minutes) — ${kindLabel(kind)}`, String(Math.round(current / 60_000))))?.trim();
@@ -303,6 +349,8 @@ export function entryItems(kind: SettingsKind, view: EntryView): SelectItem[] {
303
349
  const items: SelectItem[] = [{ value: "model", label: "Model", description: view.model }];
304
350
  if (kind === "scout") items.push({ value: "fixed", label: "Thinking", description: `${SCOUT_THINKING} (fixed for scouts)` });
305
351
  else items.push({ value: "thinking", label: "Thinking", description: view.thinking ?? "" });
352
+ items.push({ value: "fallback", label: "Fallback model", description: view.fallbackModel ?? "none · used when this model runs out of usage" });
353
+ if (kind !== "scout" && view.fallbackModel) items.push({ value: "fallbackThinking", label: "Fallback thinking", description: view.fallbackThinking ?? view.thinking ?? "" });
306
354
  if (kind !== "master") items.push({ value: "timeout", label: "Time limit", description: minutes(view.timeoutMs) });
307
355
  if (kind !== "scout" && kind !== "researcher") {
308
356
  items.push({ value: "instructions", label: "Instructions", description: view.instructions ? `${view.instructions.length} chars` : "(none)" });
@@ -317,6 +365,8 @@ async function editEntry(pi: ExtensionAPI, ctx: ExtensionContext, kind: Settings
317
365
  if (!action || action === "back") return;
318
366
  if (action === "model") await editModel(pi, ctx, kind);
319
367
  else if (action === "thinking") await editThinking(pi, ctx, kind);
368
+ else if (action === "fallback") await editFallbackModel(pi, ctx, kind);
369
+ else if (action === "fallbackThinking") await editFallbackThinking(pi, ctx, kind);
320
370
  else if (action === "timeout") await editTimeout(pi, ctx, kind);
321
371
  else if (action === "instructions") await editInstructions(pi, ctx, kind);
322
372
  }
@@ -352,23 +402,25 @@ function entryDescription(kind: SettingsKind, view: EntryView): string {
352
402
  const thinking = kind === "scout" ? SCOUT_THINKING : view.thinking;
353
403
  const custom = view.instructions ? " · custom" : "";
354
404
  const limit = kind === "master" ? "" : ` · ${minutes(view.timeoutMs)}`;
355
- return `${view.model} · ${thinking}${limit}${custom}`;
405
+ const fallback = view.fallbackModel ? ` · fallback ${view.fallbackModel}` : "";
406
+ return `${view.model} · ${thinking}${limit}${custom}${fallback}`;
356
407
  }
357
408
 
358
409
  /** The lobby's on/off settings as the settings menu lists them; `panel:*` are the Lobby tab's panes. */
359
- export type LobbySwitch = "autoOpen" | "autoAsk" | "mouse" | "issues" | `panel:${LobbyPanel}`;
410
+ export type LobbySwitch = "autoOpen" | "autoAsk" | "mouse" | "miniLine" | "issues" | `panel:${LobbyPanel}`;
360
411
 
361
- const PANEL_SWITCH_LABELS: Record<LobbyPanel, string> = { animations: "Animations in the lobby", conversation: "Conversation pane", activity: "Activity log pane", thinking: "Thinking pane" };
412
+ const PANEL_SWITCH_LABELS: Record<LobbyPanel, string> = { conversation: "Conversation pane", activity: "Activity log pane", thinking: "Thinking pane" };
362
413
 
363
414
  export const LOBBY_SWITCHES: ReadonlyArray<{ id: LobbySwitch; label: string; help: string }> = [
364
415
  { id: "autoOpen", label: "Open with a task", help: "open the lobby when this session starts or resumes a task" },
365
416
  { id: "autoAsk", label: "Ask at once", help: "put the panel's questions to you as soon as a round ends, while the Plan tab is open" },
366
417
  { id: "mouse", label: "Mouse", help: "click tabs and draft lines, scroll with the wheel (shift+drag still selects text)" },
418
+ { id: "miniLine", label: "Status line when hidden", help: "one line under the editor while the lobby is hidden: task steps, the planning round, a quick fix, or idle" },
367
419
  { id: "issues", label: "Issues tab", help: "the GitHub Issues tab" },
368
420
  ...LOBBY_PANELS.map((panel) => ({
369
421
  id: `panel:${panel}` as const,
370
422
  label: PANEL_SWITCH_LABELS[panel],
371
- help: panel === "animations" ? "the animated oracle and agents on the Lobby tab; off keeps only the task's status (they still show above pi's editor while the lobby is hidden); alt+z toggles it too" : "shown on the Lobby tab; its key in the lobby toggles it too",
423
+ help: "shown on the Lobby tab; its key in the lobby toggles it too",
372
424
  })),
373
425
  ];
374
426
 
@@ -419,6 +471,7 @@ function lobbySummary(config: BotLobbyConfig): string {
419
471
  config.lobby.autoOpen ? "opens with a task" : "opens on alt+l",
420
472
  config.lobby.autoAsk ? "asks at once" : "asks on enter",
421
473
  config.lobby.mouse ? "mouse" : "no mouse",
474
+ ...(config.lobby.miniLine ? [] : ["no status line"]),
422
475
  `planning: ${roundLimitLabel(config.lobby.maxPlanningRounds)}`,
423
476
  ...(config.lobby.issues ? ["issues tab"] : []),
424
477
  ...(hidden.length > 0 ? [`hidden: ${hidden.join(", ")}`] : []),
package/src/pi/tools.ts CHANGED
@@ -29,17 +29,17 @@ const OrchestrateSchema = Type.Object({
29
29
  question: Type.Optional(Type.String({ description: "clarify: question for the user" })),
30
30
  options: Type.Optional(Type.Array(Type.String(), { description: "clarify: optional answer choices; put your recommended one first and mark it (Recommended)" })),
31
31
  domains: Type.Optional(Type.Array(Type.String(), { description: "scout: any of designer, backend, qa" })),
32
- instruction: Type.Optional(Type.String({ description: "scout/research: what to investigate (for scout, also used to target-verify a claim)" })),
32
+ instruction: Type.Optional(Type.String({ description: "scout/research: a self-contained brief: the specific questions, where to look, the answer format you want (paths, names, versions, evidence) and what you will do with it. The agent may be a small model: assume nothing" })),
33
33
  proposal: Type.Optional(Type.String({ description: "propose: the user-facing proposal as a short `- ` bullet list, one line per change" })),
34
34
  concerns: Type.Optional(Type.Array(Type.String(), { description: "propose: concerns raised while challenging the request" })),
35
- plan: Type.Optional(Type.String({ description: "plan: the detailed internal plan (while implementing or reviewing, the full revised plan that replaces it)" })),
35
+ plan: Type.Optional(Type.String({ description: "plan: the detailed internal plan (while implementing or reviewing, the full revised plan that replaces it). Every step names its files, its concrete actions and its done criteria, and the contracts between domains are written out; no decision is left to the workers" })),
36
36
  domain: Type.Optional(Type.String({ description: "implement/research: designer, backend, or qa" })),
37
- task: Type.Optional(Type.String({ description: "implement: the concrete step for that domain's worker" })),
37
+ task: Type.Optional(Type.String({ description: "implement: a self-contained brief for that domain's worker (which may be a small, literal model): Goal, exact Files, numbered What to do with names/shapes/values, Contracts, Constraints, Done when (checkable criteria and commands), If stuck. Decide everything yourself; leave nothing to be assumed" })),
38
38
  assignments: Type.Optional(
39
39
  Type.Array(
40
40
  Type.Object({
41
41
  domain: Type.String({ description: "designer, backend, or qa" }),
42
- task: Type.String({ description: "the concrete step(s) for that domain's worker" }),
42
+ task: Type.String({ description: "a self-contained brief for that domain's worker: Goal, exact Files, numbered What to do, Contracts (in full), Constraints, Done when, If stuck. No assumptions" }),
43
43
  minutes: Type.Optional(Type.Number({ description: "under a time budget: minutes for this worker, by its scope" })),
44
44
  }),
45
45
  { description: "implement: run several domains in parallel (distinct domains); workers share files through the file desk" },
@@ -77,6 +77,7 @@ const DESCRIPTION = [
77
77
  "pass), block/resume (escalate or continue), budget (under a time budget: where it stands, or ask the",
78
78
  "user for more minutes with a reason), track (the task's path and who takes part: show it, or correct it with",
79
79
  "track=fast|full, roster and a reason), status, cancel.",
80
+ "Every instruction you give an agent is read by a possibly smaller, cheaper model that cannot infer intent: write each one as a complete, explicit brief with the goal, files, actions, contracts, constraints and done criteria, then check the report against it.",
80
81
  "The engine validates every step against the task state machine, so a rejected action means the workflow is not at that step yet.",
81
82
  ].join(" ");
82
83