@a-t-h-i/bot-lobby 0.6.5 → 0.6.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +127 -6
  2. package/package.json +1 -1
  3. package/prompts/global.md +18 -0
  4. package/prompts/master.md +77 -2
  5. package/prompts/worker.md +7 -2
  6. package/src/ask/dialog.ts +15 -3
  7. package/src/ask/state.ts +10 -1
  8. package/src/ask/tool.ts +6 -1
  9. package/src/ask/view.ts +2 -1
  10. package/src/execution/agent-runner.ts +23 -1
  11. package/src/execution/fallback.ts +75 -0
  12. package/src/index.ts +1 -1
  13. package/src/lobby/feed.ts +8 -1
  14. package/src/lobby/keys.ts +0 -1
  15. package/src/lobby/layout.ts +22 -1
  16. package/src/lobby/mini.ts +159 -0
  17. package/src/lobby/planner.ts +10 -3
  18. package/src/lobby/quickfix.ts +17 -3
  19. package/src/lobby/runtime.ts +29 -26
  20. package/src/lobby/tabs/home.ts +6 -42
  21. package/src/lobby/tabs/tasks.ts +1 -1
  22. package/src/lobby/view.ts +31 -30
  23. package/src/master/master.ts +1 -1
  24. package/src/master/research.ts +1 -0
  25. package/src/pi/activity.ts +1 -33
  26. package/src/pi/events.ts +3 -16
  27. package/src/pi/master-fallback.ts +61 -0
  28. package/src/pi/model-support.ts +18 -3
  29. package/src/pi/plan-checklist.ts +305 -0
  30. package/src/pi/run-summary.ts +12 -3
  31. package/src/pi/settings-ui.ts +65 -12
  32. package/src/pi/tools.ts +5 -4
  33. package/src/pi/ui.ts +18 -284
  34. package/src/roles/worker.ts +1 -1
  35. package/src/schemas/configuration.ts +51 -15
  36. package/src/schemas/findings.ts +2 -0
  37. package/src/workflow/brief.ts +59 -0
  38. package/src/workflow/workflow.ts +15 -2
  39. package/src/pi/expressions.ts +0 -169
  40. package/src/pi/kaomoji.ts +0 -227
  41. package/src/pi/mascot-art.ts +0 -359
  42. package/src/pi/zen-large.ts +0 -699
  43. package/src/pi/zen-metrics.ts +0 -130
  44. package/src/pi/zen.ts +0 -659
package/README.md CHANGED
@@ -2,7 +2,7 @@
2
2
 
3
3
  A [Pi](https://pi.dev) extension that turns Pi into a multi-agent software team.
4
4
 
5
- ![The bot-lobby status scene: the oracle orchestrating DEV, DESIGN, RESEARCH and QA through a task's plan](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/gallery.png)
5
+ ![The Lobby tab: your conversation with the oracle, the activity log of every agent, and their latest thoughts](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/gallery.png)
6
6
 
7
7
  `/bot-lobby <request>` (or a request typed in the lobby) starts a task, unless
8
8
  one agent can simply do it: then it goes to the [quick-fix agent](#quick-fix-or-the-team).
@@ -16,6 +16,10 @@ The rule: **LLMs decide, the engine enforces.** Agents propose; the
16
16
  extension validates every state change, permission and approval through one
17
17
  `orchestrate` tool.
18
18
 
19
+ The oracle is meant to be your most capable model; the agents can be smaller
20
+ and cheaper ones. So it never assumes they are as capable as it is: it makes
21
+ every decision itself and [briefs each agent in full](#briefing-the-agents).
22
+
19
23
  ## Install
20
24
 
21
25
  ```bash
@@ -35,7 +39,7 @@ tool names.
35
39
 
36
40
  1. `/bot-lobby add a login page` — starts a task; the lobby opens.
37
41
  2. Answer the Master's questions, then approve its proposal.
38
- 3. Watch the agents work in the lobby (`alt+l` shows or hides it).
42
+ 3. Follow the agents in the lobby (`alt+l` shows or hides it): what they do, and what they think.
39
43
 
40
44
  `/bot-lobby settings` sets each agent's model, thinking level, time limit and
41
45
  extra instructions.
@@ -129,6 +133,30 @@ passwords, payments, migrations, production, personal data), or **unclear**
129
133
  - The track shows in the lobby's activity log, the task's details on the
130
134
  Tasks tab, and `/bot-lobby status`.
131
135
 
136
+ ### Briefing the agents
137
+
138
+ Every agent may run on a smaller, cheaper model than the oracle's, one that
139
+ follows instructions well but does not infer intent. So the oracle writes
140
+ each delegation (`implement`, `scout`, `research`) as a self-contained brief
141
+ and settles every design and architecture decision itself first:
142
+
143
+ - **Goal**, the exact **Files**, numbered **What to do** with names, shapes
144
+ and values, the **Contracts** shared with other agents (repeated in full in
145
+ each brief), **Constraints**, **Done when** (checkable criteria and the
146
+ commands to run) and **If stuck**.
147
+ - Plan steps are written to the same standard, so a brief is the plan step
148
+ made explicit, never a new decision.
149
+ - Every agent is told to follow its brief and the approved plan exactly, use
150
+ the given names letter for letter, and report what does not match instead of
151
+ guessing. Workers end their report with a **Brief Check**: each "Done when"
152
+ item, met or not, with evidence.
153
+ - The oracle holds each report to its brief. Drift, a skipped criterion or a
154
+ guessed choice comes back as a fix step with a more explicit brief.
155
+ - The engine backs it up: a delegation that names nothing concrete (or a long
156
+ one with no done criteria) is sent back to the oracle once before any agent
157
+ starts. Sending the same text again goes through, so a short task that is
158
+ complete as written is never stuck. `workflow.briefCheck: false` turns it off.
159
+
132
160
  ### The full workflow
133
161
 
134
162
  - **Scouts** (read-only) investigate the domains the request touches.
@@ -188,16 +216,54 @@ everything. `workflow.freshContext: false` in the config turns this off.
188
216
  ## The lobby
189
217
 
190
218
  A full-screen view with a prompt at the bottom that talks to whatever tab is
191
- open. `alt+h` lists every key.
219
+ open. `alt+h` lists every key. It is text only: no animations, just the
220
+ conversation, the activity log and the thoughts, and a one-line status in Pi's
221
+ footer.
192
222
 
193
223
  | Tab | What it is |
194
224
  | --- | --- |
195
- | **1 Lobby** | The task's status, your conversation with the oracle, an activity log of every tool call, and each agent's latest thought |
225
+ | **1 Lobby** | Your conversation with the oracle, an activity log of every agent's steps, and each agent's latest thought |
196
226
  | **2 Tasks** | Every task and saved plan as a checklist. `s` starts a plan in a new session, `h` here; `c` comments on a plan; `a` archives, `d` deletes |
197
227
  | **3 Plan** | Plan a task with a panel of agents before building it (below) |
198
228
  | **4 Quick fix** | One agent makes a change right away, beside any running task; requests the oracle [routes here](#quick-fix-or-the-team) show up too |
199
229
  | **5 Metrics** | Run time, success rate, tokens and cost per model and agent |
200
230
 
231
+ ### Lobby
232
+
233
+ ![The Lobby tab](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/gallery.png)
234
+
235
+ Your conversation with the oracle on the left, the activity log of every
236
+ agent's steps on the right, and the agents' latest thoughts below. `alt+c`,
237
+ `alt+a` and `alt+k` hide any of the three; the prompt steers the running turn.
238
+
239
+ ### Tasks
240
+
241
+ ![The Tasks tab](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/lobby-tasks.png)
242
+
243
+ Every task as a checklist, with its track, plan progress, the approved plan
244
+ and your comments on it.
245
+
246
+ ### Plan
247
+
248
+ ![The Plan tab](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/lobby-plan.png)
249
+
250
+ The panel's questions, with recommended options, on the left; the draft plan
251
+ on the right. See [Planning](#planning).
252
+
253
+ ### Quick fix
254
+
255
+ ![The Quick fix tab](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/lobby-quickfix.png)
256
+
257
+ One agent's jobs, each with its live steps, the files it edited and its
258
+ report. See [Quick fix or the team](#quick-fix-or-the-team).
259
+
260
+ ### Metrics
261
+
262
+ ![The Metrics tab](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/lobby-metrics.png)
263
+
264
+ Run time, success rate, cost and tokens per model and agent, so you can see
265
+ which cheaper models hold up.
266
+
201
267
  Common keys: `tab` switches tabs, `esc` browses (arrows, single-key
202
268
  commands), `ctrl+f` searches, `ctrl+s` saves the plan, `alt+o` browses
203
269
  sessions, `alt+n` starts a task in a new session, `alt+s` opens settings.
@@ -207,6 +273,25 @@ Rebind any key under `lobby.keys` in the config.
207
273
  Pi session. The Lobby tab can show any session, and your prompt steers it;
208
274
  `● waiting` in the tab bar means one has a question for you.
209
275
 
276
+ **Agents at work.** The bottom line of the lobby shows the subagents running
277
+ right now at its right end (`◐ DESIGN editing 2m · DEV 40s`, a running quick
278
+ fix too), shrinking to names and then a count when the keys leave little room.
279
+
280
+ **Paging.** A pane with more lines than rows shows page buttons on its bottom
281
+ edge, `▲▲ ▲ ▼ ▼▼`: click one to scroll a page up or down (`▲▲` and `▼▼` go two
282
+ pages). The wheel, the arrows and PageUp/PageDown still work. It applies to
283
+ every scrolling pane: the conversation, activity and thinking, the plan draft,
284
+ the Tasks and Quick fix lists and details, and the metrics table.
285
+
286
+ **Status line when hidden.** With the lobby hidden (`alt+l`), one line under
287
+ Pi's editor shows where things stand: a bar of the task's plan steps (or its
288
+ stage before there is a plan) with who is working, the planning round and the
289
+ questions waiting for you, the quick fix in hand, or `idle`. It costs nothing
290
+ while nothing changes. Turn it off with `lobby.miniLine: false` (or in
291
+ `/bot-lobby settings` → Lobby).
292
+
293
+ ![The status line under the editor while the lobby is hidden: a task, planning, idle](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/lobby-status-line.png)
294
+
210
295
  The conversation keeps its newest 100 messages in memory; scroll to the top
211
296
  to load the rest.
212
297
 
@@ -241,7 +326,10 @@ can be compared by looking at them.
241
326
 
242
327
  `↑↓` move · `enter` choose · `space` pick several (multi-select) · `1`–`4`
243
328
  pick · `←→` between questions · the last row takes an answer in your own words
244
- · `esc` puts the questions away (what you answered is kept). Editor hosts that
329
+ · `esc` asks whether to leave (a second `enter`
330
+ leaves, anything else keeps you answering), so a stray press does nothing.
331
+ Questions you leave are never answered for you: the oracle waits and asks again
332
+ when you next write, and the designer asks again before it may decide. Editor hosts that
245
333
  run Pi in RPC mode get the same questions through Pi's own dialogs.
246
334
 
247
335
  **Images.** An option can also carry an `image`: a PNG, JPEG, GIF or WebP
@@ -336,7 +424,7 @@ the result.
336
424
  "scout": { "model": "anthropic/claude-haiku-4-5-20251001", "timeoutMs": 480000 },
337
425
  "planner": { "thinking": "high", "timeoutMs": 300000 },
338
426
  "lobby": { "planningPanel": ["backend", "designer", "qa", "researcher"], "maxPlanningRounds": 5 },
339
- "workflow": { "maxReviewIterations": 2, "maxParallelWorkers": 3, "stallTimeoutMs": 300000, "wrapUpAt": 0.75, "taskBudgetMinutes": 0, "fastTrack": true, "routeQuickFixes": true },
427
+ "workflow": { "maxReviewIterations": 2, "maxParallelWorkers": 3, "stallTimeoutMs": 300000, "wrapUpAt": 0.75, "taskBudgetMinutes": 0, "fastTrack": true, "briefCheck": true, "routeQuickFixes": true },
340
428
  "classifier": { "enabled": false, "provider": "auto", "effort": { "cheapModel": "inherit" } }
341
429
  }
342
430
  ```
@@ -347,9 +435,42 @@ the result.
347
435
  `off, minimal, low, medium, high, xhigh, max`, limited to what the model
348
436
  supports. Scouts always think at `low`.
349
437
  - `instructions` adds your own text to an agent's built-in prompt.
438
+ - `fallbackModel` and `fallbackThinking` on any agent (and the master): see
439
+ [Fallback models](#fallback-models).
350
440
  - Classifier thresholds and limits (`classifier.thresholds`,
351
441
  `classifier.fileHints`) are edited in the file.
352
442
 
443
+ ## Fallback models
444
+
445
+ Running the oracle on a subscription model and the agents on another provider
446
+ means one of them can run out of usage mid-task. Give each agent class a
447
+ **fallback model** and the **thinking level** to run it at (`/bot-lobby
448
+ settings` → the agent → *Fallback model* / *Fallback thinking*). When a run
449
+ fails because its model is out of usage, rate-limited, out of credit or
450
+ unavailable, it runs again on the fallback instead of failing the task.
451
+
452
+ - Works for the master, DESIGN, DEV, QA, the researcher, scouts (their fallback
453
+ thinks at `low` too), quick fixes and the planner and its panel seats.
454
+ - The exhausted model is skipped for 20 minutes, so the next agents go straight
455
+ to their fallback instead of each spending a failed run finding out.
456
+ - **The master** is your own Pi session: on a usage failure the session
457
+ switches to its fallback model and thinking level, tells you, and the oracle
458
+ carries on from where it stopped. Switch back with `/model` when your usage
459
+ returns.
460
+ - Only usage, limit and availability errors switch model; an ordinary failure
461
+ still retries on the same model. If the fallback fails the same way, the run
462
+ fails: it does not chain to a third model.
463
+ - The activity log says when an agent switched, and the run's receipts and the
464
+ Metrics tab show the model that actually ran.
465
+
466
+ ```json
467
+ {
468
+ "master": { "model": "anthropic/claude-fable-5-1", "thinking": "high", "fallbackModel": "deepseek/deepseek-v3", "fallbackThinking": "medium" },
469
+ "agents": { "backend": { "model": "zai/glm-4.6", "thinking": "medium", "fallbackModel": "deepseek/deepseek-v3", "fallbackThinking": "low" } },
470
+ "scout": { "model": "zai/glm-4.6", "fallbackModel": "deepseek/deepseek-v3" }
471
+ }
472
+ ```
473
+
353
474
  ## What the engine enforces
354
475
 
355
476
  | Rule | How |
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@a-t-h-i/bot-lobby",
3
- "version": "0.6.5",
3
+ "version": "0.6.7",
4
4
  "description": "Structured multi-agent software engineering orchestrator for Pi",
5
5
  "type": "module",
6
6
  "license": "Apache-2.0",
package/prompts/global.md CHANGED
@@ -15,6 +15,24 @@ Perform your assigned responsibility precisely and remain within your domain.
15
15
  - Do not invent requirements; ask when they are genuinely ambiguous.
16
16
  - Report conclusions, evidence, decisions, findings and blockers concisely.
17
17
 
18
+ ## Following your brief
19
+
20
+ The Master that instructs you plans the work and knows the whole task; you may
21
+ be a smaller model that sees only your part. Your brief and the approved plan
22
+ are your authority, so:
23
+
24
+ - Read the whole brief and the plan before acting, and do exactly what it
25
+ says, in its order. Do not add features, refactors or "improvements", and do
26
+ not skip steps because they look unnecessary.
27
+ - Use the names, paths, shapes and wording the brief gives you, letter for
28
+ letter. Where it leaves a detail open, choose the simplest option that
29
+ follows the existing code, and say what you chose in your report.
30
+ - Never guess at something that matters (a contract, a file that is not there,
31
+ a conflict between the brief and the code). Stop that part, finish the rest,
32
+ and report it under Blockers or Pushback with what you found.
33
+ - Check your own work against the brief's "Done when" list before you report,
34
+ criterion by criterion, and say honestly which are met and which are not.
35
+
18
36
  ## Hard rules
19
37
 
20
38
  - Do not add dependencies without approval.
package/prompts/master.md CHANGED
@@ -68,6 +68,12 @@ more), skip the researcher when research is not needed, clarify only when it
68
68
  reads the request as ambiguous — and overrule it whenever the repository says
69
69
  otherwise. It is a hint, never a rule.
70
70
 
71
+ When the user leaves your questions unanswered (they put them away, or
72
+ `ask_user_question` says so), the decision is still theirs: never assume the
73
+ answers, never fall back on the recommended options, and never carry on with
74
+ work that depends on them. Say in one short line that the questions are
75
+ waiting, end your turn, and ask again when they next write.
76
+
71
77
  When you `clarify` with options, put your recommended option first and mark
72
78
  it `(Recommended)`. When the request already makes it clearly right, the
73
79
  classifier answers for you: the reply says so, the decision is recorded, and
@@ -155,12 +161,81 @@ each `implement` task with its step number (`Step 3: ...`, or `Steps 3-4: ...`
155
161
  when one delegation covers several) so the user's checklist tracks progress
156
162
  exactly.
157
163
 
164
+ ## Briefing the agents
165
+
166
+ You are usually a far more capable model than the agents you delegate to. Scouts,
167
+ workers, the researcher and the reviewer may run on smaller, cheaper models
168
+ that follow instructions well but do not infer intent, fill gaps sensibly or
169
+ know what you know. Never assume they are as capable as you. Whatever you leave
170
+ unsaid, they will guess, and a wrong guess costs a whole agent run. Your plan
171
+ and every brief are how your goal reaches the code, so write them for a
172
+ capable but literal reader who has read nothing but the brief and the
173
+ repository.
174
+
175
+ Every `implement` task (each assignment in a parallel batch), `scout` and
176
+ `research` instruction is a self-contained brief with these parts, in this order:
177
+
178
+ 1. **Goal** — the outcome this step must produce and how it serves the user's
179
+ request and the approved plan, in one or two sentences. Name the step
180
+ number(s).
181
+ 2. **Files** — the exact paths to create or change, and the ones to leave
182
+ alone. When you do not know a path, say what to search for and where.
183
+ 3. **What to do** — numbered, concrete actions in the order to do them: names
184
+ of functions, components, endpoints, fields, types, CSS classes, strings,
185
+ values. Give the exact signature, shape or wording wherever it matters.
186
+ Write "use X", not "use a suitable library"; when a choice is left to the
187
+ agent, say which options are allowed and how to pick.
188
+ 4. **Contracts** — everything this step shares with another domain or step:
189
+ API shapes, status codes, error format, event names, shared types, data
190
+ formats, file locations. State them in full in every brief that touches
191
+ them; an agent never sees another agent's brief.
192
+ 5. **Constraints** — what it must not do: no new dependencies, no other files,
193
+ no refactors, no changed behavior outside the step, no restyling of code
194
+ it does not own. Repeat the user's explicit requirements that apply.
195
+ 6. **Done when** — a checklist of observable, checkable criteria (behaviors,
196
+ exact commands to run and what they should print, files that must exist),
197
+ including what to verify and how, with `timeout`. The agent must be able to
198
+ tell for itself whether it has finished.
199
+ 7. **If stuck** — what to do when something does not match the brief (a file is
200
+ missing, a name differs, two instructions conflict): stop that part, do not
201
+ invent a workaround, and report it under Blockers or Pushback with what it
202
+ found. Ask nothing you can answer yourself: settle it in the brief.
203
+
204
+ Rules for the brief:
205
+
206
+ - Decide first, delegate second. Every design, architecture and product
207
+ decision belongs to you; make it and write down the result. A brief must not
208
+ contain "consider", "as appropriate", "if needed", "etc.", "similar to",
209
+ "handle edge cases" or "make it look good" without the specifics. List the
210
+ edge cases; describe the look in concrete terms (layout, sizes, colors,
211
+ states).
212
+ - Say the obvious. Repeat what you already told an earlier agent, include the
213
+ conventions to follow and point to an existing file to imitate by path.
214
+ - One step, one purpose, small enough to hold in mind: a handful of files and
215
+ a few actions. Split anything larger into consecutive steps in the same
216
+ `implement` call rather than leaving the agent to sequence it. Prefer more
217
+ explicit detail to fewer, larger chunks.
218
+ - The plan's steps are written to the same standard: each step names its
219
+ files, its actions and its done criteria, so the brief is the step made
220
+ explicit, never a new decision.
221
+ - Scout and research instructions ask specific questions with the answer
222
+ format you want (paths, names, versions, yes/no plus evidence), and say what
223
+ you will do with the answer.
224
+
225
+ When a report comes back, hold it to the brief: check each **Done when**
226
+ criterion against the report's `## Brief Check`, the diff and the repository.
227
+ Drift, a skipped criterion or a guessed choice is a fix step with a corrected,
228
+ even more explicit brief that quotes the exact gap — not a reason to accept the
229
+ work, and not a reason to redo it yourself. Keep every agent on your plan: if
230
+ the code no longer matches it, say which step it deviates from and restore it.
231
+
158
232
  ## Speed
159
233
 
160
234
  Every delegation costs a full agent run, so keep the loop short:
161
235
 
162
- - Delegate fewer, larger chunks: one `implement` per domain covering its
163
- consecutive steps (`Steps 2-4: ...`) rather than one call per step.
236
+ - Delegate fewer calls, not vaguer ones: one `implement` per domain covering
237
+ its consecutive steps (`Steps 2-4: ...`) rather than one call per step, each
238
+ step still briefed in full (see Briefing the agents).
164
239
  - When steps for different domains are independent, run them together with
165
240
  `implement` `assignments` (one entry per domain). Workers then share files
166
241
  through the file desk: they claim files, queue for busy ones, and hand them
package/prompts/worker.md CHANGED
@@ -19,7 +19,9 @@ them.
19
19
 
20
20
  ## Implementation
21
21
 
22
- - Follow the approved plan.
22
+ - Follow the approved plan and your brief exactly: they are the Master's
23
+ decisions. Do not substitute your own design, rename things, or widen the
24
+ step. If the code contradicts the brief, do not improvise: report it.
23
25
  - Follow domain boundaries.
24
26
 
25
27
  If you need a new dependency, or you believe a significant architectural
@@ -73,7 +75,10 @@ repository at the same time, and files are checked out like physical documents:
73
75
 
74
76
  ## Before handoff
75
77
 
76
- - Inspect the actual diff.
78
+ - Go through the brief's "Done when" list one criterion at a time and record
79
+ each in `## Brief Check` as `- criterion — met|not met — evidence`.
80
+ - Inspect the actual diff: it must contain what the brief asked for and
81
+ nothing else.
77
82
  - Verify tests.
78
83
  - Update the temporary task scratchpad.
79
84
  - Report concise results.
package/src/ask/dialog.ts CHANGED
@@ -9,7 +9,7 @@ import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
9
9
  import { Key, matchesKey, type Component, type TUI } from "@earendil-works/pi-tui";
10
10
  import type { LobbyTheme } from "../lobby/layout.ts";
11
11
  import { lobbyTheme } from "../lobby/theme.ts";
12
- import { initialState, step, type AskKey, type AskState } from "./state.ts";
12
+ import { initialState, putAway, step, type AskKey, type AskState } from "./state.ts";
13
13
  import { renderAsk, type AskFrame } from "./view.ts";
14
14
  import { loadImages } from "./image.ts";
15
15
  import { isAbsolute, resolve } from "node:path";
@@ -68,7 +68,9 @@ export class AskDialog implements Component {
68
68
 
69
69
  /** Put the questions away from outside (the turn was aborted). */
70
70
  cancel(): void {
71
- if (!this.state.result) this.handleInput("\x1b");
71
+ if (this.state.result) return;
72
+ this.state = putAway(this.state);
73
+ this.done(this.state.result!);
72
74
  }
73
75
 
74
76
  render(width: number): string[] {
@@ -83,7 +85,17 @@ export class AskDialog implements Component {
83
85
  * cannot draw the questionnaire. Without any UI nobody can answer: the
84
86
  * result says the questions were put away.
85
87
  */
86
- export const askUser: Asker = async (questions, ctx, signal, from) => {
88
+ export const askUser: Asker = (questions, ctx, signal, from) => {
89
+ // A model may call the tool several times in one turn; pi runs them in parallel, and overlays opened together hide each other. One at a time.
90
+ const next = asking.then(() => (signal?.aborted ? { answers: [], cancelled: true } : askNow(questions, ctx, signal, from)));
91
+ asking = next.then(() => undefined, () => undefined);
92
+ return next;
93
+ };
94
+
95
+ /** Tail of the questions waiting their turn. */
96
+ let asking: Promise<void> = Promise.resolve();
97
+
98
+ const askNow: Asker = async (questions, ctx, signal, from) => {
87
99
  if (!ctx.hasUI || questions.length === 0) return { answers: [], cancelled: true };
88
100
  if ((ctx as { mode?: string }).mode === "rpc") return dialogAsker(questions, ctx, signal, from);
89
101
  // Option images are read before the questionnaire opens, so drawing it never waits on the disk.
package/src/ask/state.ts CHANGED
@@ -38,10 +38,17 @@ export interface AskState {
38
38
  /** Writing the own answer of the question in view. */
39
39
  editing: boolean;
40
40
  draft: string;
41
+ /** Esc was pressed: asked whether to leave without answering; enter leaves, any other key keeps answering. */
42
+ leaving?: boolean;
41
43
  /** Set once the user submits or puts the questions away. */
42
44
  result?: AskResult;
43
45
  }
44
46
 
47
+ /** The questions put away, with what was answered so far (also when something outside ends them). */
48
+ export function putAway(state: AskState): AskState {
49
+ return { ...state, editing: false, draft: "", leaving: false, result: { answers: answersOf(state), cancelled: true } };
50
+ }
51
+
45
52
  export function initialState(questions: readonly AskQuestion[]): AskState {
46
53
  return {
47
54
  questions,
@@ -127,6 +134,8 @@ function typing(state: AskState, key: AskKey): AskState {
127
134
  /** One key press. */
128
135
  export function step(state: AskState, key: AskKey): AskState {
129
136
  if (state.result) return state;
137
+ // Esc asks first: leaving the questions unanswered lets the oracle carry on without you, so it takes a second key.
138
+ if (state.leaving) return key.type === "enter" || (key.type === "text" && key.value.toLowerCase() === "y") ? putAway(state) : { ...state, leaving: false };
130
139
  if (state.editing) return typing(state, key);
131
140
  const question = state.questions[state.tab];
132
141
  if (!question) return { ...state, result: { answers: [], cancelled: false } };
@@ -153,7 +162,7 @@ export function step(state: AskState, key: AskKey): AskState {
153
162
  return advance(state);
154
163
  }
155
164
  case "escape":
156
- return { ...state, result: { answers: answersOf(state), cancelled: true } };
165
+ return { ...state, leaving: true };
157
166
  default:
158
167
  return state;
159
168
  }
package/src/ask/tool.ts CHANGED
@@ -65,12 +65,16 @@ export function invalidQuestions(questions: readonly AskQuestion[]): string | un
65
65
  return undefined;
66
66
  }
67
67
 
68
+ /** What the model does with questions the user left unanswered: wait, never fill them in. */
69
+ const STILL_OPEN = "Do not assume answers, do not pick the recommended options, and do not go on with work that depends on them. Say in one short line that the questions are waiting, end your turn, and put them to the user again when they next write.";
70
+ const NOT_ANSWERED = `The user left the questions without answering. ${STILL_OPEN}`;
71
+
68
72
  /** What the model reads back: each question with its answer, or that it was skipped. */
69
73
  export function answerSummary(questions: readonly AskQuestion[], result: AskResult): string {
70
74
  if (result.cancelled && result.answers.length === 0) {
71
75
  // A relay that could not ask anyone (auto mode) says why.
72
76
  if (result.globalNote) return result.globalNote;
73
- return "The user put the questions away without answering. Do not ask the same again right away: go on with your best judgement and say what you assumed, or ask something narrower.";
77
+ return NOT_ANSWERED;
74
78
  }
75
79
  const lines = questions.map((question, index) => {
76
80
  const answer = result.answers.find((entry) => entry.questionIndex === index);
@@ -82,6 +86,7 @@ export function answerSummary(questions: readonly AskQuestion[], result: AskResu
82
86
  return [
83
87
  result.cancelled ? "The user answered some questions, then put the rest away:" : "The user answered:",
84
88
  ...lines,
89
+ ...(result.cancelled ? ["", `The questions marked "(not answered)" are still open. ${STILL_OPEN}`] : []),
85
90
  ...(result.globalNote ? ["", `Note: ${result.globalNote}`] : []),
86
91
  ].join("\n");
87
92
  }
package/src/ask/view.ts CHANGED
@@ -112,6 +112,7 @@ function previewBox(preview: Preview, width: number, rows: number, theme: LobbyT
112
112
  }
113
113
 
114
114
  function hints(state: AskState, question: AskQuestion, width: number, theme?: LobbyTheme): string[] {
115
+ if (state.leaving) return wrap(paint(theme, "warning", "Leave without answering? The oracle will not guess for you. enter leaves · any other key keeps answering"), width);
115
116
  const parts = state.editing
116
117
  ? ["enter keep it", "esc back to the options"]
117
118
  : [
@@ -119,7 +120,7 @@ function hints(state: AskState, question: AskQuestion, width: number, theme?: Lo
119
120
  question.multiSelect ? "space pick · enter next" : "enter choose",
120
121
  `1-${question.options.length} pick`,
121
122
  ...(state.questions.length > 1 ? ["←→ questions"] : []),
122
- "esc put away",
123
+ "esc leave",
123
124
  ];
124
125
  return wrap(paint(theme, "dim", parts.join(" · ")), width);
125
126
  }
@@ -7,6 +7,7 @@ import { shortDuration, truncate } from "../text.ts";
7
7
  import { EditLog } from "../state/changes.ts";
8
8
  import { formatMinutes, REPORT_GRACE_MS } from "../state/budget.ts";
9
9
  import { runPiAgent, spawnPiProcess, type PiStreamEvent, type ProcessRunner, type RelayAsk } from "./pi-runner.ts";
10
+ import { isUnavailable, looksUnavailable, markUnavailable, usableFallback } from "./fallback.ts";
10
11
  import { ASK_ENV } from "../ask/relay.ts";
11
12
  import { ASK_TOOL } from "../ask/types.ts";
12
13
 
@@ -40,6 +41,10 @@ export interface AgentRequest {
40
41
  context: AgentContext;
41
42
  model?: string;
42
43
  thinking?: string;
44
+ /** Where the run goes when its model runs out of usage or is unavailable. */
45
+ fallback?: { model: string; thinking: string };
46
+ /** Set by the runner once it has switched: the model that ran out. */
47
+ fellBackFrom?: string;
43
48
  timeoutMs: number;
44
49
  cwd: string;
45
50
  signal?: AbortSignal;
@@ -135,13 +140,24 @@ function retryable(run: AgentRun): boolean {
135
140
  */
136
141
  export async function runAgent(request: AgentRequest, run: ProcessRunner = spawnPiProcess): Promise<AgentRun> {
137
142
  const startedAt = new Date().toISOString();
138
- const attempts = Math.max(1, (request.retries ?? 0) + 1);
143
+ let attempts = Math.max(1, (request.retries ?? 0) + 1);
139
144
  // A failed attempt may have edited files before the retry: the run owns every edit.
140
145
  const edits = new EditLog(request.cwd);
141
146
  let last: AgentRun | undefined;
147
+ // A model already known to be out of usage goes straight to the fallback.
148
+ const other = usableFallback(request.fallback, request.model);
149
+ if (other && isUnavailable(request.model)) request = switchToFallback(request, other, request.model ?? "the session model");
142
150
  for (let attempt = 1; attempt <= attempts; attempt++) {
143
151
  last = await runAgentOnce(request, run, attempt, startedAt, edits);
144
152
  request.onAttemptEnd?.(last);
153
+ // Out of usage (or the model unavailable): the same model would fail again, so the fallback takes the retry.
154
+ const fallback = usableFallback(request.fallback, request.model);
155
+ if (fallback && !request.fellBackFrom && last.status === "failed" && looksUnavailable(last.error)) {
156
+ if (request.model) markUnavailable(request.model);
157
+ request = switchToFallback(request, fallback, request.model ?? "the session model");
158
+ attempts += 1;
159
+ continue;
160
+ }
145
161
  if (!retryable(last)) break;
146
162
  // Under a budget a retry only uses what is left of the allotment.
147
163
  if (request.time && request.time.endsAt - Date.now() < MIN_ATTEMPT_MS) break;
@@ -150,6 +166,11 @@ export async function runAgent(request: AgentRequest, run: ProcessRunner = spawn
150
166
  return edited.length > 0 ? { ...last!, edited } : last!;
151
167
  }
152
168
 
169
+ /** The request moved to its fallback model and thinking level. */
170
+ function switchToFallback(request: AgentRequest, fallback: { model: string; thinking?: string }, from: string): AgentRequest {
171
+ return { ...request, model: fallback.model, thinking: fallback.thinking ?? request.thinking, fellBackFrom: from };
172
+ }
173
+
153
174
  /** Abort every in-flight subagent (session shutdown, user cancel). */
154
175
  export function cancelAllRuns(): void {
155
176
  for (const controller of activeControllers) controller.abort();
@@ -168,6 +189,7 @@ function baseRun(request: AgentRequest, runId: string, startedAt: string, attemp
168
189
  attempts,
169
190
  startedAt,
170
191
  ...(request.thinking ? { thinking: request.thinking } : {}),
192
+ ...(request.fellBackFrom ? { fellBackFrom: request.fellBackFrom } : {}),
171
193
  ...(request.routedFrom ? { routedFrom: request.routedFrom } : {}),
172
194
  ...(request.route ? { route: request.route } : {}),
173
195
  ...(request.time ? { allotMs: request.time.allotMs, endsAt: request.time.endsAt } : {}),
@@ -0,0 +1,75 @@
1
+ /**
2
+ * Fallback models. A subscription or key that runs out mid-task (a usage
3
+ * limit, a rate limit, no credit, the provider down or refusing the key)
4
+ * fails every run on that model the same way, so the run goes to the agent's
5
+ * configured fallback model instead, and the exhausted model is skipped for a
6
+ * while so the next agents do not each spend a failed run finding out.
7
+ */
8
+
9
+ /** What a provider says when a model cannot be used right now. Deliberately narrow: an ordinary failure is not a reason to switch model. */
10
+ const UNAVAILABLE = [
11
+ /usage limit/i, /rate.?limit/i, /\bquota\b/i, /limit (?:reached|exceeded|will reset)/i, /exceeded (?:your|the|current)/i,
12
+ /out of (?:usage|credits?|extra usage|tokens)/i, /(?:insufficient|no) (?:credits?|funds|balance|quota)/i, /credit balance/i,
13
+ /billing/i, /payment required/i, /\b402\b/, /\b429\b/, /too many requests/i, /resource.?exhausted/i,
14
+ /overloaded/i, /\b529\b/, /(?:at|over) capacity/i, /resets? (?:at|in|on)/i, /subscription/i,
15
+ /unauthori[sz]ed/i, /\b401\b/, /invalid (?:api )?key/i, /no api key/i, /authentication (?:failed|error)/i,
16
+ /model (?:is )?(?:not found|not available|unavailable)/i, /unknown model/i, /does not have access/i,
17
+ ];
18
+
19
+ /** True when a failed run's error says its model is out of usage or unavailable, so another model may succeed. */
20
+ export function looksUnavailable(error: string | undefined): boolean {
21
+ return Boolean(error && UNAVAILABLE.some((pattern) => pattern.test(error)));
22
+ }
23
+
24
+ /** How long a model that ran out is skipped before it is tried again. */
25
+ export const COOLDOWN_MS = 20 * 60 * 1000;
26
+
27
+ const exhausted = new Map<string, number>();
28
+
29
+ /** Remember that `model` cannot be used until the cooldown passes. */
30
+ export function markUnavailable(model: string, now = Date.now()): void {
31
+ exhausted.set(model, now + COOLDOWN_MS);
32
+ }
33
+
34
+ /** Whether `model` ran out recently enough to skip. */
35
+ export function isUnavailable(model: string | undefined, now = Date.now()): boolean {
36
+ if (!model) return false;
37
+ const until = exhausted.get(model);
38
+ if (until === undefined) return false;
39
+ if (until <= now) {
40
+ exhausted.delete(model);
41
+ return false;
42
+ }
43
+ return true;
44
+ }
45
+
46
+ /** Forget every exhausted model (tests). */
47
+ export function resetUnavailable(): void {
48
+ exhausted.clear();
49
+ }
50
+
51
+ /** A fallback worth switching to: a model other than the one that failed. */
52
+ export function usableFallback<F extends { model: string }>(fallback: F | undefined, current: string | undefined): F | undefined {
53
+ return fallback && fallback.model !== current ? fallback : undefined;
54
+ }
55
+
56
+ /**
57
+ * Run `attempt` on `model`; when it fails because that model is out of usage
58
+ * or unavailable, run it again on the fallback. A model already known to be
59
+ * out goes straight to the fallback. `switched` says which model took over.
60
+ */
61
+ export async function withFallback<T extends { status: string; error?: string }>(
62
+ model: string | undefined,
63
+ thinking: string,
64
+ fallback: { model: string; thinking: string } | undefined,
65
+ attempt: (model: string | undefined, thinking: string) => Promise<T>,
66
+ ): Promise<{ result: T; switchedFrom?: string }> {
67
+ const other = usableFallback(fallback, model);
68
+ if (other && isUnavailable(model)) return { result: await attempt(other.model, other.thinking), switchedFrom: model ?? "the session model" };
69
+ const result = await attempt(model, thinking);
70
+ if (other && result.status === "failed" && looksUnavailable(result.error)) {
71
+ if (model) markUnavailable(model);
72
+ return { result: await attempt(other.model, other.thinking), switchedFrom: model ?? "the session model" };
73
+ }
74
+ return { result };
75
+ }
package/src/index.ts CHANGED
@@ -21,7 +21,7 @@ export default function (pi: ExtensionAPI): void {
21
21
  // Before the lifecycle, so a task that ended while the oracle was idle is closed before it builds the next turn's prompt.
22
22
  registerFreshContext(pi, CONFIG_DIR_NAME);
23
23
  registerLifecycle(pi, CONFIG_DIR_NAME);
24
- // After the lifecycle, so the lobby opens over a task the widget state already knows.
24
+ // After the lifecycle, so the lobby opens over a task the status already knows.
25
25
  registerLobbyEvents(pi, CONFIG_DIR_NAME);
26
26
  registerOwner(pi, CONFIG_DIR_NAME);
27
27
  onTransition((task) => pingTransition(task));