@a-t-h-i/bot-lobby 0.6.5 → 0.6.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +127 -6
- package/package.json +1 -1
- package/prompts/global.md +18 -0
- package/prompts/master.md +77 -2
- package/prompts/worker.md +7 -2
- package/src/ask/dialog.ts +15 -3
- package/src/ask/state.ts +10 -1
- package/src/ask/tool.ts +6 -1
- package/src/ask/view.ts +2 -1
- package/src/execution/agent-runner.ts +23 -1
- package/src/execution/fallback.ts +75 -0
- package/src/index.ts +1 -1
- package/src/lobby/feed.ts +8 -1
- package/src/lobby/keys.ts +0 -1
- package/src/lobby/layout.ts +22 -1
- package/src/lobby/mini.ts +159 -0
- package/src/lobby/planner.ts +10 -3
- package/src/lobby/quickfix.ts +17 -3
- package/src/lobby/runtime.ts +29 -26
- package/src/lobby/tabs/home.ts +6 -42
- package/src/lobby/tabs/tasks.ts +1 -1
- package/src/lobby/view.ts +31 -30
- package/src/master/master.ts +1 -1
- package/src/master/research.ts +1 -0
- package/src/pi/activity.ts +1 -33
- package/src/pi/events.ts +3 -16
- package/src/pi/master-fallback.ts +61 -0
- package/src/pi/model-support.ts +18 -3
- package/src/pi/plan-checklist.ts +305 -0
- package/src/pi/run-summary.ts +12 -3
- package/src/pi/settings-ui.ts +65 -12
- package/src/pi/tools.ts +5 -4
- package/src/pi/ui.ts +18 -284
- package/src/roles/worker.ts +1 -1
- package/src/schemas/configuration.ts +51 -15
- package/src/schemas/findings.ts +2 -0
- package/src/workflow/brief.ts +59 -0
- package/src/workflow/workflow.ts +15 -2
- package/src/pi/expressions.ts +0 -169
- package/src/pi/kaomoji.ts +0 -227
- package/src/pi/mascot-art.ts +0 -359
- package/src/pi/zen-large.ts +0 -699
- package/src/pi/zen-metrics.ts +0 -130
- package/src/pi/zen.ts +0 -659
package/README.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
A [Pi](https://pi.dev) extension that turns Pi into a multi-agent software team.
|
|
4
4
|
|
|
5
|
-

|
|
6
6
|
|
|
7
7
|
`/bot-lobby <request>` (or a request typed in the lobby) starts a task, unless
|
|
8
8
|
one agent can simply do it: then it goes to the [quick-fix agent](#quick-fix-or-the-team).
|
|
@@ -16,6 +16,10 @@ The rule: **LLMs decide, the engine enforces.** Agents propose; the
|
|
|
16
16
|
extension validates every state change, permission and approval through one
|
|
17
17
|
`orchestrate` tool.
|
|
18
18
|
|
|
19
|
+
The oracle is meant to be your most capable model; the agents can be smaller
|
|
20
|
+
and cheaper ones. So it never assumes they are as capable as it is: it makes
|
|
21
|
+
every decision itself and [briefs each agent in full](#briefing-the-agents).
|
|
22
|
+
|
|
19
23
|
## Install
|
|
20
24
|
|
|
21
25
|
```bash
|
|
@@ -35,7 +39,7 @@ tool names.
|
|
|
35
39
|
|
|
36
40
|
1. `/bot-lobby add a login page` — starts a task; the lobby opens.
|
|
37
41
|
2. Answer the Master's questions, then approve its proposal.
|
|
38
|
-
3.
|
|
42
|
+
3. Follow the agents in the lobby (`alt+l` shows or hides it): what they do, and what they think.
|
|
39
43
|
|
|
40
44
|
`/bot-lobby settings` sets each agent's model, thinking level, time limit and
|
|
41
45
|
extra instructions.
|
|
@@ -129,6 +133,30 @@ passwords, payments, migrations, production, personal data), or **unclear**
|
|
|
129
133
|
- The track shows in the lobby's activity log, the task's details on the
|
|
130
134
|
Tasks tab, and `/bot-lobby status`.
|
|
131
135
|
|
|
136
|
+
### Briefing the agents
|
|
137
|
+
|
|
138
|
+
Every agent may run on a smaller, cheaper model than the oracle's, one that
|
|
139
|
+
follows instructions well but does not infer intent. So the oracle writes
|
|
140
|
+
each delegation (`implement`, `scout`, `research`) as a self-contained brief
|
|
141
|
+
and settles every design and architecture decision itself first:
|
|
142
|
+
|
|
143
|
+
- **Goal**, the exact **Files**, numbered **What to do** with names, shapes
|
|
144
|
+
and values, the **Contracts** shared with other agents (repeated in full in
|
|
145
|
+
each brief), **Constraints**, **Done when** (checkable criteria and the
|
|
146
|
+
commands to run) and **If stuck**.
|
|
147
|
+
- Plan steps are written to the same standard, so a brief is the plan step
|
|
148
|
+
made explicit, never a new decision.
|
|
149
|
+
- Every agent is told to follow its brief and the approved plan exactly, use
|
|
150
|
+
the given names letter for letter, and report what does not match instead of
|
|
151
|
+
guessing. Workers end their report with a **Brief Check**: each "Done when"
|
|
152
|
+
item, met or not, with evidence.
|
|
153
|
+
- The oracle holds each report to its brief. Drift, a skipped criterion or a
|
|
154
|
+
guessed choice comes back as a fix step with a more explicit brief.
|
|
155
|
+
- The engine backs it up: a delegation that names nothing concrete (or a long
|
|
156
|
+
one with no done criteria) is sent back to the oracle once before any agent
|
|
157
|
+
starts. Sending the same text again goes through, so a short task that is
|
|
158
|
+
complete as written is never stuck. `workflow.briefCheck: false` turns it off.
|
|
159
|
+
|
|
132
160
|
### The full workflow
|
|
133
161
|
|
|
134
162
|
- **Scouts** (read-only) investigate the domains the request touches.
|
|
@@ -188,16 +216,54 @@ everything. `workflow.freshContext: false` in the config turns this off.
|
|
|
188
216
|
## The lobby
|
|
189
217
|
|
|
190
218
|
A full-screen view with a prompt at the bottom that talks to whatever tab is
|
|
191
|
-
open. `alt+h` lists every key.
|
|
219
|
+
open. `alt+h` lists every key. It is text only: no animations, just the
|
|
220
|
+
conversation, the activity log and the thoughts, and a one-line status in Pi's
|
|
221
|
+
footer.
|
|
192
222
|
|
|
193
223
|
| Tab | What it is |
|
|
194
224
|
| --- | --- |
|
|
195
|
-
| **1 Lobby** |
|
|
225
|
+
| **1 Lobby** | Your conversation with the oracle, an activity log of every agent's steps, and each agent's latest thought |
|
|
196
226
|
| **2 Tasks** | Every task and saved plan as a checklist. `s` starts a plan in a new session, `h` here; `c` comments on a plan; `a` archives, `d` deletes |
|
|
197
227
|
| **3 Plan** | Plan a task with a panel of agents before building it (below) |
|
|
198
228
|
| **4 Quick fix** | One agent makes a change right away, beside any running task; requests the oracle [routes here](#quick-fix-or-the-team) show up too |
|
|
199
229
|
| **5 Metrics** | Run time, success rate, tokens and cost per model and agent |
|
|
200
230
|
|
|
231
|
+
### Lobby
|
|
232
|
+
|
|
233
|
+

|
|
234
|
+
|
|
235
|
+
Your conversation with the oracle on the left, the activity log of every
|
|
236
|
+
agent's steps on the right, and the agents' latest thoughts below. `alt+c`,
|
|
237
|
+
`alt+a` and `alt+k` hide any of the three; the prompt steers the running turn.
|
|
238
|
+
|
|
239
|
+
### Tasks
|
|
240
|
+
|
|
241
|
+

|
|
242
|
+
|
|
243
|
+
Every task as a checklist, with its track, plan progress, the approved plan
|
|
244
|
+
and your comments on it.
|
|
245
|
+
|
|
246
|
+
### Plan
|
|
247
|
+
|
|
248
|
+

|
|
249
|
+
|
|
250
|
+
The panel's questions, with recommended options, on the left; the draft plan
|
|
251
|
+
on the right. See [Planning](#planning).
|
|
252
|
+
|
|
253
|
+
### Quick fix
|
|
254
|
+
|
|
255
|
+

|
|
256
|
+
|
|
257
|
+
One agent's jobs, each with its live steps, the files it edited and its
|
|
258
|
+
report. See [Quick fix or the team](#quick-fix-or-the-team).
|
|
259
|
+
|
|
260
|
+
### Metrics
|
|
261
|
+
|
|
262
|
+

|
|
263
|
+
|
|
264
|
+
Run time, success rate, cost and tokens per model and agent, so you can see
|
|
265
|
+
which cheaper models hold up.
|
|
266
|
+
|
|
201
267
|
Common keys: `tab` switches tabs, `esc` browses (arrows, single-key
|
|
202
268
|
commands), `ctrl+f` searches, `ctrl+s` saves the plan, `alt+o` browses
|
|
203
269
|
sessions, `alt+n` starts a task in a new session, `alt+s` opens settings.
|
|
@@ -207,6 +273,25 @@ Rebind any key under `lobby.keys` in the config.
|
|
|
207
273
|
Pi session. The Lobby tab can show any session, and your prompt steers it;
|
|
208
274
|
`● waiting` in the tab bar means one has a question for you.
|
|
209
275
|
|
|
276
|
+
**Agents at work.** The bottom line of the lobby shows the subagents running
|
|
277
|
+
right now at its right end (`◐ DESIGN editing 2m · DEV 40s`, a running quick
|
|
278
|
+
fix too), shrinking to names and then a count when the keys leave little room.
|
|
279
|
+
|
|
280
|
+
**Paging.** A pane with more lines than rows shows page buttons on its bottom
|
|
281
|
+
edge, `▲▲ ▲ ▼ ▼▼`: click one to scroll a page up or down (`▲▲` and `▼▼` go two
|
|
282
|
+
pages). The wheel, the arrows and PageUp/PageDown still work. It applies to
|
|
283
|
+
every scrolling pane: the conversation, activity and thinking, the plan draft,
|
|
284
|
+
the Tasks and Quick fix lists and details, and the metrics table.
|
|
285
|
+
|
|
286
|
+
**Status line when hidden.** With the lobby hidden (`alt+l`), one line under
|
|
287
|
+
Pi's editor shows where things stand: a bar of the task's plan steps (or its
|
|
288
|
+
stage before there is a plan) with who is working, the planning round and the
|
|
289
|
+
questions waiting for you, the quick fix in hand, or `idle`. It costs nothing
|
|
290
|
+
while nothing changes. Turn it off with `lobby.miniLine: false` (or in
|
|
291
|
+
`/bot-lobby settings` → Lobby).
|
|
292
|
+
|
|
293
|
+

|
|
294
|
+
|
|
210
295
|
The conversation keeps its newest 100 messages in memory; scroll to the top
|
|
211
296
|
to load the rest.
|
|
212
297
|
|
|
@@ -241,7 +326,10 @@ can be compared by looking at them.
|
|
|
241
326
|
|
|
242
327
|
`↑↓` move · `enter` choose · `space` pick several (multi-select) · `1`–`4`
|
|
243
328
|
pick · `←→` between questions · the last row takes an answer in your own words
|
|
244
|
-
· `esc`
|
|
329
|
+
· `esc` asks whether to leave (a second `enter`
|
|
330
|
+
leaves, anything else keeps you answering), so a stray press does nothing.
|
|
331
|
+
Questions you leave are never answered for you: the oracle waits and asks again
|
|
332
|
+
when you next write, and the designer asks again before it may decide. Editor hosts that
|
|
245
333
|
run Pi in RPC mode get the same questions through Pi's own dialogs.
|
|
246
334
|
|
|
247
335
|
**Images.** An option can also carry an `image`: a PNG, JPEG, GIF or WebP
|
|
@@ -336,7 +424,7 @@ the result.
|
|
|
336
424
|
"scout": { "model": "anthropic/claude-haiku-4-5-20251001", "timeoutMs": 480000 },
|
|
337
425
|
"planner": { "thinking": "high", "timeoutMs": 300000 },
|
|
338
426
|
"lobby": { "planningPanel": ["backend", "designer", "qa", "researcher"], "maxPlanningRounds": 5 },
|
|
339
|
-
"workflow": { "maxReviewIterations": 2, "maxParallelWorkers": 3, "stallTimeoutMs": 300000, "wrapUpAt": 0.75, "taskBudgetMinutes": 0, "fastTrack": true, "routeQuickFixes": true },
|
|
427
|
+
"workflow": { "maxReviewIterations": 2, "maxParallelWorkers": 3, "stallTimeoutMs": 300000, "wrapUpAt": 0.75, "taskBudgetMinutes": 0, "fastTrack": true, "briefCheck": true, "routeQuickFixes": true },
|
|
340
428
|
"classifier": { "enabled": false, "provider": "auto", "effort": { "cheapModel": "inherit" } }
|
|
341
429
|
}
|
|
342
430
|
```
|
|
@@ -347,9 +435,42 @@ the result.
|
|
|
347
435
|
`off, minimal, low, medium, high, xhigh, max`, limited to what the model
|
|
348
436
|
supports. Scouts always think at `low`.
|
|
349
437
|
- `instructions` adds your own text to an agent's built-in prompt.
|
|
438
|
+
- `fallbackModel` and `fallbackThinking` on any agent (and the master): see
|
|
439
|
+
[Fallback models](#fallback-models).
|
|
350
440
|
- Classifier thresholds and limits (`classifier.thresholds`,
|
|
351
441
|
`classifier.fileHints`) are edited in the file.
|
|
352
442
|
|
|
443
|
+
## Fallback models
|
|
444
|
+
|
|
445
|
+
Running the oracle on a subscription model and the agents on another provider
|
|
446
|
+
means one of them can run out of usage mid-task. Give each agent class a
|
|
447
|
+
**fallback model** and the **thinking level** to run it at (`/bot-lobby
|
|
448
|
+
settings` → the agent → *Fallback model* / *Fallback thinking*). When a run
|
|
449
|
+
fails because its model is out of usage, rate-limited, out of credit or
|
|
450
|
+
unavailable, it runs again on the fallback instead of failing the task.
|
|
451
|
+
|
|
452
|
+
- Works for the master, DESIGN, DEV, QA, the researcher, scouts (their fallback
|
|
453
|
+
thinks at `low` too), quick fixes and the planner and its panel seats.
|
|
454
|
+
- The exhausted model is skipped for 20 minutes, so the next agents go straight
|
|
455
|
+
to their fallback instead of each spending a failed run finding out.
|
|
456
|
+
- **The master** is your own Pi session: on a usage failure the session
|
|
457
|
+
switches to its fallback model and thinking level, tells you, and the oracle
|
|
458
|
+
carries on from where it stopped. Switch back with `/model` when your usage
|
|
459
|
+
returns.
|
|
460
|
+
- Only usage, limit and availability errors switch model; an ordinary failure
|
|
461
|
+
still retries on the same model. If the fallback fails the same way, the run
|
|
462
|
+
fails: it does not chain to a third model.
|
|
463
|
+
- The activity log says when an agent switched, and the run's receipts and the
|
|
464
|
+
Metrics tab show the model that actually ran.
|
|
465
|
+
|
|
466
|
+
```json
|
|
467
|
+
{
|
|
468
|
+
"master": { "model": "anthropic/claude-fable-5-1", "thinking": "high", "fallbackModel": "deepseek/deepseek-v3", "fallbackThinking": "medium" },
|
|
469
|
+
"agents": { "backend": { "model": "zai/glm-4.6", "thinking": "medium", "fallbackModel": "deepseek/deepseek-v3", "fallbackThinking": "low" } },
|
|
470
|
+
"scout": { "model": "zai/glm-4.6", "fallbackModel": "deepseek/deepseek-v3" }
|
|
471
|
+
}
|
|
472
|
+
```
|
|
473
|
+
|
|
353
474
|
## What the engine enforces
|
|
354
475
|
|
|
355
476
|
| Rule | How |
|
package/package.json
CHANGED
package/prompts/global.md
CHANGED
|
@@ -15,6 +15,24 @@ Perform your assigned responsibility precisely and remain within your domain.
|
|
|
15
15
|
- Do not invent requirements; ask when they are genuinely ambiguous.
|
|
16
16
|
- Report conclusions, evidence, decisions, findings and blockers concisely.
|
|
17
17
|
|
|
18
|
+
## Following your brief
|
|
19
|
+
|
|
20
|
+
The Master that instructs you plans the work and knows the whole task; you may
|
|
21
|
+
be a smaller model that sees only your part. Your brief and the approved plan
|
|
22
|
+
are your authority, so:
|
|
23
|
+
|
|
24
|
+
- Read the whole brief and the plan before acting, and do exactly what it
|
|
25
|
+
says, in its order. Do not add features, refactors or "improvements", and do
|
|
26
|
+
not skip steps because they look unnecessary.
|
|
27
|
+
- Use the names, paths, shapes and wording the brief gives you, letter for
|
|
28
|
+
letter. Where it leaves a detail open, choose the simplest option that
|
|
29
|
+
follows the existing code, and say what you chose in your report.
|
|
30
|
+
- Never guess at something that matters (a contract, a file that is not there,
|
|
31
|
+
a conflict between the brief and the code). Stop that part, finish the rest,
|
|
32
|
+
and report it under Blockers or Pushback with what you found.
|
|
33
|
+
- Check your own work against the brief's "Done when" list before you report,
|
|
34
|
+
criterion by criterion, and say honestly which are met and which are not.
|
|
35
|
+
|
|
18
36
|
## Hard rules
|
|
19
37
|
|
|
20
38
|
- Do not add dependencies without approval.
|
package/prompts/master.md
CHANGED
|
@@ -68,6 +68,12 @@ more), skip the researcher when research is not needed, clarify only when it
|
|
|
68
68
|
reads the request as ambiguous — and overrule it whenever the repository says
|
|
69
69
|
otherwise. It is a hint, never a rule.
|
|
70
70
|
|
|
71
|
+
When the user leaves your questions unanswered (they put them away, or
|
|
72
|
+
`ask_user_question` says so), the decision is still theirs: never assume the
|
|
73
|
+
answers, never fall back on the recommended options, and never carry on with
|
|
74
|
+
work that depends on them. Say in one short line that the questions are
|
|
75
|
+
waiting, end your turn, and ask again when they next write.
|
|
76
|
+
|
|
71
77
|
When you `clarify` with options, put your recommended option first and mark
|
|
72
78
|
it `(Recommended)`. When the request already makes it clearly right, the
|
|
73
79
|
classifier answers for you: the reply says so, the decision is recorded, and
|
|
@@ -155,12 +161,81 @@ each `implement` task with its step number (`Step 3: ...`, or `Steps 3-4: ...`
|
|
|
155
161
|
when one delegation covers several) so the user's checklist tracks progress
|
|
156
162
|
exactly.
|
|
157
163
|
|
|
164
|
+
## Briefing the agents
|
|
165
|
+
|
|
166
|
+
You are usually a far more capable model than the agents you delegate to. Scouts,
|
|
167
|
+
workers, the researcher and the reviewer may run on smaller, cheaper models
|
|
168
|
+
that follow instructions well but do not infer intent, fill gaps sensibly or
|
|
169
|
+
know what you know. Never assume they are as capable as you. Whatever you leave
|
|
170
|
+
unsaid, they will guess, and a wrong guess costs a whole agent run. Your plan
|
|
171
|
+
and every brief are how your goal reaches the code, so write them for a
|
|
172
|
+
capable but literal reader who has read nothing but the brief and the
|
|
173
|
+
repository.
|
|
174
|
+
|
|
175
|
+
Every `implement` task (each assignment in a parallel batch), `scout` and
|
|
176
|
+
`research` instruction is a self-contained brief with these parts, in this order:
|
|
177
|
+
|
|
178
|
+
1. **Goal** — the outcome this step must produce and how it serves the user's
|
|
179
|
+
request and the approved plan, in one or two sentences. Name the step
|
|
180
|
+
number(s).
|
|
181
|
+
2. **Files** — the exact paths to create or change, and the ones to leave
|
|
182
|
+
alone. When you do not know a path, say what to search for and where.
|
|
183
|
+
3. **What to do** — numbered, concrete actions in the order to do them: names
|
|
184
|
+
of functions, components, endpoints, fields, types, CSS classes, strings,
|
|
185
|
+
values. Give the exact signature, shape or wording wherever it matters.
|
|
186
|
+
Write "use X", not "use a suitable library"; when a choice is left to the
|
|
187
|
+
agent, say which options are allowed and how to pick.
|
|
188
|
+
4. **Contracts** — everything this step shares with another domain or step:
|
|
189
|
+
API shapes, status codes, error format, event names, shared types, data
|
|
190
|
+
formats, file locations. State them in full in every brief that touches
|
|
191
|
+
them; an agent never sees another agent's brief.
|
|
192
|
+
5. **Constraints** — what it must not do: no new dependencies, no other files,
|
|
193
|
+
no refactors, no changed behavior outside the step, no restyling of code
|
|
194
|
+
it does not own. Repeat the user's explicit requirements that apply.
|
|
195
|
+
6. **Done when** — a checklist of observable, checkable criteria (behaviors,
|
|
196
|
+
exact commands to run and what they should print, files that must exist),
|
|
197
|
+
including what to verify and how, with `timeout`. The agent must be able to
|
|
198
|
+
tell for itself whether it has finished.
|
|
199
|
+
7. **If stuck** — what to do when something does not match the brief (a file is
|
|
200
|
+
missing, a name differs, two instructions conflict): stop that part, do not
|
|
201
|
+
invent a workaround, and report it under Blockers or Pushback with what it
|
|
202
|
+
found. Ask nothing you can answer yourself: settle it in the brief.
|
|
203
|
+
|
|
204
|
+
Rules for the brief:
|
|
205
|
+
|
|
206
|
+
- Decide first, delegate second. Every design, architecture and product
|
|
207
|
+
decision belongs to you; make it and write down the result. A brief must not
|
|
208
|
+
contain "consider", "as appropriate", "if needed", "etc.", "similar to",
|
|
209
|
+
"handle edge cases" or "make it look good" without the specifics. List the
|
|
210
|
+
edge cases; describe the look in concrete terms (layout, sizes, colors,
|
|
211
|
+
states).
|
|
212
|
+
- Say the obvious. Repeat what you already told an earlier agent, include the
|
|
213
|
+
conventions to follow and point to an existing file to imitate by path.
|
|
214
|
+
- One step, one purpose, small enough to hold in mind: a handful of files and
|
|
215
|
+
a few actions. Split anything larger into consecutive steps in the same
|
|
216
|
+
`implement` call rather than leaving the agent to sequence it. Prefer more
|
|
217
|
+
explicit detail to fewer, larger chunks.
|
|
218
|
+
- The plan's steps are written to the same standard: each step names its
|
|
219
|
+
files, its actions and its done criteria, so the brief is the step made
|
|
220
|
+
explicit, never a new decision.
|
|
221
|
+
- Scout and research instructions ask specific questions with the answer
|
|
222
|
+
format you want (paths, names, versions, yes/no plus evidence), and say what
|
|
223
|
+
you will do with the answer.
|
|
224
|
+
|
|
225
|
+
When a report comes back, hold it to the brief: check each **Done when**
|
|
226
|
+
criterion against the report's `## Brief Check`, the diff and the repository.
|
|
227
|
+
Drift, a skipped criterion or a guessed choice is a fix step with a corrected,
|
|
228
|
+
even more explicit brief that quotes the exact gap — not a reason to accept the
|
|
229
|
+
work, and not a reason to redo it yourself. Keep every agent on your plan: if
|
|
230
|
+
the code no longer matches it, say which step it deviates from and restore it.
|
|
231
|
+
|
|
158
232
|
## Speed
|
|
159
233
|
|
|
160
234
|
Every delegation costs a full agent run, so keep the loop short:
|
|
161
235
|
|
|
162
|
-
- Delegate fewer,
|
|
163
|
-
consecutive steps (`Steps 2-4: ...`) rather than one call per step
|
|
236
|
+
- Delegate fewer calls, not vaguer ones: one `implement` per domain covering
|
|
237
|
+
its consecutive steps (`Steps 2-4: ...`) rather than one call per step, each
|
|
238
|
+
step still briefed in full (see Briefing the agents).
|
|
164
239
|
- When steps for different domains are independent, run them together with
|
|
165
240
|
`implement` `assignments` (one entry per domain). Workers then share files
|
|
166
241
|
through the file desk: they claim files, queue for busy ones, and hand them
|
package/prompts/worker.md
CHANGED
|
@@ -19,7 +19,9 @@ them.
|
|
|
19
19
|
|
|
20
20
|
## Implementation
|
|
21
21
|
|
|
22
|
-
- Follow the approved plan
|
|
22
|
+
- Follow the approved plan and your brief exactly: they are the Master's
|
|
23
|
+
decisions. Do not substitute your own design, rename things, or widen the
|
|
24
|
+
step. If the code contradicts the brief, do not improvise: report it.
|
|
23
25
|
- Follow domain boundaries.
|
|
24
26
|
|
|
25
27
|
If you need a new dependency, or you believe a significant architectural
|
|
@@ -73,7 +75,10 @@ repository at the same time, and files are checked out like physical documents:
|
|
|
73
75
|
|
|
74
76
|
## Before handoff
|
|
75
77
|
|
|
76
|
-
-
|
|
78
|
+
- Go through the brief's "Done when" list one criterion at a time and record
|
|
79
|
+
each in `## Brief Check` as `- criterion — met|not met — evidence`.
|
|
80
|
+
- Inspect the actual diff: it must contain what the brief asked for and
|
|
81
|
+
nothing else.
|
|
77
82
|
- Verify tests.
|
|
78
83
|
- Update the temporary task scratchpad.
|
|
79
84
|
- Report concise results.
|
package/src/ask/dialog.ts
CHANGED
|
@@ -9,7 +9,7 @@ import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
|
9
9
|
import { Key, matchesKey, type Component, type TUI } from "@earendil-works/pi-tui";
|
|
10
10
|
import type { LobbyTheme } from "../lobby/layout.ts";
|
|
11
11
|
import { lobbyTheme } from "../lobby/theme.ts";
|
|
12
|
-
import { initialState, step, type AskKey, type AskState } from "./state.ts";
|
|
12
|
+
import { initialState, putAway, step, type AskKey, type AskState } from "./state.ts";
|
|
13
13
|
import { renderAsk, type AskFrame } from "./view.ts";
|
|
14
14
|
import { loadImages } from "./image.ts";
|
|
15
15
|
import { isAbsolute, resolve } from "node:path";
|
|
@@ -68,7 +68,9 @@ export class AskDialog implements Component {
|
|
|
68
68
|
|
|
69
69
|
/** Put the questions away from outside (the turn was aborted). */
|
|
70
70
|
cancel(): void {
|
|
71
|
-
if (
|
|
71
|
+
if (this.state.result) return;
|
|
72
|
+
this.state = putAway(this.state);
|
|
73
|
+
this.done(this.state.result!);
|
|
72
74
|
}
|
|
73
75
|
|
|
74
76
|
render(width: number): string[] {
|
|
@@ -83,7 +85,17 @@ export class AskDialog implements Component {
|
|
|
83
85
|
* cannot draw the questionnaire. Without any UI nobody can answer: the
|
|
84
86
|
* result says the questions were put away.
|
|
85
87
|
*/
|
|
86
|
-
export const askUser: Asker =
|
|
88
|
+
export const askUser: Asker = (questions, ctx, signal, from) => {
|
|
89
|
+
// A model may call the tool several times in one turn; pi runs them in parallel, and overlays opened together hide each other. One at a time.
|
|
90
|
+
const next = asking.then(() => (signal?.aborted ? { answers: [], cancelled: true } : askNow(questions, ctx, signal, from)));
|
|
91
|
+
asking = next.then(() => undefined, () => undefined);
|
|
92
|
+
return next;
|
|
93
|
+
};
|
|
94
|
+
|
|
95
|
+
/** Tail of the questions waiting their turn. */
|
|
96
|
+
let asking: Promise<void> = Promise.resolve();
|
|
97
|
+
|
|
98
|
+
const askNow: Asker = async (questions, ctx, signal, from) => {
|
|
87
99
|
if (!ctx.hasUI || questions.length === 0) return { answers: [], cancelled: true };
|
|
88
100
|
if ((ctx as { mode?: string }).mode === "rpc") return dialogAsker(questions, ctx, signal, from);
|
|
89
101
|
// Option images are read before the questionnaire opens, so drawing it never waits on the disk.
|
package/src/ask/state.ts
CHANGED
|
@@ -38,10 +38,17 @@ export interface AskState {
|
|
|
38
38
|
/** Writing the own answer of the question in view. */
|
|
39
39
|
editing: boolean;
|
|
40
40
|
draft: string;
|
|
41
|
+
/** Esc was pressed: asked whether to leave without answering; enter leaves, any other key keeps answering. */
|
|
42
|
+
leaving?: boolean;
|
|
41
43
|
/** Set once the user submits or puts the questions away. */
|
|
42
44
|
result?: AskResult;
|
|
43
45
|
}
|
|
44
46
|
|
|
47
|
+
/** The questions put away, with what was answered so far (also when something outside ends them). */
|
|
48
|
+
export function putAway(state: AskState): AskState {
|
|
49
|
+
return { ...state, editing: false, draft: "", leaving: false, result: { answers: answersOf(state), cancelled: true } };
|
|
50
|
+
}
|
|
51
|
+
|
|
45
52
|
export function initialState(questions: readonly AskQuestion[]): AskState {
|
|
46
53
|
return {
|
|
47
54
|
questions,
|
|
@@ -127,6 +134,8 @@ function typing(state: AskState, key: AskKey): AskState {
|
|
|
127
134
|
/** One key press. */
|
|
128
135
|
export function step(state: AskState, key: AskKey): AskState {
|
|
129
136
|
if (state.result) return state;
|
|
137
|
+
// Esc asks first: leaving the questions unanswered lets the oracle carry on without you, so it takes a second key.
|
|
138
|
+
if (state.leaving) return key.type === "enter" || (key.type === "text" && key.value.toLowerCase() === "y") ? putAway(state) : { ...state, leaving: false };
|
|
130
139
|
if (state.editing) return typing(state, key);
|
|
131
140
|
const question = state.questions[state.tab];
|
|
132
141
|
if (!question) return { ...state, result: { answers: [], cancelled: false } };
|
|
@@ -153,7 +162,7 @@ export function step(state: AskState, key: AskKey): AskState {
|
|
|
153
162
|
return advance(state);
|
|
154
163
|
}
|
|
155
164
|
case "escape":
|
|
156
|
-
return { ...state,
|
|
165
|
+
return { ...state, leaving: true };
|
|
157
166
|
default:
|
|
158
167
|
return state;
|
|
159
168
|
}
|
package/src/ask/tool.ts
CHANGED
|
@@ -65,12 +65,16 @@ export function invalidQuestions(questions: readonly AskQuestion[]): string | un
|
|
|
65
65
|
return undefined;
|
|
66
66
|
}
|
|
67
67
|
|
|
68
|
+
/** What the model does with questions the user left unanswered: wait, never fill them in. */
|
|
69
|
+
const STILL_OPEN = "Do not assume answers, do not pick the recommended options, and do not go on with work that depends on them. Say in one short line that the questions are waiting, end your turn, and put them to the user again when they next write.";
|
|
70
|
+
const NOT_ANSWERED = `The user left the questions without answering. ${STILL_OPEN}`;
|
|
71
|
+
|
|
68
72
|
/** What the model reads back: each question with its answer, or that it was skipped. */
|
|
69
73
|
export function answerSummary(questions: readonly AskQuestion[], result: AskResult): string {
|
|
70
74
|
if (result.cancelled && result.answers.length === 0) {
|
|
71
75
|
// A relay that could not ask anyone (auto mode) says why.
|
|
72
76
|
if (result.globalNote) return result.globalNote;
|
|
73
|
-
return
|
|
77
|
+
return NOT_ANSWERED;
|
|
74
78
|
}
|
|
75
79
|
const lines = questions.map((question, index) => {
|
|
76
80
|
const answer = result.answers.find((entry) => entry.questionIndex === index);
|
|
@@ -82,6 +86,7 @@ export function answerSummary(questions: readonly AskQuestion[], result: AskResu
|
|
|
82
86
|
return [
|
|
83
87
|
result.cancelled ? "The user answered some questions, then put the rest away:" : "The user answered:",
|
|
84
88
|
...lines,
|
|
89
|
+
...(result.cancelled ? ["", `The questions marked "(not answered)" are still open. ${STILL_OPEN}`] : []),
|
|
85
90
|
...(result.globalNote ? ["", `Note: ${result.globalNote}`] : []),
|
|
86
91
|
].join("\n");
|
|
87
92
|
}
|
package/src/ask/view.ts
CHANGED
|
@@ -112,6 +112,7 @@ function previewBox(preview: Preview, width: number, rows: number, theme: LobbyT
|
|
|
112
112
|
}
|
|
113
113
|
|
|
114
114
|
function hints(state: AskState, question: AskQuestion, width: number, theme?: LobbyTheme): string[] {
|
|
115
|
+
if (state.leaving) return wrap(paint(theme, "warning", "Leave without answering? The oracle will not guess for you. enter leaves · any other key keeps answering"), width);
|
|
115
116
|
const parts = state.editing
|
|
116
117
|
? ["enter keep it", "esc back to the options"]
|
|
117
118
|
: [
|
|
@@ -119,7 +120,7 @@ function hints(state: AskState, question: AskQuestion, width: number, theme?: Lo
|
|
|
119
120
|
question.multiSelect ? "space pick · enter next" : "enter choose",
|
|
120
121
|
`1-${question.options.length} pick`,
|
|
121
122
|
...(state.questions.length > 1 ? ["←→ questions"] : []),
|
|
122
|
-
"esc
|
|
123
|
+
"esc leave",
|
|
123
124
|
];
|
|
124
125
|
return wrap(paint(theme, "dim", parts.join(" · ")), width);
|
|
125
126
|
}
|
|
@@ -7,6 +7,7 @@ import { shortDuration, truncate } from "../text.ts";
|
|
|
7
7
|
import { EditLog } from "../state/changes.ts";
|
|
8
8
|
import { formatMinutes, REPORT_GRACE_MS } from "../state/budget.ts";
|
|
9
9
|
import { runPiAgent, spawnPiProcess, type PiStreamEvent, type ProcessRunner, type RelayAsk } from "./pi-runner.ts";
|
|
10
|
+
import { isUnavailable, looksUnavailable, markUnavailable, usableFallback } from "./fallback.ts";
|
|
10
11
|
import { ASK_ENV } from "../ask/relay.ts";
|
|
11
12
|
import { ASK_TOOL } from "../ask/types.ts";
|
|
12
13
|
|
|
@@ -40,6 +41,10 @@ export interface AgentRequest {
|
|
|
40
41
|
context: AgentContext;
|
|
41
42
|
model?: string;
|
|
42
43
|
thinking?: string;
|
|
44
|
+
/** Where the run goes when its model runs out of usage or is unavailable. */
|
|
45
|
+
fallback?: { model: string; thinking: string };
|
|
46
|
+
/** Set by the runner once it has switched: the model that ran out. */
|
|
47
|
+
fellBackFrom?: string;
|
|
43
48
|
timeoutMs: number;
|
|
44
49
|
cwd: string;
|
|
45
50
|
signal?: AbortSignal;
|
|
@@ -135,13 +140,24 @@ function retryable(run: AgentRun): boolean {
|
|
|
135
140
|
*/
|
|
136
141
|
export async function runAgent(request: AgentRequest, run: ProcessRunner = spawnPiProcess): Promise<AgentRun> {
|
|
137
142
|
const startedAt = new Date().toISOString();
|
|
138
|
-
|
|
143
|
+
let attempts = Math.max(1, (request.retries ?? 0) + 1);
|
|
139
144
|
// A failed attempt may have edited files before the retry: the run owns every edit.
|
|
140
145
|
const edits = new EditLog(request.cwd);
|
|
141
146
|
let last: AgentRun | undefined;
|
|
147
|
+
// A model already known to be out of usage goes straight to the fallback.
|
|
148
|
+
const other = usableFallback(request.fallback, request.model);
|
|
149
|
+
if (other && isUnavailable(request.model)) request = switchToFallback(request, other, request.model ?? "the session model");
|
|
142
150
|
for (let attempt = 1; attempt <= attempts; attempt++) {
|
|
143
151
|
last = await runAgentOnce(request, run, attempt, startedAt, edits);
|
|
144
152
|
request.onAttemptEnd?.(last);
|
|
153
|
+
// Out of usage (or the model unavailable): the same model would fail again, so the fallback takes the retry.
|
|
154
|
+
const fallback = usableFallback(request.fallback, request.model);
|
|
155
|
+
if (fallback && !request.fellBackFrom && last.status === "failed" && looksUnavailable(last.error)) {
|
|
156
|
+
if (request.model) markUnavailable(request.model);
|
|
157
|
+
request = switchToFallback(request, fallback, request.model ?? "the session model");
|
|
158
|
+
attempts += 1;
|
|
159
|
+
continue;
|
|
160
|
+
}
|
|
145
161
|
if (!retryable(last)) break;
|
|
146
162
|
// Under a budget a retry only uses what is left of the allotment.
|
|
147
163
|
if (request.time && request.time.endsAt - Date.now() < MIN_ATTEMPT_MS) break;
|
|
@@ -150,6 +166,11 @@ export async function runAgent(request: AgentRequest, run: ProcessRunner = spawn
|
|
|
150
166
|
return edited.length > 0 ? { ...last!, edited } : last!;
|
|
151
167
|
}
|
|
152
168
|
|
|
169
|
+
/** The request moved to its fallback model and thinking level. */
|
|
170
|
+
function switchToFallback(request: AgentRequest, fallback: { model: string; thinking?: string }, from: string): AgentRequest {
|
|
171
|
+
return { ...request, model: fallback.model, thinking: fallback.thinking ?? request.thinking, fellBackFrom: from };
|
|
172
|
+
}
|
|
173
|
+
|
|
153
174
|
/** Abort every in-flight subagent (session shutdown, user cancel). */
|
|
154
175
|
export function cancelAllRuns(): void {
|
|
155
176
|
for (const controller of activeControllers) controller.abort();
|
|
@@ -168,6 +189,7 @@ function baseRun(request: AgentRequest, runId: string, startedAt: string, attemp
|
|
|
168
189
|
attempts,
|
|
169
190
|
startedAt,
|
|
170
191
|
...(request.thinking ? { thinking: request.thinking } : {}),
|
|
192
|
+
...(request.fellBackFrom ? { fellBackFrom: request.fellBackFrom } : {}),
|
|
171
193
|
...(request.routedFrom ? { routedFrom: request.routedFrom } : {}),
|
|
172
194
|
...(request.route ? { route: request.route } : {}),
|
|
173
195
|
...(request.time ? { allotMs: request.time.allotMs, endsAt: request.time.endsAt } : {}),
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Fallback models. A subscription or key that runs out mid-task (a usage
|
|
3
|
+
* limit, a rate limit, no credit, the provider down or refusing the key)
|
|
4
|
+
* fails every run on that model the same way, so the run goes to the agent's
|
|
5
|
+
* configured fallback model instead, and the exhausted model is skipped for a
|
|
6
|
+
* while so the next agents do not each spend a failed run finding out.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
/** What a provider says when a model cannot be used right now. Deliberately narrow: an ordinary failure is not a reason to switch model. */
|
|
10
|
+
const UNAVAILABLE = [
|
|
11
|
+
/usage limit/i, /rate.?limit/i, /\bquota\b/i, /limit (?:reached|exceeded|will reset)/i, /exceeded (?:your|the|current)/i,
|
|
12
|
+
/out of (?:usage|credits?|extra usage|tokens)/i, /(?:insufficient|no) (?:credits?|funds|balance|quota)/i, /credit balance/i,
|
|
13
|
+
/billing/i, /payment required/i, /\b402\b/, /\b429\b/, /too many requests/i, /resource.?exhausted/i,
|
|
14
|
+
/overloaded/i, /\b529\b/, /(?:at|over) capacity/i, /resets? (?:at|in|on)/i, /subscription/i,
|
|
15
|
+
/unauthori[sz]ed/i, /\b401\b/, /invalid (?:api )?key/i, /no api key/i, /authentication (?:failed|error)/i,
|
|
16
|
+
/model (?:is )?(?:not found|not available|unavailable)/i, /unknown model/i, /does not have access/i,
|
|
17
|
+
];
|
|
18
|
+
|
|
19
|
+
/** True when a failed run's error says its model is out of usage or unavailable, so another model may succeed. */
|
|
20
|
+
export function looksUnavailable(error: string | undefined): boolean {
|
|
21
|
+
return Boolean(error && UNAVAILABLE.some((pattern) => pattern.test(error)));
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
/** How long a model that ran out is skipped before it is tried again. */
|
|
25
|
+
export const COOLDOWN_MS = 20 * 60 * 1000;
|
|
26
|
+
|
|
27
|
+
const exhausted = new Map<string, number>();
|
|
28
|
+
|
|
29
|
+
/** Remember that `model` cannot be used until the cooldown passes. */
|
|
30
|
+
export function markUnavailable(model: string, now = Date.now()): void {
|
|
31
|
+
exhausted.set(model, now + COOLDOWN_MS);
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
/** Whether `model` ran out recently enough to skip. */
|
|
35
|
+
export function isUnavailable(model: string | undefined, now = Date.now()): boolean {
|
|
36
|
+
if (!model) return false;
|
|
37
|
+
const until = exhausted.get(model);
|
|
38
|
+
if (until === undefined) return false;
|
|
39
|
+
if (until <= now) {
|
|
40
|
+
exhausted.delete(model);
|
|
41
|
+
return false;
|
|
42
|
+
}
|
|
43
|
+
return true;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/** Forget every exhausted model (tests). */
|
|
47
|
+
export function resetUnavailable(): void {
|
|
48
|
+
exhausted.clear();
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** A fallback worth switching to: a model other than the one that failed. */
|
|
52
|
+
export function usableFallback<F extends { model: string }>(fallback: F | undefined, current: string | undefined): F | undefined {
|
|
53
|
+
return fallback && fallback.model !== current ? fallback : undefined;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Run `attempt` on `model`; when it fails because that model is out of usage
|
|
58
|
+
* or unavailable, run it again on the fallback. A model already known to be
|
|
59
|
+
* out goes straight to the fallback. `switched` says which model took over.
|
|
60
|
+
*/
|
|
61
|
+
export async function withFallback<T extends { status: string; error?: string }>(
|
|
62
|
+
model: string | undefined,
|
|
63
|
+
thinking: string,
|
|
64
|
+
fallback: { model: string; thinking: string } | undefined,
|
|
65
|
+
attempt: (model: string | undefined, thinking: string) => Promise<T>,
|
|
66
|
+
): Promise<{ result: T; switchedFrom?: string }> {
|
|
67
|
+
const other = usableFallback(fallback, model);
|
|
68
|
+
if (other && isUnavailable(model)) return { result: await attempt(other.model, other.thinking), switchedFrom: model ?? "the session model" };
|
|
69
|
+
const result = await attempt(model, thinking);
|
|
70
|
+
if (other && result.status === "failed" && looksUnavailable(result.error)) {
|
|
71
|
+
if (model) markUnavailable(model);
|
|
72
|
+
return { result: await attempt(other.model, other.thinking), switchedFrom: model ?? "the session model" };
|
|
73
|
+
}
|
|
74
|
+
return { result };
|
|
75
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -21,7 +21,7 @@ export default function (pi: ExtensionAPI): void {
|
|
|
21
21
|
// Before the lifecycle, so a task that ended while the oracle was idle is closed before it builds the next turn's prompt.
|
|
22
22
|
registerFreshContext(pi, CONFIG_DIR_NAME);
|
|
23
23
|
registerLifecycle(pi, CONFIG_DIR_NAME);
|
|
24
|
-
// After the lifecycle, so the lobby opens over a task the
|
|
24
|
+
// After the lifecycle, so the lobby opens over a task the status already knows.
|
|
25
25
|
registerLobbyEvents(pi, CONFIG_DIR_NAME);
|
|
26
26
|
registerOwner(pi, CONFIG_DIR_NAME);
|
|
27
27
|
onTransition((task) => pingTransition(task));
|