@a-t-h-i/bot-lobby 0.6.4 → 0.6.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +150 -12
  2. package/package.json +12 -3
  3. package/prompts/backend.md +3 -1
  4. package/prompts/designer.md +24 -1
  5. package/prompts/global.md +18 -0
  6. package/prompts/master.md +122 -20
  7. package/prompts/quickfix.md +16 -3
  8. package/prompts/worker.md +7 -2
  9. package/src/agents/backend.ts +2 -2
  10. package/src/agents/designer.ts +2 -2
  11. package/src/ask/dialog.ts +11 -1
  12. package/src/classifier/triage.ts +24 -7
  13. package/src/index.ts +4 -1
  14. package/src/lobby/feed.ts +7 -2
  15. package/src/lobby/keys.ts +0 -1
  16. package/src/lobby/quickfix.ts +29 -3
  17. package/src/lobby/runtime.ts +29 -26
  18. package/src/lobby/tabs/home.ts +6 -42
  19. package/src/lobby/tabs/quickfix.ts +1 -1
  20. package/src/lobby/tabs/tasks.ts +3 -2
  21. package/src/lobby/view.ts +18 -23
  22. package/src/master/decisions.ts +10 -2
  23. package/src/pi/activity.ts +1 -33
  24. package/src/pi/commands.ts +22 -11
  25. package/src/pi/events.ts +3 -17
  26. package/src/pi/plan-checklist.ts +305 -0
  27. package/src/pi/route.ts +180 -0
  28. package/src/pi/run-summary.ts +12 -3
  29. package/src/pi/settings-ui.ts +2 -2
  30. package/src/pi/start-task.ts +51 -5
  31. package/src/pi/tools.ts +13 -7
  32. package/src/pi/ui.ts +18 -284
  33. package/src/roles/worker.ts +11 -3
  34. package/src/schemas/configuration.ts +19 -6
  35. package/src/schemas/task.ts +31 -0
  36. package/src/workflow/brief.ts +59 -0
  37. package/src/workflow/track.ts +436 -0
  38. package/src/workflow/workflow.ts +151 -10
  39. package/src/pi/expressions.ts +0 -169
  40. package/src/pi/kaomoji.ts +0 -227
  41. package/src/pi/mascot-art.ts +0 -359
  42. package/src/pi/zen-large.ts +0 -699
  43. package/src/pi/zen-metrics.ts +0 -130
  44. package/src/pi/zen.ts +0 -659
package/README.md CHANGED
@@ -2,7 +2,11 @@
2
2
 
3
3
  A [Pi](https://pi.dev) extension that turns Pi into a multi-agent software team.
4
4
 
5
- `/bot-lobby <request>` starts a task. Your Pi session becomes the **Master**
5
+ ![The Lobby tab: your conversation with the oracle, the activity log of every agent, and their latest thoughts](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/gallery.png)
6
+
7
+ `/bot-lobby <request>` (or a request typed in the lobby) starts a task, unless
8
+ one agent can simply do it: then it goes to the [quick-fix agent](#quick-fix-or-the-team).
9
+ For a task, your Pi session becomes the **Master**
6
10
  (the "oracle"): it scouts the codebase, proposes a plan, and delegates the
7
11
  work to three domain agents — **Designer+Frontend**, **Backend** and **QA** —
8
12
  each running in its own isolated `pi` process. QA's reviewer is the quality
@@ -12,6 +16,10 @@ The rule: **LLMs decide, the engine enforces.** Agents propose; the
12
16
  extension validates every state change, permission and approval through one
13
17
  `orchestrate` tool.
14
18
 
19
+ The oracle is meant to be your most capable model; the agents can be smaller
20
+ and cheaper ones. So it never assumes they are as capable as it is: it makes
21
+ every decision itself and [briefs each agent in full](#briefing-the-agents).
22
+
15
23
  ## Install
16
24
 
17
25
  ```bash
@@ -31,7 +39,7 @@ tool names.
31
39
 
32
40
  1. `/bot-lobby add a login page` — starts a task; the lobby opens.
33
41
  2. Answer the Master's questions, then approve its proposal.
34
- 3. Watch the agents work in the lobby (`alt+l` shows or hides it).
42
+ 3. Follow the agents in the lobby (`alt+l` shows or hides it): what they do, and what they think.
35
43
 
36
44
  `/bot-lobby settings` sets each agent's model, thinking level, time limit and
37
45
  extra instructions.
@@ -41,7 +49,7 @@ extra instructions.
41
49
  | Command | Does |
42
50
  | --- | --- |
43
51
  | `/bot-lobby` | Open the lobby (`alt+l`) |
44
- | `/bot-lobby <request>` | Start a task (`--task` if it begins with a command word, `--auto` to run unattended, `--budget 90m` to give it a time budget) |
52
+ | `/bot-lobby <request>` | Start a request: a [quick fix](#quick-fix-or-the-team) when one agent can do it alone, else a task (`--task` to always make it a task, also when it begins with a command word; `--auto` to run unattended, `--budget 90m` to give it a time budget, `--fast` / `--full` to pick its [track](#fast-track-or-full-workflow)) |
45
53
  | `/bot-lobby budget [90m\|off]` | Show or set this session's task time budget |
46
54
  | `/bot-lobby status \| tasks \| runs [id]` | Current task, all tasks, recent agent runs |
47
55
  | `/bot-lobby approve \| amend <text> \| decline` | Answer the proposal |
@@ -54,15 +62,106 @@ extra instructions.
54
62
  | `/bot-lobby knowledge` | Knowledge file sizes |
55
63
  | `/bot-lobby minimize \| restore` | Hide bot-lobby in this session (`ctrl+shift+m`) |
56
64
 
65
+ ## Quick fix or the team
66
+
67
+ Before any task exists, bot-lobby asks whether **one agent can just do it**:
68
+ in one file or one area, with nothing to agree between frontend and backend,
69
+ no unfamiliar codebase to survey, nothing risky and no decision you must make
70
+ first. [Jev](#the-classifier-jev) answers when it is on (`classifier.thresholds.quickFixAt`,
71
+ 0.7); plain rules answer otherwise, and whenever Jev is unsure. A request that
72
+ says it is self-contained ("a single page", "in one html file") counts even
73
+ when it is rich.
74
+
75
+ When it reads that way, the oracle confirms in one step, without reading
76
+ files or planning (`route_request`), and says so: *this looks like a quick
77
+ feature, the quick-fix agent is on it*. The lobby then hands the request to
78
+ the quick-fix agent and switches to the **Quick fix** tab, where you follow
79
+ it. No scouts, proposal, plan or QA. A quick feature (bigger than a small
80
+ change, but in one place) runs on its builder's model, thinking and time limit
81
+ (DESIGN's for a page) instead of the quick-fix defaults.
82
+
83
+ Everything else, or anything the oracle judges needs the team, starts as a
84
+ task below. `--task` always makes a task; `workflow.routeQuickFixes: false`
85
+ turns routing off. Without the lobby (a background session, RPC mode) every
86
+ request is a task.
87
+
57
88
  ## How a task runs
58
89
 
59
90
  ```
60
- request → clarify → scout → propose → approve → plan → implement → QA gate → complete
91
+ full workflow: request → clarify → scout → propose → approve → plan → implement → QA gate → complete
92
+ fast track: request → implement (only the agents it needs) → QA, if it needs tests → complete
61
93
  ```
62
94
 
95
+ ### Fast track or full workflow
96
+
97
+ The moment a task starts, bot-lobby reads the request and decides how serious
98
+ it is: its size, who has to take part, and whether it takes the **fast
99
+ track** or the **full workflow**. The read is instant and costs no tokens
100
+ (plain rules, refined by [Jev](#the-classifier-jev) when that is on).
101
+
102
+ | The request… | Who takes part |
103
+ | --- | --- |
104
+ | changes anything that runs in the browser: screens, components, styles, copy, canvas or three.js graphics | DESIGN (frontend) |
105
+ | changes an API, the database, auth, jobs or other server-side code | DEV (backend) |
106
+ | needs tests: asks for them, fixes a bug, or changes backend logic | QA |
107
+ | depends on outside facts: latest versions, docs, standards, third-party APIs | RESEARCH |
108
+
109
+ A **small, clear, low-risk** request takes the fast track: the oracle hands
110
+ it straight to those agents, with no scouts, no proposal to approve and no
111
+ plan document (the engine keeps a short plan whose steps are the
112
+ delegations, so the checklist still works). QA takes part only when the
113
+ change needs tests (its worker writing and running them as the last step, or
114
+ the QA gate); without it the task completes as soon as the work is in and
115
+ checked. A copy change is one agent run.
116
+
117
+ Everything else takes the full workflow below: anything **medium or large**
118
+ (a new page, flow or endpoint, a refactor, an upgrade, a vague or
119
+ many-part request), **serious** whatever its size (security, auth,
120
+ passwords, payments, migrations, production, personal data), or **unclear**
121
+ (too short, vague, or asking to investigate first).
122
+
123
+ - The oracle glances at the read once and acts on it, or corrects it with
124
+ `orchestrate action=track`: the full workflow for a change bigger than it
125
+ reads, the fast track for one that is smaller (before its work is planned),
126
+ or a member added or dropped.
127
+ - Once work is under way a track only gets stricter: the full workflow or
128
+ more members, never QA dropped. A fast task whose worker asks for a new
129
+ dependency or an architecture change gets QA.
130
+ - `/bot-lobby --fast <request>` or `--full <request>` decides it yourself;
131
+ the oracle never moves a `--full` task to the fast track.
132
+ `workflow.fastTrack: false` puts every task on the full workflow.
133
+ - The track shows in the lobby's activity log, the task's details on the
134
+ Tasks tab, and `/bot-lobby status`.
135
+
136
+ ### Briefing the agents
137
+
138
+ Every agent may run on a smaller, cheaper model than the oracle's, one that
139
+ follows instructions well but does not infer intent. So the oracle writes
140
+ each delegation (`implement`, `scout`, `research`) as a self-contained brief
141
+ and settles every design and architecture decision itself first:
142
+
143
+ - **Goal**, the exact **Files**, numbered **What to do** with names, shapes
144
+ and values, the **Contracts** shared with other agents (repeated in full in
145
+ each brief), **Constraints**, **Done when** (checkable criteria and the
146
+ commands to run) and **If stuck**.
147
+ - Plan steps are written to the same standard, so a brief is the plan step
148
+ made explicit, never a new decision.
149
+ - Every agent is told to follow its brief and the approved plan exactly, use
150
+ the given names letter for letter, and report what does not match instead of
151
+ guessing. Workers end their report with a **Brief Check**: each "Done when"
152
+ item, met or not, with evidence.
153
+ - The oracle holds each report to its brief. Drift, a skipped criterion or a
154
+ guessed choice comes back as a fix step with a more explicit brief.
155
+ - The engine backs it up: a delegation that names nothing concrete (or a long
156
+ one with no done criteria) is sent back to the oracle once before any agent
157
+ starts. Sending the same text again goes through, so a short task that is
158
+ complete as written is never stuck. `workflow.briefCheck: false` turns it off.
159
+
160
+ ### The full workflow
161
+
63
162
  - **Scouts** (read-only) investigate the domains the request touches.
64
163
  - The Master **proposes** a short bullet list; nothing is built until you
65
- approve it. Small single-domain changes may skip scouting and the proposal.
164
+ approve it.
66
165
  - **Workers** implement plan steps, one domain each. Several can run in
67
166
  parallel; they share files through a **file desk** (claim a file, queue for
68
167
  a busy one, hand it over with a note).
@@ -117,16 +216,54 @@ everything. `workflow.freshContext: false` in the config turns this off.
117
216
  ## The lobby
118
217
 
119
218
  A full-screen view with a prompt at the bottom that talks to whatever tab is
120
- open. `alt+h` lists every key.
219
+ open. `alt+h` lists every key. It is text only: no animations, just the
220
+ conversation, the activity log and the thoughts, and a one-line status in Pi's
221
+ footer.
121
222
 
122
223
  | Tab | What it is |
123
224
  | --- | --- |
124
- | **1 Lobby** | The task's status, your conversation with the oracle, an activity log of every tool call, and each agent's latest thought |
225
+ | **1 Lobby** | Your conversation with the oracle, an activity log of every agent's steps, and each agent's latest thought |
125
226
  | **2 Tasks** | Every task and saved plan as a checklist. `s` starts a plan in a new session, `h` here; `c` comments on a plan; `a` archives, `d` deletes |
126
227
  | **3 Plan** | Plan a task with a panel of agents before building it (below) |
127
- | **4 Quick fix** | One agent makes a small change right away, beside any running task |
228
+ | **4 Quick fix** | One agent makes a change right away, beside any running task; requests the oracle [routes here](#quick-fix-or-the-team) show up too |
128
229
  | **5 Metrics** | Run time, success rate, tokens and cost per model and agent |
129
230
 
231
+ ### Lobby
232
+
233
+ ![The Lobby tab](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/gallery.png)
234
+
235
+ Your conversation with the oracle on the left, the activity log of every
236
+ agent's steps on the right, and the agents' latest thoughts below. `alt+c`,
237
+ `alt+a` and `alt+k` hide any of the three; the prompt steers the running turn.
238
+
239
+ ### Tasks
240
+
241
+ ![The Tasks tab](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/lobby-tasks.png)
242
+
243
+ Every task as a checklist, with its track, plan progress, the approved plan
244
+ and your comments on it.
245
+
246
+ ### Plan
247
+
248
+ ![The Plan tab](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/lobby-plan.png)
249
+
250
+ The panel's questions, with recommended options, on the left; the draft plan
251
+ on the right. See [Planning](#planning).
252
+
253
+ ### Quick fix
254
+
255
+ ![The Quick fix tab](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/lobby-quickfix.png)
256
+
257
+ One agent's jobs, each with its live steps, the files it edited and its
258
+ report. See [Quick fix or the team](#quick-fix-or-the-team).
259
+
260
+ ### Metrics
261
+
262
+ ![The Metrics tab](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/lobby-metrics.png)
263
+
264
+ Run time, success rate, cost and tokens per model and agent, so you can see
265
+ which cheaper models hold up.
266
+
130
267
  Common keys: `tab` switches tabs, `esc` browses (arrows, single-key
131
268
  commands), `ctrl+f` searches, `ctrl+s` saves the plan, `alt+o` browses
132
269
  sessions, `alt+n` starts a task in a new session, `alt+s` opens settings.
@@ -236,7 +373,8 @@ else TypeSafe.
236
373
  | Planning seats | Each round, only the seats the idea or your latest answers touch sit; `1`–`4` pins a seat |
237
374
  | Obvious answers | Answers a question itself when the conversation already makes the recommended option clearly right (≥ 0.9); listed under Assumptions |
238
375
  | File hints | Agents start with a short list of the files they most likely need, and get a `find_relevant_files` tool |
239
- | Task triage | The Master gets hints (size, domains, research needed); a quick fix that is really a task is held (`r` run anyway, `t` make it a task) |
376
+ | Quick fix or task | Whether one engineer can do a new request alone decides whether it goes to the [quick-fix agent](#quick-fix-or-the-team) (the oracle confirms) |
377
+ | Task triage | The task's [track](#fast-track-or-full-workflow) and roster use its read (size, domains, research, ambiguity), and the Master gets it as hints; a quick fix that is really a task (large, and not one engineer's work) is held (`r` run anyway, `t` make it a task) |
240
378
  | Effort routing | Simple steps run one thinking level lower; trivial ones on a **cheaper model** you pick. A routed run that falls short re-runs on your normal settings |
241
379
 
242
380
  **It never gets in the way:** any failure, timeout or missing key means
@@ -264,7 +402,7 @@ the result.
264
402
  "scout": { "model": "anthropic/claude-haiku-4-5-20251001", "timeoutMs": 480000 },
265
403
  "planner": { "thinking": "high", "timeoutMs": 300000 },
266
404
  "lobby": { "planningPanel": ["backend", "designer", "qa", "researcher"], "maxPlanningRounds": 5 },
267
- "workflow": { "maxReviewIterations": 2, "maxParallelWorkers": 3, "stallTimeoutMs": 300000, "wrapUpAt": 0.75, "taskBudgetMinutes": 0 },
405
+ "workflow": { "maxReviewIterations": 2, "maxParallelWorkers": 3, "stallTimeoutMs": 300000, "wrapUpAt": 0.75, "taskBudgetMinutes": 0, "fastTrack": true, "briefCheck": true, "routeQuickFixes": true },
268
406
  "classifier": { "enabled": false, "provider": "auto", "effort": { "cheapModel": "inherit" } }
269
407
  }
270
408
  ```
@@ -283,11 +421,11 @@ the result.
283
421
  | Rule | How |
284
422
  | --- | --- |
285
423
  | Steps happen in order | A state machine validates every action |
286
- | Nothing is built before approval | `implement` refuses earlier states |
424
+ | Nothing is built before approval | On the full workflow `implement` refuses earlier states; only a fast-track task (small, clear, low-risk) starts straight away |
287
425
  | Scouts and the QA gate can't edit code | Scouts get read-only tools; the QA gate adds only `bash` for tests |
288
426
  | New dependencies and architecture changes need approval | Parsed from worker reports; the domain is blocked until resolved |
289
427
  | Only the Master writes knowledge | Agents can only propose it |
290
- | "Done" is earned | Needs a plan, a passing QA gate that ran checks (or your explicit acceptance), and no open blockers |
428
+ | "Done" is earned | Needs a plan, a passing QA gate that ran checks (or your explicit acceptance), and no open blockers; on the fast track, a finished worker step, and QA's part only when the change needs tests |
291
429
  | Parallel workers don't clobber files | Edits need a file-desk claim |
292
430
  | A crash doesn't corrupt a task | State is on disk; tasks resume from their state |
293
431
 
package/package.json CHANGED
@@ -1,11 +1,19 @@
1
1
  {
2
2
  "name": "@a-t-h-i/bot-lobby",
3
- "version": "0.6.4",
3
+ "version": "0.6.6",
4
4
  "description": "Structured multi-agent software engineering orchestrator for Pi",
5
5
  "type": "module",
6
6
  "license": "Apache-2.0",
7
7
  "keywords": [
8
- "pi-package"
8
+ "pi-package",
9
+ "pi",
10
+ "pi-extension",
11
+ "pi-coding-agent",
12
+ "multi-agent",
13
+ "subagents",
14
+ "orchestrator",
15
+ "code-review",
16
+ "web-search"
9
17
  ],
10
18
  "repository": {
11
19
  "type": "git",
@@ -25,7 +33,8 @@
25
33
  "pi": {
26
34
  "extensions": [
27
35
  "./src/index.ts"
28
- ]
36
+ ],
37
+ "image": "https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/gallery.png"
29
38
  },
30
39
  "scripts": {
31
40
  "typecheck": "tsc --noEmit",
@@ -2,7 +2,9 @@
2
2
 
3
3
  You own backend engineering: API, business logic, data models, database,
4
4
  authentication, authorization, integrations, backend architecture, security,
5
- reliability and backend performance.
5
+ reliability and backend performance. Code that runs in the browser — pages,
6
+ client-side logic, canvas and WebGL/three.js graphics — is the designer's,
7
+ however much logic it holds.
6
8
 
7
9
  ## API design
8
10
 
@@ -2,7 +2,9 @@
2
2
 
3
3
  You own UI/UX and frontend engineering: user experience, interaction design,
4
4
  visual consistency, frontend implementation, responsive behavior,
5
- accessibility, frontend performance and the design language.
5
+ accessibility, frontend performance and the design language. Everything that
6
+ runs in the browser is yours, including client-side logic and canvas,
7
+ WebGL/three.js graphics.
6
8
 
7
9
  You have real visual taste. Your work is calm, considered and quietly
8
10
  delightful: the understated, crafted aesthetic of Anthropic's latest models,
@@ -66,6 +68,27 @@ one-off values.
66
68
  - **No magic numbers:** every size, space, color, radius, shadow and duration
67
69
  comes from a token or the existing scale.
68
70
 
71
+ ## Graphics and 3D
72
+
73
+ When the work is a scene (canvas, WebGL, three.js), how it looks is the
74
+ product, and a generic render is a failed one.
75
+
76
+ - **Light it like a photograph:** image-based lighting (an environment map,
77
+ e.g. PMREM with `RoomEnvironment`) plus one key light with soft shadows;
78
+ ACES or AgX tone mapping and sRGB output. Never flat ambient light.
79
+ - **Physical materials:** `MeshPhysicalMaterial` with values from the real
80
+ thing — glass with transmission, thickness, IOR about 1.5 and a faint tint;
81
+ wood, brass or stone with sensible roughness and metalness. Liquids and
82
+ grains read by their own color, sheen and scale, not by a texture.
83
+ - **Proportions and framing:** model the object after a real reference; frame
84
+ it so it fills most of the view at rest, on every screen size, with a quiet
85
+ backdrop (a soft gradient, a floor that fades out). No default grey planes,
86
+ giant ground discs or horizon lines cutting through the shot.
87
+ - **Restraint:** fewer effects done well beat many done badly; hold 60fps on a
88
+ phone with quality tiers rather than dropping detail everywhere.
89
+ - **Look at it:** render a frame when the project has a headless browser, and
90
+ fix what you see; otherwise say in your report that it was not seen.
91
+
69
92
  ## Interaction and motion
70
93
 
71
94
  You like interactive UIs that give the user subtle, fun feedback — never
package/prompts/global.md CHANGED
@@ -15,6 +15,24 @@ Perform your assigned responsibility precisely and remain within your domain.
15
15
  - Do not invent requirements; ask when they are genuinely ambiguous.
16
16
  - Report conclusions, evidence, decisions, findings and blockers concisely.
17
17
 
18
+ ## Following your brief
19
+
20
+ The Master that instructs you plans the work and knows the whole task; you may
21
+ be a smaller model that sees only your part. Your brief and the approved plan
22
+ are your authority, so:
23
+
24
+ - Read the whole brief and the plan before acting, and do exactly what it
25
+ says, in its order. Do not add features, refactors or "improvements", and do
26
+ not skip steps because they look unnecessary.
27
+ - Use the names, paths, shapes and wording the brief gives you, letter for
28
+ letter. Where it leaves a detail open, choose the simplest option that
29
+ follows the existing code, and say what you chose in your report.
30
+ - Never guess at something that matters (a contract, a file that is not there,
31
+ a conflict between the brief and the code). Stop that part, finish the rest,
32
+ and report it under Blockers or Pushback with what you found.
33
+ - Check your own work against the brief's "Done when" list before you report,
34
+ criterion by criterion, and say honestly which are met and which are not.
35
+
18
36
  ## Hard rules
19
37
 
20
38
  - Do not add dependencies without approval.
package/prompts/master.md CHANGED
@@ -17,32 +17,55 @@ validates every step — state transitions, role permissions, approval gates and
17
17
  completion authority. If it rejects an action, read the error and adjust; never
18
18
  work around it. Do not rely on prompts to enforce permissions or state.
19
19
 
20
+ ## Task track
21
+
22
+ Every task starts on a track the engine read from the request (the `Track`
23
+ line in your context): how serious it is, and so who takes part and how much
24
+ process it gets. Fewer steps win whenever the result is the same.
25
+
26
+ - **Fast track** — a small, clear, low-risk change. No scouts, no proposal,
27
+ no plan document: delegate straight away with `orchestrate action=implement`,
28
+ opening each task with `Step N:` (the engine keeps the plan and the
29
+ checklist). Only the roster takes part: DESIGN for frontend work, DEV for
30
+ backend work, QA when the change needs tests (its worker writing and
31
+ running them as the last step, or the QA gate), and the researcher when a
32
+ decision needs outside facts (summon it first). Several domains: one
33
+ `implement` with `assignments`, each task stating the contract between
34
+ them. When the work is in, check `git diff --stat` and the report, then
35
+ `complete`; without QA on the roster there is no QA gate.
36
+ - **Full workflow** — everything else: the steps below, ending with the QA
37
+ gate.
38
+
39
+ The read is quick and can be wrong, so glance at it once and move on: confirm
40
+ it by acting on it, or correct it with `orchestrate action=track` (`track`,
41
+ `roster`, `reason`). Go full when the change is bigger, riskier (security,
42
+ money, data, migrations, production) or less clear than it reads; take the
43
+ fast track when a full-workflow task turns out small and clear (before its
44
+ work is planned). Add a member the roster is missing, or drop one it does not
45
+ need, before delegating. Once work is under way a track only gets stricter:
46
+ the full workflow or more members, never QA dropped; a fast task whose worker
47
+ asks for a dependency or an architecture change gets QA added. The user's
48
+ `--full` and a disabled fast track keep the full workflow.
49
+
20
50
  ## Before implementation
21
51
 
22
- For feature-level work: understand the request; clarify with
52
+ For full-workflow work: understand the request; clarify with
23
53
  `orchestrate action=clarify` when necessary; challenge it when there is a real
24
54
  technical, security, reliability, UX or maintainability concern; select and run
25
55
  relevant Scouts; review findings and target-verify important claims against the
26
56
  repository; synthesize and present a short `- ` bullet-list proposal; then wait for
27
- approval, amendment, or decline. Do not start feature implementation before
28
- approval.
29
-
30
- For a trivial, single-domain request you may skip the Scout round and the
31
- proposal ceremony: state the short plan, delegate the step, and verify the diff
32
- directly. The engine allows `clarifying -> awaiting_approval -> planning`, so no
33
- state override is needed. Skip only when the change is small, obvious and
34
- confined to one domain.
57
+ approval, amendment, or decline. Do not start full-workflow implementation
58
+ before approval.
35
59
 
36
60
  ## Classifier hints
37
61
 
38
62
  When the classifier is on, your task context carries a **Classifier
39
63
  triage**: a fast model's read of the request (size, the domains it touches,
40
64
  whether it needs outside research, whether it is ambiguous, its kind, likely
41
- files) and a suggested path. Use it to skip reasoning you do not need —
42
- scout only the domains it marks (0.5 or more), skip scouting when the task
43
- is trivial or small in one domain and likely files are named, skip the
44
- researcher when research is not needed, clarify only when it reads the
45
- request as ambiguous — and overrule it whenever the repository says
65
+ files) and a suggested path; the task's track was chosen with it. Use it to
66
+ skip reasoning you do not need — scout only the domains it marks (0.5 or
67
+ more), skip the researcher when research is not needed, clarify only when it
68
+ reads the request as ambiguous — and overrule it whenever the repository says
46
69
  otherwise. It is a hint, never a rule.
47
70
 
48
71
  When you `clarify` with options, put your recommended option first and mark
@@ -119,17 +142,94 @@ Assign work to the correct domain; never ask one domain to do another's. A
119
142
  cross-domain dependency is reported to you, and you decide whether another
120
143
  domain needs a task.
121
144
 
145
+ Assign by where the code runs, not by how much logic it holds. Everything that
146
+ runs in the browser — pages, components, client-side state and logic, canvas,
147
+ WebGL/three.js scenes, shaders, client-side physics — is DESIGN's
148
+ (Designer+Frontend); DEV (backend) owns server-side code, data, APIs and
149
+ integrations. One file has one owner: never split a file between domains, and
150
+ never give DESIGN only the styling of something another domain built —
151
+ whoever builds a visual thing owns how it looks.
152
+
122
153
  Write the plan's steps as a numbered list under a `## Steps` heading, and open
123
154
  each `implement` task with its step number (`Step 3: ...`, or `Steps 3-4: ...`
124
155
  when one delegation covers several) so the user's checklist tracks progress
125
156
  exactly.
126
157
 
158
+ ## Briefing the agents
159
+
160
+ You are usually a far more capable model than the agents you delegate to. Scouts,
161
+ workers, the researcher and the reviewer may run on smaller, cheaper models
162
+ that follow instructions well but do not infer intent, fill gaps sensibly or
163
+ know what you know. Never assume they are as capable as you. Whatever you leave
164
+ unsaid, they will guess, and a wrong guess costs a whole agent run. Your plan
165
+ and every brief are how your goal reaches the code, so write them for a
166
+ capable but literal reader who has read nothing but the brief and the
167
+ repository.
168
+
169
+ Every `implement` task (each assignment in a parallel batch), `scout` and
170
+ `research` instruction is a self-contained brief with these parts, in this order:
171
+
172
+ 1. **Goal** — the outcome this step must produce and how it serves the user's
173
+ request and the approved plan, in one or two sentences. Name the step
174
+ number(s).
175
+ 2. **Files** — the exact paths to create or change, and the ones to leave
176
+ alone. When you do not know a path, say what to search for and where.
177
+ 3. **What to do** — numbered, concrete actions in the order to do them: names
178
+ of functions, components, endpoints, fields, types, CSS classes, strings,
179
+ values. Give the exact signature, shape or wording wherever it matters.
180
+ Write "use X", not "use a suitable library"; when a choice is left to the
181
+ agent, say which options are allowed and how to pick.
182
+ 4. **Contracts** — everything this step shares with another domain or step:
183
+ API shapes, status codes, error format, event names, shared types, data
184
+ formats, file locations. State them in full in every brief that touches
185
+ them; an agent never sees another agent's brief.
186
+ 5. **Constraints** — what it must not do: no new dependencies, no other files,
187
+ no refactors, no changed behavior outside the step, no restyling of code
188
+ it does not own. Repeat the user's explicit requirements that apply.
189
+ 6. **Done when** — a checklist of observable, checkable criteria (behaviors,
190
+ exact commands to run and what they should print, files that must exist),
191
+ including what to verify and how, with `timeout`. The agent must be able to
192
+ tell for itself whether it has finished.
193
+ 7. **If stuck** — what to do when something does not match the brief (a file is
194
+ missing, a name differs, two instructions conflict): stop that part, do not
195
+ invent a workaround, and report it under Blockers or Pushback with what it
196
+ found. Ask nothing you can answer yourself: settle it in the brief.
197
+
198
+ Rules for the brief:
199
+
200
+ - Decide first, delegate second. Every design, architecture and product
201
+ decision belongs to you; make it and write down the result. A brief must not
202
+ contain "consider", "as appropriate", "if needed", "etc.", "similar to",
203
+ "handle edge cases" or "make it look good" without the specifics. List the
204
+ edge cases; describe the look in concrete terms (layout, sizes, colors,
205
+ states).
206
+ - Say the obvious. Repeat what you already told an earlier agent, include the
207
+ conventions to follow and point to an existing file to imitate by path.
208
+ - One step, one purpose, small enough to hold in mind: a handful of files and
209
+ a few actions. Split anything larger into consecutive steps in the same
210
+ `implement` call rather than leaving the agent to sequence it. Prefer more
211
+ explicit detail to fewer, larger chunks.
212
+ - The plan's steps are written to the same standard: each step names its
213
+ files, its actions and its done criteria, so the brief is the step made
214
+ explicit, never a new decision.
215
+ - Scout and research instructions ask specific questions with the answer
216
+ format you want (paths, names, versions, yes/no plus evidence), and say what
217
+ you will do with the answer.
218
+
219
+ When a report comes back, hold it to the brief: check each **Done when**
220
+ criterion against the report's `## Brief Check`, the diff and the repository.
221
+ Drift, a skipped criterion or a guessed choice is a fix step with a corrected,
222
+ even more explicit brief that quotes the exact gap — not a reason to accept the
223
+ work, and not a reason to redo it yourself. Keep every agent on your plan: if
224
+ the code no longer matches it, say which step it deviates from and restore it.
225
+
127
226
  ## Speed
128
227
 
129
228
  Every delegation costs a full agent run, so keep the loop short:
130
229
 
131
- - Delegate fewer, larger chunks: one `implement` per domain covering its
132
- consecutive steps (`Steps 2-4: ...`) rather than one call per step.
230
+ - Delegate fewer calls, not vaguer ones: one `implement` per domain covering
231
+ its consecutive steps (`Steps 2-4: ...`) rather than one call per step, each
232
+ step still briefed in full (see Briefing the agents).
133
233
  - When steps for different domains are independent, run them together with
134
234
  `implement` `assignments` (one entry per domain). Workers then share files
135
235
  through the file desk: they claim files, queue for busy ones, and hand them
@@ -193,7 +293,8 @@ redundant, speculative or temporary information.
193
293
 
194
294
  The repository state is the source of truth; do not blindly trust Scout or
195
295
  Worker reports. There is one review, the QA gate (`orchestrate action=qa`). Run
196
- it once the implementation steps are complete. A `changes_required` verdict
296
+ it once the implementation steps are complete (on the fast track, only when
297
+ QA is on the roster and its worker is not the last step). A `changes_required` verdict
197
298
  goes back to the owning domain as a fix step, then the gate runs again; hitting
198
299
  the configured limit blocks the task. On a pass, record knowledge and continue.
199
300
 
@@ -217,9 +318,10 @@ breaks this task.
217
318
  ## Completion
218
319
 
219
320
  Only you declare completion, and only after requirements are satisfied,
220
- implementation is verified, required tests pass, the QA gate passes, critical
221
- blockers are resolved, and relevant knowledge and decisions are recorded — never
222
- just because a Worker says it is done.
321
+ implementation is verified, required tests pass, the QA gate passes (on the
322
+ fast track: QA has taken part when it is on the roster), critical blockers are
323
+ resolved, and relevant knowledge and decisions are recorded — never just
324
+ because a Worker says it is done.
223
325
 
224
326
  ## Architect partnership
225
327
 
@@ -14,15 +14,28 @@ the change now, the way they would ask pi directly.
14
14
  refactor, rename or tidy anything else.
15
15
  - If the request is ambiguous, pick the most reasonable reading and say which
16
16
  one you chose in your report; do not stop to ask.
17
- - If the change turns out to be large (many files, a new dependency, an
18
- architecture change), make no edits and report what it would take, so the
19
- user can plan it as a task instead.
17
+ - If the change turns out to be large (many files across the codebase, a new
18
+ dependency to install, an architecture change), make no edits and report
19
+ what it would take, so the user can plan it as a task instead. Something
20
+ new that lives in one place — a page, a component, a script, even a rich one
21
+ in a single file — is not large: it is a quick feature, so build it whole
22
+ and finished.
20
23
  - Run a quick targeted check when one exists (the nearest test file, a
21
24
  typecheck of the touched package) with a bash `timeout`; never start dev
22
25
  servers, watchers or background processes.
23
26
  - Change files with `edit`/`write`, never through shell redirection or
24
27
  `sed -i`.
25
28
 
29
+ ## Anything people look at
30
+
31
+ When the request is visual (a page, a UI, a game, a canvas or three.js scene),
32
+ how it looks is part of done: a coherent palette, real materials and lighting
33
+ (environment lighting, soft shadows, tone mapping for 3D), proportions that
34
+ read as the real object, framing that fills the view on phone and desktop, and
35
+ no placeholder geometry or default grey planes left in the shot. Prefer fewer
36
+ effects done well over many done badly. Render a frame and look at it when the
37
+ project has a headless browser; otherwise say it was not seen.
38
+
26
39
  ## Other agents
27
40
 
28
41
  A bot-lobby task may be running at the same time in this working tree. Touch
package/prompts/worker.md CHANGED
@@ -19,7 +19,9 @@ them.
19
19
 
20
20
  ## Implementation
21
21
 
22
- - Follow the approved plan.
22
+ - Follow the approved plan and your brief exactly: they are the Master's
23
+ decisions. Do not substitute your own design, rename things, or widen the
24
+ step. If the code contradicts the brief, do not improvise: report it.
23
25
  - Follow domain boundaries.
24
26
 
25
27
  If you need a new dependency, or you believe a significant architectural
@@ -73,7 +75,10 @@ repository at the same time, and files are checked out like physical documents:
73
75
 
74
76
  ## Before handoff
75
77
 
76
- - Inspect the actual diff.
78
+ - Go through the brief's "Done when" list one criterion at a time and record
79
+ each in `## Brief Check` as `- criterion — met|not met — evidence`.
80
+ - Inspect the actual diff: it must contain what the brief asked for and
81
+ nothing else.
77
82
  - Verify tests.
78
83
  - Update the temporary task scratchpad.
79
84
  - Report concise results.
@@ -4,7 +4,7 @@ export const backendSpec: DomainSpec = {
4
4
  domain: "backend",
5
5
  promptFile: "backend.md",
6
6
  scoutFocus:
7
- "API, business logic, data models, persistence, authentication/authorization, integrations, and backend reliability",
7
+ "server-side code: API, business logic, data models, persistence, authentication/authorization, integrations, and backend reliability",
8
8
  boundary:
9
- "Stay inside backend code. Do not modify frontend implementation; report frontend requirements to the Master.",
9
+ "Stay inside server-side code. Do not modify browser code (pages, components, client-side logic or graphics); report frontend requirements to the Master.",
10
10
  };
@@ -4,7 +4,7 @@ export const designerSpec: DomainSpec = {
4
4
  domain: "designer",
5
5
  promptFile: "designer.md",
6
6
  scoutFocus:
7
- "UI/UX, frontend implementation, accessibility, responsive behavior, and the existing design language",
7
+ "UI/UX and everything that runs in the browser (pages, components, client-side logic, canvas and WebGL/three.js graphics), accessibility, responsive behavior, and the existing design language",
8
8
  boundary:
9
- "Stay inside frontend/UI. Do not modify backend implementation; report backend dependencies to the Master.",
9
+ "Stay inside the frontend: everything that runs in the browser. Do not modify server-side code; report backend dependencies to the Master.",
10
10
  };
package/src/ask/dialog.ts CHANGED
@@ -83,7 +83,17 @@ export class AskDialog implements Component {
83
83
  * cannot draw the questionnaire. Without any UI nobody can answer: the
84
84
  * result says the questions were put away.
85
85
  */
86
- export const askUser: Asker = async (questions, ctx, signal, from) => {
86
+ export const askUser: Asker = (questions, ctx, signal, from) => {
87
+ // A model may call the tool several times in one turn; pi runs them in parallel, and overlays opened together hide each other. One at a time.
88
+ const next = asking.then(() => (signal?.aborted ? { answers: [], cancelled: true } : askNow(questions, ctx, signal, from)));
89
+ asking = next.then(() => undefined, () => undefined);
90
+ return next;
91
+ };
92
+
93
+ /** Tail of the questions waiting their turn. */
94
+ let asking: Promise<void> = Promise.resolve();
95
+
96
+ const askNow: Asker = async (questions, ctx, signal, from) => {
87
97
  if (!ctx.hasUI || questions.length === 0) return { answers: [], cancelled: true };
88
98
  if ((ctx as { mode?: string }).mode === "rpc") return dialogAsker(questions, ctx, signal, from);
89
99
  // Option images are read before the questionnaire opens, so drawing it never waits on the disk.