@a-t-h-i/bot-lobby 0.6.4 → 0.6.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +150 -12
- package/package.json +12 -3
- package/prompts/backend.md +3 -1
- package/prompts/designer.md +24 -1
- package/prompts/global.md +18 -0
- package/prompts/master.md +122 -20
- package/prompts/quickfix.md +16 -3
- package/prompts/worker.md +7 -2
- package/src/agents/backend.ts +2 -2
- package/src/agents/designer.ts +2 -2
- package/src/ask/dialog.ts +11 -1
- package/src/classifier/triage.ts +24 -7
- package/src/index.ts +4 -1
- package/src/lobby/feed.ts +7 -2
- package/src/lobby/keys.ts +0 -1
- package/src/lobby/quickfix.ts +29 -3
- package/src/lobby/runtime.ts +29 -26
- package/src/lobby/tabs/home.ts +6 -42
- package/src/lobby/tabs/quickfix.ts +1 -1
- package/src/lobby/tabs/tasks.ts +3 -2
- package/src/lobby/view.ts +18 -23
- package/src/master/decisions.ts +10 -2
- package/src/pi/activity.ts +1 -33
- package/src/pi/commands.ts +22 -11
- package/src/pi/events.ts +3 -17
- package/src/pi/plan-checklist.ts +305 -0
- package/src/pi/route.ts +180 -0
- package/src/pi/run-summary.ts +12 -3
- package/src/pi/settings-ui.ts +2 -2
- package/src/pi/start-task.ts +51 -5
- package/src/pi/tools.ts +13 -7
- package/src/pi/ui.ts +18 -284
- package/src/roles/worker.ts +11 -3
- package/src/schemas/configuration.ts +19 -6
- package/src/schemas/task.ts +31 -0
- package/src/workflow/brief.ts +59 -0
- package/src/workflow/track.ts +436 -0
- package/src/workflow/workflow.ts +151 -10
- package/src/pi/expressions.ts +0 -169
- package/src/pi/kaomoji.ts +0 -227
- package/src/pi/mascot-art.ts +0 -359
- package/src/pi/zen-large.ts +0 -699
- package/src/pi/zen-metrics.ts +0 -130
- package/src/pi/zen.ts +0 -659
package/README.md
CHANGED
|
@@ -2,7 +2,11 @@
|
|
|
2
2
|
|
|
3
3
|
A [Pi](https://pi.dev) extension that turns Pi into a multi-agent software team.
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+

|
|
6
|
+
|
|
7
|
+
`/bot-lobby <request>` (or a request typed in the lobby) starts a task, unless
|
|
8
|
+
one agent can simply do it: then it goes to the [quick-fix agent](#quick-fix-or-the-team).
|
|
9
|
+
For a task, your Pi session becomes the **Master**
|
|
6
10
|
(the "oracle"): it scouts the codebase, proposes a plan, and delegates the
|
|
7
11
|
work to three domain agents — **Designer+Frontend**, **Backend** and **QA** —
|
|
8
12
|
each running in its own isolated `pi` process. QA's reviewer is the quality
|
|
@@ -12,6 +16,10 @@ The rule: **LLMs decide, the engine enforces.** Agents propose; the
|
|
|
12
16
|
extension validates every state change, permission and approval through one
|
|
13
17
|
`orchestrate` tool.
|
|
14
18
|
|
|
19
|
+
The oracle is meant to be your most capable model; the agents can be smaller
|
|
20
|
+
and cheaper ones. So it never assumes they are as capable as it is: it makes
|
|
21
|
+
every decision itself and [briefs each agent in full](#briefing-the-agents).
|
|
22
|
+
|
|
15
23
|
## Install
|
|
16
24
|
|
|
17
25
|
```bash
|
|
@@ -31,7 +39,7 @@ tool names.
|
|
|
31
39
|
|
|
32
40
|
1. `/bot-lobby add a login page` — starts a task; the lobby opens.
|
|
33
41
|
2. Answer the Master's questions, then approve its proposal.
|
|
34
|
-
3.
|
|
42
|
+
3. Follow the agents in the lobby (`alt+l` shows or hides it): what they do, and what they think.
|
|
35
43
|
|
|
36
44
|
`/bot-lobby settings` sets each agent's model, thinking level, time limit and
|
|
37
45
|
extra instructions.
|
|
@@ -41,7 +49,7 @@ extra instructions.
|
|
|
41
49
|
| Command | Does |
|
|
42
50
|
| --- | --- |
|
|
43
51
|
| `/bot-lobby` | Open the lobby (`alt+l`) |
|
|
44
|
-
| `/bot-lobby <request>` | Start a task (`--task`
|
|
52
|
+
| `/bot-lobby <request>` | Start a request: a [quick fix](#quick-fix-or-the-team) when one agent can do it alone, else a task (`--task` to always make it a task, also when it begins with a command word; `--auto` to run unattended, `--budget 90m` to give it a time budget, `--fast` / `--full` to pick its [track](#fast-track-or-full-workflow)) |
|
|
45
53
|
| `/bot-lobby budget [90m\|off]` | Show or set this session's task time budget |
|
|
46
54
|
| `/bot-lobby status \| tasks \| runs [id]` | Current task, all tasks, recent agent runs |
|
|
47
55
|
| `/bot-lobby approve \| amend <text> \| decline` | Answer the proposal |
|
|
@@ -54,15 +62,106 @@ extra instructions.
|
|
|
54
62
|
| `/bot-lobby knowledge` | Knowledge file sizes |
|
|
55
63
|
| `/bot-lobby minimize \| restore` | Hide bot-lobby in this session (`ctrl+shift+m`) |
|
|
56
64
|
|
|
65
|
+
## Quick fix or the team
|
|
66
|
+
|
|
67
|
+
Before any task exists, bot-lobby asks whether **one agent can just do it**:
|
|
68
|
+
in one file or one area, with nothing to agree between frontend and backend,
|
|
69
|
+
no unfamiliar codebase to survey, nothing risky and no decision you must make
|
|
70
|
+
first. [Jev](#the-classifier-jev) answers when it is on (`classifier.thresholds.quickFixAt`,
|
|
71
|
+
0.7); plain rules answer otherwise, and whenever Jev is unsure. A request that
|
|
72
|
+
says it is self-contained ("a single page", "in one html file") counts even
|
|
73
|
+
when it is rich.
|
|
74
|
+
|
|
75
|
+
When it reads that way, the oracle confirms in one step, without reading
|
|
76
|
+
files or planning (`route_request`), and says so: *this looks like a quick
|
|
77
|
+
feature, the quick-fix agent is on it*. The lobby then hands the request to
|
|
78
|
+
the quick-fix agent and switches to the **Quick fix** tab, where you follow
|
|
79
|
+
it. No scouts, proposal, plan or QA. A quick feature (bigger than a small
|
|
80
|
+
change, but in one place) runs on its builder's model, thinking and time limit
|
|
81
|
+
(DESIGN's for a page) instead of the quick-fix defaults.
|
|
82
|
+
|
|
83
|
+
Everything else, or anything the oracle judges needs the team, starts as a
|
|
84
|
+
task below. `--task` always makes a task; `workflow.routeQuickFixes: false`
|
|
85
|
+
turns routing off. Without the lobby (a background session, RPC mode) every
|
|
86
|
+
request is a task.
|
|
87
|
+
|
|
57
88
|
## How a task runs
|
|
58
89
|
|
|
59
90
|
```
|
|
60
|
-
request → clarify → scout → propose → approve → plan → implement → QA gate → complete
|
|
91
|
+
full workflow: request → clarify → scout → propose → approve → plan → implement → QA gate → complete
|
|
92
|
+
fast track: request → implement (only the agents it needs) → QA, if it needs tests → complete
|
|
61
93
|
```
|
|
62
94
|
|
|
95
|
+
### Fast track or full workflow
|
|
96
|
+
|
|
97
|
+
The moment a task starts, bot-lobby reads the request and decides how serious
|
|
98
|
+
it is: its size, who has to take part, and whether it takes the **fast
|
|
99
|
+
track** or the **full workflow**. The read is instant and costs no tokens
|
|
100
|
+
(plain rules, refined by [Jev](#the-classifier-jev) when that is on).
|
|
101
|
+
|
|
102
|
+
| The request… | Who takes part |
|
|
103
|
+
| --- | --- |
|
|
104
|
+
| changes anything that runs in the browser: screens, components, styles, copy, canvas or three.js graphics | DESIGN (frontend) |
|
|
105
|
+
| changes an API, the database, auth, jobs or other server-side code | DEV (backend) |
|
|
106
|
+
| needs tests: asks for them, fixes a bug, or changes backend logic | QA |
|
|
107
|
+
| depends on outside facts: latest versions, docs, standards, third-party APIs | RESEARCH |
|
|
108
|
+
|
|
109
|
+
A **small, clear, low-risk** request takes the fast track: the oracle hands
|
|
110
|
+
it straight to those agents, with no scouts, no proposal to approve and no
|
|
111
|
+
plan document (the engine keeps a short plan whose steps are the
|
|
112
|
+
delegations, so the checklist still works). QA takes part only when the
|
|
113
|
+
change needs tests (its worker writing and running them as the last step, or
|
|
114
|
+
the QA gate); without it the task completes as soon as the work is in and
|
|
115
|
+
checked. A copy change is one agent run.
|
|
116
|
+
|
|
117
|
+
Everything else takes the full workflow below: anything **medium or large**
|
|
118
|
+
(a new page, flow or endpoint, a refactor, an upgrade, a vague or
|
|
119
|
+
many-part request), **serious** whatever its size (security, auth,
|
|
120
|
+
passwords, payments, migrations, production, personal data), or **unclear**
|
|
121
|
+
(too short, vague, or asking to investigate first).
|
|
122
|
+
|
|
123
|
+
- The oracle glances at the read once and acts on it, or corrects it with
|
|
124
|
+
`orchestrate action=track`: the full workflow for a change bigger than it
|
|
125
|
+
reads, the fast track for one that is smaller (before its work is planned),
|
|
126
|
+
or a member added or dropped.
|
|
127
|
+
- Once work is under way a track only gets stricter: the full workflow or
|
|
128
|
+
more members, never QA dropped. A fast task whose worker asks for a new
|
|
129
|
+
dependency or an architecture change gets QA.
|
|
130
|
+
- `/bot-lobby --fast <request>` or `--full <request>` decides it yourself;
|
|
131
|
+
the oracle never moves a `--full` task to the fast track.
|
|
132
|
+
`workflow.fastTrack: false` puts every task on the full workflow.
|
|
133
|
+
- The track shows in the lobby's activity log, the task's details on the
|
|
134
|
+
Tasks tab, and `/bot-lobby status`.
|
|
135
|
+
|
|
136
|
+
### Briefing the agents
|
|
137
|
+
|
|
138
|
+
Every agent may run on a smaller, cheaper model than the oracle's, one that
|
|
139
|
+
follows instructions well but does not infer intent. So the oracle writes
|
|
140
|
+
each delegation (`implement`, `scout`, `research`) as a self-contained brief
|
|
141
|
+
and settles every design and architecture decision itself first:
|
|
142
|
+
|
|
143
|
+
- **Goal**, the exact **Files**, numbered **What to do** with names, shapes
|
|
144
|
+
and values, the **Contracts** shared with other agents (repeated in full in
|
|
145
|
+
each brief), **Constraints**, **Done when** (checkable criteria and the
|
|
146
|
+
commands to run) and **If stuck**.
|
|
147
|
+
- Plan steps are written to the same standard, so a brief is the plan step
|
|
148
|
+
made explicit, never a new decision.
|
|
149
|
+
- Every agent is told to follow its brief and the approved plan exactly, use
|
|
150
|
+
the given names letter for letter, and report what does not match instead of
|
|
151
|
+
guessing. Workers end their report with a **Brief Check**: each "Done when"
|
|
152
|
+
item, met or not, with evidence.
|
|
153
|
+
- The oracle holds each report to its brief. Drift, a skipped criterion or a
|
|
154
|
+
guessed choice comes back as a fix step with a more explicit brief.
|
|
155
|
+
- The engine backs it up: a delegation that names nothing concrete (or a long
|
|
156
|
+
one with no done criteria) is sent back to the oracle once before any agent
|
|
157
|
+
starts. Sending the same text again goes through, so a short task that is
|
|
158
|
+
complete as written is never stuck. `workflow.briefCheck: false` turns it off.
|
|
159
|
+
|
|
160
|
+
### The full workflow
|
|
161
|
+
|
|
63
162
|
- **Scouts** (read-only) investigate the domains the request touches.
|
|
64
163
|
- The Master **proposes** a short bullet list; nothing is built until you
|
|
65
|
-
approve it.
|
|
164
|
+
approve it.
|
|
66
165
|
- **Workers** implement plan steps, one domain each. Several can run in
|
|
67
166
|
parallel; they share files through a **file desk** (claim a file, queue for
|
|
68
167
|
a busy one, hand it over with a note).
|
|
@@ -117,16 +216,54 @@ everything. `workflow.freshContext: false` in the config turns this off.
|
|
|
117
216
|
## The lobby
|
|
118
217
|
|
|
119
218
|
A full-screen view with a prompt at the bottom that talks to whatever tab is
|
|
120
|
-
open. `alt+h` lists every key.
|
|
219
|
+
open. `alt+h` lists every key. It is text only: no animations, just the
|
|
220
|
+
conversation, the activity log and the thoughts, and a one-line status in Pi's
|
|
221
|
+
footer.
|
|
121
222
|
|
|
122
223
|
| Tab | What it is |
|
|
123
224
|
| --- | --- |
|
|
124
|
-
| **1 Lobby** |
|
|
225
|
+
| **1 Lobby** | Your conversation with the oracle, an activity log of every agent's steps, and each agent's latest thought |
|
|
125
226
|
| **2 Tasks** | Every task and saved plan as a checklist. `s` starts a plan in a new session, `h` here; `c` comments on a plan; `a` archives, `d` deletes |
|
|
126
227
|
| **3 Plan** | Plan a task with a panel of agents before building it (below) |
|
|
127
|
-
| **4 Quick fix** | One agent makes a
|
|
228
|
+
| **4 Quick fix** | One agent makes a change right away, beside any running task; requests the oracle [routes here](#quick-fix-or-the-team) show up too |
|
|
128
229
|
| **5 Metrics** | Run time, success rate, tokens and cost per model and agent |
|
|
129
230
|
|
|
231
|
+
### Lobby
|
|
232
|
+
|
|
233
|
+

|
|
234
|
+
|
|
235
|
+
Your conversation with the oracle on the left, the activity log of every
|
|
236
|
+
agent's steps on the right, and the agents' latest thoughts below. `alt+c`,
|
|
237
|
+
`alt+a` and `alt+k` hide any of the three; the prompt steers the running turn.
|
|
238
|
+
|
|
239
|
+
### Tasks
|
|
240
|
+
|
|
241
|
+

|
|
242
|
+
|
|
243
|
+
Every task as a checklist, with its track, plan progress, the approved plan
|
|
244
|
+
and your comments on it.
|
|
245
|
+
|
|
246
|
+
### Plan
|
|
247
|
+
|
|
248
|
+

|
|
249
|
+
|
|
250
|
+
The panel's questions, with recommended options, on the left; the draft plan
|
|
251
|
+
on the right. See [Planning](#planning).
|
|
252
|
+
|
|
253
|
+
### Quick fix
|
|
254
|
+
|
|
255
|
+

|
|
256
|
+
|
|
257
|
+
One agent's jobs, each with its live steps, the files it edited and its
|
|
258
|
+
report. See [Quick fix or the team](#quick-fix-or-the-team).
|
|
259
|
+
|
|
260
|
+
### Metrics
|
|
261
|
+
|
|
262
|
+

|
|
263
|
+
|
|
264
|
+
Run time, success rate, cost and tokens per model and agent, so you can see
|
|
265
|
+
which cheaper models hold up.
|
|
266
|
+
|
|
130
267
|
Common keys: `tab` switches tabs, `esc` browses (arrows, single-key
|
|
131
268
|
commands), `ctrl+f` searches, `ctrl+s` saves the plan, `alt+o` browses
|
|
132
269
|
sessions, `alt+n` starts a task in a new session, `alt+s` opens settings.
|
|
@@ -236,7 +373,8 @@ else TypeSafe.
|
|
|
236
373
|
| Planning seats | Each round, only the seats the idea or your latest answers touch sit; `1`–`4` pins a seat |
|
|
237
374
|
| Obvious answers | Answers a question itself when the conversation already makes the recommended option clearly right (≥ 0.9); listed under Assumptions |
|
|
238
375
|
| File hints | Agents start with a short list of the files they most likely need, and get a `find_relevant_files` tool |
|
|
239
|
-
|
|
|
376
|
+
| Quick fix or task | Whether one engineer can do a new request alone decides whether it goes to the [quick-fix agent](#quick-fix-or-the-team) (the oracle confirms) |
|
|
377
|
+
| Task triage | The task's [track](#fast-track-or-full-workflow) and roster use its read (size, domains, research, ambiguity), and the Master gets it as hints; a quick fix that is really a task (large, and not one engineer's work) is held (`r` run anyway, `t` make it a task) |
|
|
240
378
|
| Effort routing | Simple steps run one thinking level lower; trivial ones on a **cheaper model** you pick. A routed run that falls short re-runs on your normal settings |
|
|
241
379
|
|
|
242
380
|
**It never gets in the way:** any failure, timeout or missing key means
|
|
@@ -264,7 +402,7 @@ the result.
|
|
|
264
402
|
"scout": { "model": "anthropic/claude-haiku-4-5-20251001", "timeoutMs": 480000 },
|
|
265
403
|
"planner": { "thinking": "high", "timeoutMs": 300000 },
|
|
266
404
|
"lobby": { "planningPanel": ["backend", "designer", "qa", "researcher"], "maxPlanningRounds": 5 },
|
|
267
|
-
"workflow": { "maxReviewIterations": 2, "maxParallelWorkers": 3, "stallTimeoutMs": 300000, "wrapUpAt": 0.75, "taskBudgetMinutes": 0 },
|
|
405
|
+
"workflow": { "maxReviewIterations": 2, "maxParallelWorkers": 3, "stallTimeoutMs": 300000, "wrapUpAt": 0.75, "taskBudgetMinutes": 0, "fastTrack": true, "briefCheck": true, "routeQuickFixes": true },
|
|
268
406
|
"classifier": { "enabled": false, "provider": "auto", "effort": { "cheapModel": "inherit" } }
|
|
269
407
|
}
|
|
270
408
|
```
|
|
@@ -283,11 +421,11 @@ the result.
|
|
|
283
421
|
| Rule | How |
|
|
284
422
|
| --- | --- |
|
|
285
423
|
| Steps happen in order | A state machine validates every action |
|
|
286
|
-
| Nothing is built before approval | `implement` refuses earlier states |
|
|
424
|
+
| Nothing is built before approval | On the full workflow `implement` refuses earlier states; only a fast-track task (small, clear, low-risk) starts straight away |
|
|
287
425
|
| Scouts and the QA gate can't edit code | Scouts get read-only tools; the QA gate adds only `bash` for tests |
|
|
288
426
|
| New dependencies and architecture changes need approval | Parsed from worker reports; the domain is blocked until resolved |
|
|
289
427
|
| Only the Master writes knowledge | Agents can only propose it |
|
|
290
|
-
| "Done" is earned | Needs a plan, a passing QA gate that ran checks (or your explicit acceptance), and no open blockers |
|
|
428
|
+
| "Done" is earned | Needs a plan, a passing QA gate that ran checks (or your explicit acceptance), and no open blockers; on the fast track, a finished worker step, and QA's part only when the change needs tests |
|
|
291
429
|
| Parallel workers don't clobber files | Edits need a file-desk claim |
|
|
292
430
|
| A crash doesn't corrupt a task | State is on disk; tasks resume from their state |
|
|
293
431
|
|
package/package.json
CHANGED
|
@@ -1,11 +1,19 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@a-t-h-i/bot-lobby",
|
|
3
|
-
"version": "0.6.
|
|
3
|
+
"version": "0.6.6",
|
|
4
4
|
"description": "Structured multi-agent software engineering orchestrator for Pi",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "Apache-2.0",
|
|
7
7
|
"keywords": [
|
|
8
|
-
"pi-package"
|
|
8
|
+
"pi-package",
|
|
9
|
+
"pi",
|
|
10
|
+
"pi-extension",
|
|
11
|
+
"pi-coding-agent",
|
|
12
|
+
"multi-agent",
|
|
13
|
+
"subagents",
|
|
14
|
+
"orchestrator",
|
|
15
|
+
"code-review",
|
|
16
|
+
"web-search"
|
|
9
17
|
],
|
|
10
18
|
"repository": {
|
|
11
19
|
"type": "git",
|
|
@@ -25,7 +33,8 @@
|
|
|
25
33
|
"pi": {
|
|
26
34
|
"extensions": [
|
|
27
35
|
"./src/index.ts"
|
|
28
|
-
]
|
|
36
|
+
],
|
|
37
|
+
"image": "https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/gallery.png"
|
|
29
38
|
},
|
|
30
39
|
"scripts": {
|
|
31
40
|
"typecheck": "tsc --noEmit",
|
package/prompts/backend.md
CHANGED
|
@@ -2,7 +2,9 @@
|
|
|
2
2
|
|
|
3
3
|
You own backend engineering: API, business logic, data models, database,
|
|
4
4
|
authentication, authorization, integrations, backend architecture, security,
|
|
5
|
-
reliability and backend performance.
|
|
5
|
+
reliability and backend performance. Code that runs in the browser — pages,
|
|
6
|
+
client-side logic, canvas and WebGL/three.js graphics — is the designer's,
|
|
7
|
+
however much logic it holds.
|
|
6
8
|
|
|
7
9
|
## API design
|
|
8
10
|
|
package/prompts/designer.md
CHANGED
|
@@ -2,7 +2,9 @@
|
|
|
2
2
|
|
|
3
3
|
You own UI/UX and frontend engineering: user experience, interaction design,
|
|
4
4
|
visual consistency, frontend implementation, responsive behavior,
|
|
5
|
-
accessibility, frontend performance and the design language.
|
|
5
|
+
accessibility, frontend performance and the design language. Everything that
|
|
6
|
+
runs in the browser is yours, including client-side logic and canvas,
|
|
7
|
+
WebGL/three.js graphics.
|
|
6
8
|
|
|
7
9
|
You have real visual taste. Your work is calm, considered and quietly
|
|
8
10
|
delightful: the understated, crafted aesthetic of Anthropic's latest models,
|
|
@@ -66,6 +68,27 @@ one-off values.
|
|
|
66
68
|
- **No magic numbers:** every size, space, color, radius, shadow and duration
|
|
67
69
|
comes from a token or the existing scale.
|
|
68
70
|
|
|
71
|
+
## Graphics and 3D
|
|
72
|
+
|
|
73
|
+
When the work is a scene (canvas, WebGL, three.js), how it looks is the
|
|
74
|
+
product, and a generic render is a failed one.
|
|
75
|
+
|
|
76
|
+
- **Light it like a photograph:** image-based lighting (an environment map,
|
|
77
|
+
e.g. PMREM with `RoomEnvironment`) plus one key light with soft shadows;
|
|
78
|
+
ACES or AgX tone mapping and sRGB output. Never flat ambient light.
|
|
79
|
+
- **Physical materials:** `MeshPhysicalMaterial` with values from the real
|
|
80
|
+
thing — glass with transmission, thickness, IOR about 1.5 and a faint tint;
|
|
81
|
+
wood, brass or stone with sensible roughness and metalness. Liquids and
|
|
82
|
+
grains read by their own color, sheen and scale, not by a texture.
|
|
83
|
+
- **Proportions and framing:** model the object after a real reference; frame
|
|
84
|
+
it so it fills most of the view at rest, on every screen size, with a quiet
|
|
85
|
+
backdrop (a soft gradient, a floor that fades out). No default grey planes,
|
|
86
|
+
giant ground discs or horizon lines cutting through the shot.
|
|
87
|
+
- **Restraint:** fewer effects done well beat many done badly; hold 60fps on a
|
|
88
|
+
phone with quality tiers rather than dropping detail everywhere.
|
|
89
|
+
- **Look at it:** render a frame when the project has a headless browser, and
|
|
90
|
+
fix what you see; otherwise say in your report that it was not seen.
|
|
91
|
+
|
|
69
92
|
## Interaction and motion
|
|
70
93
|
|
|
71
94
|
You like interactive UIs that give the user subtle, fun feedback — never
|
package/prompts/global.md
CHANGED
|
@@ -15,6 +15,24 @@ Perform your assigned responsibility precisely and remain within your domain.
|
|
|
15
15
|
- Do not invent requirements; ask when they are genuinely ambiguous.
|
|
16
16
|
- Report conclusions, evidence, decisions, findings and blockers concisely.
|
|
17
17
|
|
|
18
|
+
## Following your brief
|
|
19
|
+
|
|
20
|
+
The Master that instructs you plans the work and knows the whole task; you may
|
|
21
|
+
be a smaller model that sees only your part. Your brief and the approved plan
|
|
22
|
+
are your authority, so:
|
|
23
|
+
|
|
24
|
+
- Read the whole brief and the plan before acting, and do exactly what it
|
|
25
|
+
says, in its order. Do not add features, refactors or "improvements", and do
|
|
26
|
+
not skip steps because they look unnecessary.
|
|
27
|
+
- Use the names, paths, shapes and wording the brief gives you, letter for
|
|
28
|
+
letter. Where it leaves a detail open, choose the simplest option that
|
|
29
|
+
follows the existing code, and say what you chose in your report.
|
|
30
|
+
- Never guess at something that matters (a contract, a file that is not there,
|
|
31
|
+
a conflict between the brief and the code). Stop that part, finish the rest,
|
|
32
|
+
and report it under Blockers or Pushback with what you found.
|
|
33
|
+
- Check your own work against the brief's "Done when" list before you report,
|
|
34
|
+
criterion by criterion, and say honestly which are met and which are not.
|
|
35
|
+
|
|
18
36
|
## Hard rules
|
|
19
37
|
|
|
20
38
|
- Do not add dependencies without approval.
|
package/prompts/master.md
CHANGED
|
@@ -17,32 +17,55 @@ validates every step — state transitions, role permissions, approval gates and
|
|
|
17
17
|
completion authority. If it rejects an action, read the error and adjust; never
|
|
18
18
|
work around it. Do not rely on prompts to enforce permissions or state.
|
|
19
19
|
|
|
20
|
+
## Task track
|
|
21
|
+
|
|
22
|
+
Every task starts on a track the engine read from the request (the `Track`
|
|
23
|
+
line in your context): how serious it is, and so who takes part and how much
|
|
24
|
+
process it gets. Fewer steps win whenever the result is the same.
|
|
25
|
+
|
|
26
|
+
- **Fast track** — a small, clear, low-risk change. No scouts, no proposal,
|
|
27
|
+
no plan document: delegate straight away with `orchestrate action=implement`,
|
|
28
|
+
opening each task with `Step N:` (the engine keeps the plan and the
|
|
29
|
+
checklist). Only the roster takes part: DESIGN for frontend work, DEV for
|
|
30
|
+
backend work, QA when the change needs tests (its worker writing and
|
|
31
|
+
running them as the last step, or the QA gate), and the researcher when a
|
|
32
|
+
decision needs outside facts (summon it first). Several domains: one
|
|
33
|
+
`implement` with `assignments`, each task stating the contract between
|
|
34
|
+
them. When the work is in, check `git diff --stat` and the report, then
|
|
35
|
+
`complete`; without QA on the roster there is no QA gate.
|
|
36
|
+
- **Full workflow** — everything else: the steps below, ending with the QA
|
|
37
|
+
gate.
|
|
38
|
+
|
|
39
|
+
The read is quick and can be wrong, so glance at it once and move on: confirm
|
|
40
|
+
it by acting on it, or correct it with `orchestrate action=track` (`track`,
|
|
41
|
+
`roster`, `reason`). Go full when the change is bigger, riskier (security,
|
|
42
|
+
money, data, migrations, production) or less clear than it reads; take the
|
|
43
|
+
fast track when a full-workflow task turns out small and clear (before its
|
|
44
|
+
work is planned). Add a member the roster is missing, or drop one it does not
|
|
45
|
+
need, before delegating. Once work is under way a track only gets stricter:
|
|
46
|
+
the full workflow or more members, never QA dropped; a fast task whose worker
|
|
47
|
+
asks for a dependency or an architecture change gets QA added. The user's
|
|
48
|
+
`--full` and a disabled fast track keep the full workflow.
|
|
49
|
+
|
|
20
50
|
## Before implementation
|
|
21
51
|
|
|
22
|
-
For
|
|
52
|
+
For full-workflow work: understand the request; clarify with
|
|
23
53
|
`orchestrate action=clarify` when necessary; challenge it when there is a real
|
|
24
54
|
technical, security, reliability, UX or maintainability concern; select and run
|
|
25
55
|
relevant Scouts; review findings and target-verify important claims against the
|
|
26
56
|
repository; synthesize and present a short `- ` bullet-list proposal; then wait for
|
|
27
|
-
approval, amendment, or decline. Do not start
|
|
28
|
-
approval.
|
|
29
|
-
|
|
30
|
-
For a trivial, single-domain request you may skip the Scout round and the
|
|
31
|
-
proposal ceremony: state the short plan, delegate the step, and verify the diff
|
|
32
|
-
directly. The engine allows `clarifying -> awaiting_approval -> planning`, so no
|
|
33
|
-
state override is needed. Skip only when the change is small, obvious and
|
|
34
|
-
confined to one domain.
|
|
57
|
+
approval, amendment, or decline. Do not start full-workflow implementation
|
|
58
|
+
before approval.
|
|
35
59
|
|
|
36
60
|
## Classifier hints
|
|
37
61
|
|
|
38
62
|
When the classifier is on, your task context carries a **Classifier
|
|
39
63
|
triage**: a fast model's read of the request (size, the domains it touches,
|
|
40
64
|
whether it needs outside research, whether it is ambiguous, its kind, likely
|
|
41
|
-
files) and a suggested path
|
|
42
|
-
scout only the domains it marks (0.5 or
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
request as ambiguous — and overrule it whenever the repository says
|
|
65
|
+
files) and a suggested path; the task's track was chosen with it. Use it to
|
|
66
|
+
skip reasoning you do not need — scout only the domains it marks (0.5 or
|
|
67
|
+
more), skip the researcher when research is not needed, clarify only when it
|
|
68
|
+
reads the request as ambiguous — and overrule it whenever the repository says
|
|
46
69
|
otherwise. It is a hint, never a rule.
|
|
47
70
|
|
|
48
71
|
When you `clarify` with options, put your recommended option first and mark
|
|
@@ -119,17 +142,94 @@ Assign work to the correct domain; never ask one domain to do another's. A
|
|
|
119
142
|
cross-domain dependency is reported to you, and you decide whether another
|
|
120
143
|
domain needs a task.
|
|
121
144
|
|
|
145
|
+
Assign by where the code runs, not by how much logic it holds. Everything that
|
|
146
|
+
runs in the browser — pages, components, client-side state and logic, canvas,
|
|
147
|
+
WebGL/three.js scenes, shaders, client-side physics — is DESIGN's
|
|
148
|
+
(Designer+Frontend); DEV (backend) owns server-side code, data, APIs and
|
|
149
|
+
integrations. One file has one owner: never split a file between domains, and
|
|
150
|
+
never give DESIGN only the styling of something another domain built —
|
|
151
|
+
whoever builds a visual thing owns how it looks.
|
|
152
|
+
|
|
122
153
|
Write the plan's steps as a numbered list under a `## Steps` heading, and open
|
|
123
154
|
each `implement` task with its step number (`Step 3: ...`, or `Steps 3-4: ...`
|
|
124
155
|
when one delegation covers several) so the user's checklist tracks progress
|
|
125
156
|
exactly.
|
|
126
157
|
|
|
158
|
+
## Briefing the agents
|
|
159
|
+
|
|
160
|
+
You are usually a far more capable model than the agents you delegate to. Scouts,
|
|
161
|
+
workers, the researcher and the reviewer may run on smaller, cheaper models
|
|
162
|
+
that follow instructions well but do not infer intent, fill gaps sensibly or
|
|
163
|
+
know what you know. Never assume they are as capable as you. Whatever you leave
|
|
164
|
+
unsaid, they will guess, and a wrong guess costs a whole agent run. Your plan
|
|
165
|
+
and every brief are how your goal reaches the code, so write them for a
|
|
166
|
+
capable but literal reader who has read nothing but the brief and the
|
|
167
|
+
repository.
|
|
168
|
+
|
|
169
|
+
Every `implement` task (each assignment in a parallel batch), `scout` and
|
|
170
|
+
`research` instruction is a self-contained brief with these parts, in this order:
|
|
171
|
+
|
|
172
|
+
1. **Goal** — the outcome this step must produce and how it serves the user's
|
|
173
|
+
request and the approved plan, in one or two sentences. Name the step
|
|
174
|
+
number(s).
|
|
175
|
+
2. **Files** — the exact paths to create or change, and the ones to leave
|
|
176
|
+
alone. When you do not know a path, say what to search for and where.
|
|
177
|
+
3. **What to do** — numbered, concrete actions in the order to do them: names
|
|
178
|
+
of functions, components, endpoints, fields, types, CSS classes, strings,
|
|
179
|
+
values. Give the exact signature, shape or wording wherever it matters.
|
|
180
|
+
Write "use X", not "use a suitable library"; when a choice is left to the
|
|
181
|
+
agent, say which options are allowed and how to pick.
|
|
182
|
+
4. **Contracts** — everything this step shares with another domain or step:
|
|
183
|
+
API shapes, status codes, error format, event names, shared types, data
|
|
184
|
+
formats, file locations. State them in full in every brief that touches
|
|
185
|
+
them; an agent never sees another agent's brief.
|
|
186
|
+
5. **Constraints** — what it must not do: no new dependencies, no other files,
|
|
187
|
+
no refactors, no changed behavior outside the step, no restyling of code
|
|
188
|
+
it does not own. Repeat the user's explicit requirements that apply.
|
|
189
|
+
6. **Done when** — a checklist of observable, checkable criteria (behaviors,
|
|
190
|
+
exact commands to run and what they should print, files that must exist),
|
|
191
|
+
including what to verify and how, with `timeout`. The agent must be able to
|
|
192
|
+
tell for itself whether it has finished.
|
|
193
|
+
7. **If stuck** — what to do when something does not match the brief (a file is
|
|
194
|
+
missing, a name differs, two instructions conflict): stop that part, do not
|
|
195
|
+
invent a workaround, and report it under Blockers or Pushback with what it
|
|
196
|
+
found. Ask nothing you can answer yourself: settle it in the brief.
|
|
197
|
+
|
|
198
|
+
Rules for the brief:
|
|
199
|
+
|
|
200
|
+
- Decide first, delegate second. Every design, architecture and product
|
|
201
|
+
decision belongs to you; make it and write down the result. A brief must not
|
|
202
|
+
contain "consider", "as appropriate", "if needed", "etc.", "similar to",
|
|
203
|
+
"handle edge cases" or "make it look good" without the specifics. List the
|
|
204
|
+
edge cases; describe the look in concrete terms (layout, sizes, colors,
|
|
205
|
+
states).
|
|
206
|
+
- Say the obvious. Repeat what you already told an earlier agent, include the
|
|
207
|
+
conventions to follow and point to an existing file to imitate by path.
|
|
208
|
+
- One step, one purpose, small enough to hold in mind: a handful of files and
|
|
209
|
+
a few actions. Split anything larger into consecutive steps in the same
|
|
210
|
+
`implement` call rather than leaving the agent to sequence it. Prefer more
|
|
211
|
+
explicit detail to fewer, larger chunks.
|
|
212
|
+
- The plan's steps are written to the same standard: each step names its
|
|
213
|
+
files, its actions and its done criteria, so the brief is the step made
|
|
214
|
+
explicit, never a new decision.
|
|
215
|
+
- Scout and research instructions ask specific questions with the answer
|
|
216
|
+
format you want (paths, names, versions, yes/no plus evidence), and say what
|
|
217
|
+
you will do with the answer.
|
|
218
|
+
|
|
219
|
+
When a report comes back, hold it to the brief: check each **Done when**
|
|
220
|
+
criterion against the report's `## Brief Check`, the diff and the repository.
|
|
221
|
+
Drift, a skipped criterion or a guessed choice is a fix step with a corrected,
|
|
222
|
+
even more explicit brief that quotes the exact gap — not a reason to accept the
|
|
223
|
+
work, and not a reason to redo it yourself. Keep every agent on your plan: if
|
|
224
|
+
the code no longer matches it, say which step it deviates from and restore it.
|
|
225
|
+
|
|
127
226
|
## Speed
|
|
128
227
|
|
|
129
228
|
Every delegation costs a full agent run, so keep the loop short:
|
|
130
229
|
|
|
131
|
-
- Delegate fewer,
|
|
132
|
-
consecutive steps (`Steps 2-4: ...`) rather than one call per step
|
|
230
|
+
- Delegate fewer calls, not vaguer ones: one `implement` per domain covering
|
|
231
|
+
its consecutive steps (`Steps 2-4: ...`) rather than one call per step, each
|
|
232
|
+
step still briefed in full (see Briefing the agents).
|
|
133
233
|
- When steps for different domains are independent, run them together with
|
|
134
234
|
`implement` `assignments` (one entry per domain). Workers then share files
|
|
135
235
|
through the file desk: they claim files, queue for busy ones, and hand them
|
|
@@ -193,7 +293,8 @@ redundant, speculative or temporary information.
|
|
|
193
293
|
|
|
194
294
|
The repository state is the source of truth; do not blindly trust Scout or
|
|
195
295
|
Worker reports. There is one review, the QA gate (`orchestrate action=qa`). Run
|
|
196
|
-
it once the implementation steps are complete
|
|
296
|
+
it once the implementation steps are complete (on the fast track, only when
|
|
297
|
+
QA is on the roster and its worker is not the last step). A `changes_required` verdict
|
|
197
298
|
goes back to the owning domain as a fix step, then the gate runs again; hitting
|
|
198
299
|
the configured limit blocks the task. On a pass, record knowledge and continue.
|
|
199
300
|
|
|
@@ -217,9 +318,10 @@ breaks this task.
|
|
|
217
318
|
## Completion
|
|
218
319
|
|
|
219
320
|
Only you declare completion, and only after requirements are satisfied,
|
|
220
|
-
implementation is verified, required tests pass, the QA gate passes
|
|
221
|
-
|
|
222
|
-
|
|
321
|
+
implementation is verified, required tests pass, the QA gate passes (on the
|
|
322
|
+
fast track: QA has taken part when it is on the roster), critical blockers are
|
|
323
|
+
resolved, and relevant knowledge and decisions are recorded — never just
|
|
324
|
+
because a Worker says it is done.
|
|
223
325
|
|
|
224
326
|
## Architect partnership
|
|
225
327
|
|
package/prompts/quickfix.md
CHANGED
|
@@ -14,15 +14,28 @@ the change now, the way they would ask pi directly.
|
|
|
14
14
|
refactor, rename or tidy anything else.
|
|
15
15
|
- If the request is ambiguous, pick the most reasonable reading and say which
|
|
16
16
|
one you chose in your report; do not stop to ask.
|
|
17
|
-
- If the change turns out to be large (many files, a new
|
|
18
|
-
architecture change), make no edits and report
|
|
19
|
-
user can plan it as a task instead.
|
|
17
|
+
- If the change turns out to be large (many files across the codebase, a new
|
|
18
|
+
dependency to install, an architecture change), make no edits and report
|
|
19
|
+
what it would take, so the user can plan it as a task instead. Something
|
|
20
|
+
new that lives in one place — a page, a component, a script, even a rich one
|
|
21
|
+
in a single file — is not large: it is a quick feature, so build it whole
|
|
22
|
+
and finished.
|
|
20
23
|
- Run a quick targeted check when one exists (the nearest test file, a
|
|
21
24
|
typecheck of the touched package) with a bash `timeout`; never start dev
|
|
22
25
|
servers, watchers or background processes.
|
|
23
26
|
- Change files with `edit`/`write`, never through shell redirection or
|
|
24
27
|
`sed -i`.
|
|
25
28
|
|
|
29
|
+
## Anything people look at
|
|
30
|
+
|
|
31
|
+
When the request is visual (a page, a UI, a game, a canvas or three.js scene),
|
|
32
|
+
how it looks is part of done: a coherent palette, real materials and lighting
|
|
33
|
+
(environment lighting, soft shadows, tone mapping for 3D), proportions that
|
|
34
|
+
read as the real object, framing that fills the view on phone and desktop, and
|
|
35
|
+
no placeholder geometry or default grey planes left in the shot. Prefer fewer
|
|
36
|
+
effects done well over many done badly. Render a frame and look at it when the
|
|
37
|
+
project has a headless browser; otherwise say it was not seen.
|
|
38
|
+
|
|
26
39
|
## Other agents
|
|
27
40
|
|
|
28
41
|
A bot-lobby task may be running at the same time in this working tree. Touch
|
package/prompts/worker.md
CHANGED
|
@@ -19,7 +19,9 @@ them.
|
|
|
19
19
|
|
|
20
20
|
## Implementation
|
|
21
21
|
|
|
22
|
-
- Follow the approved plan
|
|
22
|
+
- Follow the approved plan and your brief exactly: they are the Master's
|
|
23
|
+
decisions. Do not substitute your own design, rename things, or widen the
|
|
24
|
+
step. If the code contradicts the brief, do not improvise: report it.
|
|
23
25
|
- Follow domain boundaries.
|
|
24
26
|
|
|
25
27
|
If you need a new dependency, or you believe a significant architectural
|
|
@@ -73,7 +75,10 @@ repository at the same time, and files are checked out like physical documents:
|
|
|
73
75
|
|
|
74
76
|
## Before handoff
|
|
75
77
|
|
|
76
|
-
-
|
|
78
|
+
- Go through the brief's "Done when" list one criterion at a time and record
|
|
79
|
+
each in `## Brief Check` as `- criterion — met|not met — evidence`.
|
|
80
|
+
- Inspect the actual diff: it must contain what the brief asked for and
|
|
81
|
+
nothing else.
|
|
77
82
|
- Verify tests.
|
|
78
83
|
- Update the temporary task scratchpad.
|
|
79
84
|
- Report concise results.
|
package/src/agents/backend.ts
CHANGED
|
@@ -4,7 +4,7 @@ export const backendSpec: DomainSpec = {
|
|
|
4
4
|
domain: "backend",
|
|
5
5
|
promptFile: "backend.md",
|
|
6
6
|
scoutFocus:
|
|
7
|
-
"API, business logic, data models, persistence, authentication/authorization, integrations, and backend reliability",
|
|
7
|
+
"server-side code: API, business logic, data models, persistence, authentication/authorization, integrations, and backend reliability",
|
|
8
8
|
boundary:
|
|
9
|
-
"Stay inside
|
|
9
|
+
"Stay inside server-side code. Do not modify browser code (pages, components, client-side logic or graphics); report frontend requirements to the Master.",
|
|
10
10
|
};
|
package/src/agents/designer.ts
CHANGED
|
@@ -4,7 +4,7 @@ export const designerSpec: DomainSpec = {
|
|
|
4
4
|
domain: "designer",
|
|
5
5
|
promptFile: "designer.md",
|
|
6
6
|
scoutFocus:
|
|
7
|
-
"UI/UX,
|
|
7
|
+
"UI/UX and everything that runs in the browser (pages, components, client-side logic, canvas and WebGL/three.js graphics), accessibility, responsive behavior, and the existing design language",
|
|
8
8
|
boundary:
|
|
9
|
-
"Stay inside frontend
|
|
9
|
+
"Stay inside the frontend: everything that runs in the browser. Do not modify server-side code; report backend dependencies to the Master.",
|
|
10
10
|
};
|
package/src/ask/dialog.ts
CHANGED
|
@@ -83,7 +83,17 @@ export class AskDialog implements Component {
|
|
|
83
83
|
* cannot draw the questionnaire. Without any UI nobody can answer: the
|
|
84
84
|
* result says the questions were put away.
|
|
85
85
|
*/
|
|
86
|
-
export const askUser: Asker =
|
|
86
|
+
export const askUser: Asker = (questions, ctx, signal, from) => {
|
|
87
|
+
// A model may call the tool several times in one turn; pi runs them in parallel, and overlays opened together hide each other. One at a time.
|
|
88
|
+
const next = asking.then(() => (signal?.aborted ? { answers: [], cancelled: true } : askNow(questions, ctx, signal, from)));
|
|
89
|
+
asking = next.then(() => undefined, () => undefined);
|
|
90
|
+
return next;
|
|
91
|
+
};
|
|
92
|
+
|
|
93
|
+
/** Tail of the questions waiting their turn. */
|
|
94
|
+
let asking: Promise<void> = Promise.resolve();
|
|
95
|
+
|
|
96
|
+
const askNow: Asker = async (questions, ctx, signal, from) => {
|
|
87
97
|
if (!ctx.hasUI || questions.length === 0) return { answers: [], cancelled: true };
|
|
88
98
|
if ((ctx as { mode?: string }).mode === "rpc") return dialogAsker(questions, ctx, signal, from);
|
|
89
99
|
// Option images are read before the questionnaire opens, so drawing it never waits on the disk.
|