@a-t-h-i/bot-lobby 0.6.4 → 0.6.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +81 -9
- package/package.json +12 -3
- package/prompts/backend.md +3 -1
- package/prompts/designer.md +24 -1
- package/prompts/master.md +51 -18
- package/prompts/quickfix.md +16 -3
- package/src/agents/backend.ts +2 -2
- package/src/agents/designer.ts +2 -2
- package/src/classifier/triage.ts +24 -7
- package/src/index.ts +3 -0
- package/src/lobby/feed.ts +6 -1
- package/src/lobby/quickfix.ts +29 -3
- package/src/lobby/runtime.ts +27 -4
- package/src/lobby/tabs/quickfix.ts +1 -1
- package/src/lobby/tabs/tasks.ts +2 -1
- package/src/lobby/view.ts +9 -0
- package/src/master/decisions.ts +10 -2
- package/src/pi/commands.ts +22 -11
- package/src/pi/events.ts +2 -1
- package/src/pi/route.ts +180 -0
- package/src/pi/start-task.ts +51 -5
- package/src/pi/tools.ts +8 -3
- package/src/roles/worker.ts +10 -2
- package/src/schemas/configuration.ts +11 -0
- package/src/schemas/task.ts +31 -0
- package/src/workflow/track.ts +436 -0
- package/src/workflow/workflow.ts +148 -10
package/README.md
CHANGED
|
@@ -2,7 +2,11 @@
|
|
|
2
2
|
|
|
3
3
|
A [Pi](https://pi.dev) extension that turns Pi into a multi-agent software team.
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+

|
|
6
|
+
|
|
7
|
+
`/bot-lobby <request>` (or a request typed in the lobby) starts a task, unless
|
|
8
|
+
one agent can simply do it: then it goes to the [quick-fix agent](#quick-fix-or-the-team).
|
|
9
|
+
For a task, your Pi session becomes the **Master**
|
|
6
10
|
(the "oracle"): it scouts the codebase, proposes a plan, and delegates the
|
|
7
11
|
work to three domain agents — **Designer+Frontend**, **Backend** and **QA** —
|
|
8
12
|
each running in its own isolated `pi` process. QA's reviewer is the quality
|
|
@@ -41,7 +45,7 @@ extra instructions.
|
|
|
41
45
|
| Command | Does |
|
|
42
46
|
| --- | --- |
|
|
43
47
|
| `/bot-lobby` | Open the lobby (`alt+l`) |
|
|
44
|
-
| `/bot-lobby <request>` | Start a task (`--task`
|
|
48
|
+
| `/bot-lobby <request>` | Start a request: a [quick fix](#quick-fix-or-the-team) when one agent can do it alone, else a task (`--task` to always make it a task, also when it begins with a command word; `--auto` to run unattended, `--budget 90m` to give it a time budget, `--fast` / `--full` to pick its [track](#fast-track-or-full-workflow)) |
|
|
45
49
|
| `/bot-lobby budget [90m\|off]` | Show or set this session's task time budget |
|
|
46
50
|
| `/bot-lobby status \| tasks \| runs [id]` | Current task, all tasks, recent agent runs |
|
|
47
51
|
| `/bot-lobby approve \| amend <text> \| decline` | Answer the proposal |
|
|
@@ -54,15 +58,82 @@ extra instructions.
|
|
|
54
58
|
| `/bot-lobby knowledge` | Knowledge file sizes |
|
|
55
59
|
| `/bot-lobby minimize \| restore` | Hide bot-lobby in this session (`ctrl+shift+m`) |
|
|
56
60
|
|
|
61
|
+
## Quick fix or the team
|
|
62
|
+
|
|
63
|
+
Before any task exists, bot-lobby asks whether **one agent can just do it**:
|
|
64
|
+
in one file or one area, with nothing to agree between frontend and backend,
|
|
65
|
+
no unfamiliar codebase to survey, nothing risky and no decision you must make
|
|
66
|
+
first. [Jev](#the-classifier-jev) answers when it is on (`classifier.thresholds.quickFixAt`,
|
|
67
|
+
0.7); plain rules answer otherwise, and whenever Jev is unsure. A request that
|
|
68
|
+
says it is self-contained ("a single page", "in one html file") counts even
|
|
69
|
+
when it is rich.
|
|
70
|
+
|
|
71
|
+
When it reads that way, the oracle confirms in one step, without reading
|
|
72
|
+
files or planning (`route_request`), and says so: *this looks like a quick
|
|
73
|
+
feature, the quick-fix agent is on it*. The lobby then hands the request to
|
|
74
|
+
the quick-fix agent and switches to the **Quick fix** tab, where you follow
|
|
75
|
+
it. No scouts, proposal, plan or QA. A quick feature (bigger than a small
|
|
76
|
+
change, but in one place) runs on its builder's model, thinking and time limit
|
|
77
|
+
(DESIGN's for a page) instead of the quick-fix defaults.
|
|
78
|
+
|
|
79
|
+
Everything else, or anything the oracle judges needs the team, starts as a
|
|
80
|
+
task below. `--task` always makes a task; `workflow.routeQuickFixes: false`
|
|
81
|
+
turns routing off. Without the lobby (a background session, RPC mode) every
|
|
82
|
+
request is a task.
|
|
83
|
+
|
|
57
84
|
## How a task runs
|
|
58
85
|
|
|
59
86
|
```
|
|
60
|
-
request → clarify → scout → propose → approve → plan → implement → QA gate → complete
|
|
87
|
+
full workflow: request → clarify → scout → propose → approve → plan → implement → QA gate → complete
|
|
88
|
+
fast track: request → implement (only the agents it needs) → QA, if it needs tests → complete
|
|
61
89
|
```
|
|
62
90
|
|
|
91
|
+
### Fast track or full workflow
|
|
92
|
+
|
|
93
|
+
The moment a task starts, bot-lobby reads the request and decides how serious
|
|
94
|
+
it is: its size, who has to take part, and whether it takes the **fast
|
|
95
|
+
track** or the **full workflow**. The read is instant and costs no tokens
|
|
96
|
+
(plain rules, refined by [Jev](#the-classifier-jev) when that is on).
|
|
97
|
+
|
|
98
|
+
| The request… | Who takes part |
|
|
99
|
+
| --- | --- |
|
|
100
|
+
| changes anything that runs in the browser: screens, components, styles, copy, canvas or three.js graphics | DESIGN (frontend) |
|
|
101
|
+
| changes an API, the database, auth, jobs or other server-side code | DEV (backend) |
|
|
102
|
+
| needs tests: asks for them, fixes a bug, or changes backend logic | QA |
|
|
103
|
+
| depends on outside facts: latest versions, docs, standards, third-party APIs | RESEARCH |
|
|
104
|
+
|
|
105
|
+
A **small, clear, low-risk** request takes the fast track: the oracle hands
|
|
106
|
+
it straight to those agents, with no scouts, no proposal to approve and no
|
|
107
|
+
plan document (the engine keeps a short plan whose steps are the
|
|
108
|
+
delegations, so the checklist still works). QA takes part only when the
|
|
109
|
+
change needs tests (its worker writing and running them as the last step, or
|
|
110
|
+
the QA gate); without it the task completes as soon as the work is in and
|
|
111
|
+
checked. A copy change is one agent run.
|
|
112
|
+
|
|
113
|
+
Everything else takes the full workflow below: anything **medium or large**
|
|
114
|
+
(a new page, flow or endpoint, a refactor, an upgrade, a vague or
|
|
115
|
+
many-part request), **serious** whatever its size (security, auth,
|
|
116
|
+
passwords, payments, migrations, production, personal data), or **unclear**
|
|
117
|
+
(too short, vague, or asking to investigate first).
|
|
118
|
+
|
|
119
|
+
- The oracle glances at the read once and acts on it, or corrects it with
|
|
120
|
+
`orchestrate action=track`: the full workflow for a change bigger than it
|
|
121
|
+
reads, the fast track for one that is smaller (before its work is planned),
|
|
122
|
+
or a member added or dropped.
|
|
123
|
+
- Once work is under way a track only gets stricter: the full workflow or
|
|
124
|
+
more members, never QA dropped. A fast task whose worker asks for a new
|
|
125
|
+
dependency or an architecture change gets QA.
|
|
126
|
+
- `/bot-lobby --fast <request>` or `--full <request>` decides it yourself;
|
|
127
|
+
the oracle never moves a `--full` task to the fast track.
|
|
128
|
+
`workflow.fastTrack: false` puts every task on the full workflow.
|
|
129
|
+
- The track shows in the lobby's activity log, the task's details on the
|
|
130
|
+
Tasks tab, and `/bot-lobby status`.
|
|
131
|
+
|
|
132
|
+
### The full workflow
|
|
133
|
+
|
|
63
134
|
- **Scouts** (read-only) investigate the domains the request touches.
|
|
64
135
|
- The Master **proposes** a short bullet list; nothing is built until you
|
|
65
|
-
approve it.
|
|
136
|
+
approve it.
|
|
66
137
|
- **Workers** implement plan steps, one domain each. Several can run in
|
|
67
138
|
parallel; they share files through a **file desk** (claim a file, queue for
|
|
68
139
|
a busy one, hand it over with a note).
|
|
@@ -124,7 +195,7 @@ open. `alt+h` lists every key.
|
|
|
124
195
|
| **1 Lobby** | The task's status, your conversation with the oracle, an activity log of every tool call, and each agent's latest thought |
|
|
125
196
|
| **2 Tasks** | Every task and saved plan as a checklist. `s` starts a plan in a new session, `h` here; `c` comments on a plan; `a` archives, `d` deletes |
|
|
126
197
|
| **3 Plan** | Plan a task with a panel of agents before building it (below) |
|
|
127
|
-
| **4 Quick fix** | One agent makes a
|
|
198
|
+
| **4 Quick fix** | One agent makes a change right away, beside any running task; requests the oracle [routes here](#quick-fix-or-the-team) show up too |
|
|
128
199
|
| **5 Metrics** | Run time, success rate, tokens and cost per model and agent |
|
|
129
200
|
|
|
130
201
|
Common keys: `tab` switches tabs, `esc` browses (arrows, single-key
|
|
@@ -236,7 +307,8 @@ else TypeSafe.
|
|
|
236
307
|
| Planning seats | Each round, only the seats the idea or your latest answers touch sit; `1`–`4` pins a seat |
|
|
237
308
|
| Obvious answers | Answers a question itself when the conversation already makes the recommended option clearly right (≥ 0.9); listed under Assumptions |
|
|
238
309
|
| File hints | Agents start with a short list of the files they most likely need, and get a `find_relevant_files` tool |
|
|
239
|
-
|
|
|
310
|
+
| Quick fix or task | Whether one engineer can do a new request alone decides whether it goes to the [quick-fix agent](#quick-fix-or-the-team) (the oracle confirms) |
|
|
311
|
+
| Task triage | The task's [track](#fast-track-or-full-workflow) and roster use its read (size, domains, research, ambiguity), and the Master gets it as hints; a quick fix that is really a task (large, and not one engineer's work) is held (`r` run anyway, `t` make it a task) |
|
|
240
312
|
| Effort routing | Simple steps run one thinking level lower; trivial ones on a **cheaper model** you pick. A routed run that falls short re-runs on your normal settings |
|
|
241
313
|
|
|
242
314
|
**It never gets in the way:** any failure, timeout or missing key means
|
|
@@ -264,7 +336,7 @@ the result.
|
|
|
264
336
|
"scout": { "model": "anthropic/claude-haiku-4-5-20251001", "timeoutMs": 480000 },
|
|
265
337
|
"planner": { "thinking": "high", "timeoutMs": 300000 },
|
|
266
338
|
"lobby": { "planningPanel": ["backend", "designer", "qa", "researcher"], "maxPlanningRounds": 5 },
|
|
267
|
-
"workflow": { "maxReviewIterations": 2, "maxParallelWorkers": 3, "stallTimeoutMs": 300000, "wrapUpAt": 0.75, "taskBudgetMinutes": 0 },
|
|
339
|
+
"workflow": { "maxReviewIterations": 2, "maxParallelWorkers": 3, "stallTimeoutMs": 300000, "wrapUpAt": 0.75, "taskBudgetMinutes": 0, "fastTrack": true, "routeQuickFixes": true },
|
|
268
340
|
"classifier": { "enabled": false, "provider": "auto", "effort": { "cheapModel": "inherit" } }
|
|
269
341
|
}
|
|
270
342
|
```
|
|
@@ -283,11 +355,11 @@ the result.
|
|
|
283
355
|
| Rule | How |
|
|
284
356
|
| --- | --- |
|
|
285
357
|
| Steps happen in order | A state machine validates every action |
|
|
286
|
-
| Nothing is built before approval | `implement` refuses earlier states |
|
|
358
|
+
| Nothing is built before approval | On the full workflow `implement` refuses earlier states; only a fast-track task (small, clear, low-risk) starts straight away |
|
|
287
359
|
| Scouts and the QA gate can't edit code | Scouts get read-only tools; the QA gate adds only `bash` for tests |
|
|
288
360
|
| New dependencies and architecture changes need approval | Parsed from worker reports; the domain is blocked until resolved |
|
|
289
361
|
| Only the Master writes knowledge | Agents can only propose it |
|
|
290
|
-
| "Done" is earned | Needs a plan, a passing QA gate that ran checks (or your explicit acceptance), and no open blockers |
|
|
362
|
+
| "Done" is earned | Needs a plan, a passing QA gate that ran checks (or your explicit acceptance), and no open blockers; on the fast track, a finished worker step, and QA's part only when the change needs tests |
|
|
291
363
|
| Parallel workers don't clobber files | Edits need a file-desk claim |
|
|
292
364
|
| A crash doesn't corrupt a task | State is on disk; tasks resume from their state |
|
|
293
365
|
|
package/package.json
CHANGED
|
@@ -1,11 +1,19 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@a-t-h-i/bot-lobby",
|
|
3
|
-
"version": "0.6.
|
|
3
|
+
"version": "0.6.5",
|
|
4
4
|
"description": "Structured multi-agent software engineering orchestrator for Pi",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "Apache-2.0",
|
|
7
7
|
"keywords": [
|
|
8
|
-
"pi-package"
|
|
8
|
+
"pi-package",
|
|
9
|
+
"pi",
|
|
10
|
+
"pi-extension",
|
|
11
|
+
"pi-coding-agent",
|
|
12
|
+
"multi-agent",
|
|
13
|
+
"subagents",
|
|
14
|
+
"orchestrator",
|
|
15
|
+
"code-review",
|
|
16
|
+
"web-search"
|
|
9
17
|
],
|
|
10
18
|
"repository": {
|
|
11
19
|
"type": "git",
|
|
@@ -25,7 +33,8 @@
|
|
|
25
33
|
"pi": {
|
|
26
34
|
"extensions": [
|
|
27
35
|
"./src/index.ts"
|
|
28
|
-
]
|
|
36
|
+
],
|
|
37
|
+
"image": "https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/gallery.png"
|
|
29
38
|
},
|
|
30
39
|
"scripts": {
|
|
31
40
|
"typecheck": "tsc --noEmit",
|
package/prompts/backend.md
CHANGED
|
@@ -2,7 +2,9 @@
|
|
|
2
2
|
|
|
3
3
|
You own backend engineering: API, business logic, data models, database,
|
|
4
4
|
authentication, authorization, integrations, backend architecture, security,
|
|
5
|
-
reliability and backend performance.
|
|
5
|
+
reliability and backend performance. Code that runs in the browser — pages,
|
|
6
|
+
client-side logic, canvas and WebGL/three.js graphics — is the designer's,
|
|
7
|
+
however much logic it holds.
|
|
6
8
|
|
|
7
9
|
## API design
|
|
8
10
|
|
package/prompts/designer.md
CHANGED
|
@@ -2,7 +2,9 @@
|
|
|
2
2
|
|
|
3
3
|
You own UI/UX and frontend engineering: user experience, interaction design,
|
|
4
4
|
visual consistency, frontend implementation, responsive behavior,
|
|
5
|
-
accessibility, frontend performance and the design language.
|
|
5
|
+
accessibility, frontend performance and the design language. Everything that
|
|
6
|
+
runs in the browser is yours, including client-side logic and canvas,
|
|
7
|
+
WebGL/three.js graphics.
|
|
6
8
|
|
|
7
9
|
You have real visual taste. Your work is calm, considered and quietly
|
|
8
10
|
delightful: the understated, crafted aesthetic of Anthropic's latest models,
|
|
@@ -66,6 +68,27 @@ one-off values.
|
|
|
66
68
|
- **No magic numbers:** every size, space, color, radius, shadow and duration
|
|
67
69
|
comes from a token or the existing scale.
|
|
68
70
|
|
|
71
|
+
## Graphics and 3D
|
|
72
|
+
|
|
73
|
+
When the work is a scene (canvas, WebGL, three.js), how it looks is the
|
|
74
|
+
product, and a generic render is a failed one.
|
|
75
|
+
|
|
76
|
+
- **Light it like a photograph:** image-based lighting (an environment map,
|
|
77
|
+
e.g. PMREM with `RoomEnvironment`) plus one key light with soft shadows;
|
|
78
|
+
ACES or AgX tone mapping and sRGB output. Never flat ambient light.
|
|
79
|
+
- **Physical materials:** `MeshPhysicalMaterial` with values from the real
|
|
80
|
+
thing — glass with transmission, thickness, IOR about 1.5 and a faint tint;
|
|
81
|
+
wood, brass or stone with sensible roughness and metalness. Liquids and
|
|
82
|
+
grains read by their own color, sheen and scale, not by a texture.
|
|
83
|
+
- **Proportions and framing:** model the object after a real reference; frame
|
|
84
|
+
it so it fills most of the view at rest, on every screen size, with a quiet
|
|
85
|
+
backdrop (a soft gradient, a floor that fades out). No default grey planes,
|
|
86
|
+
giant ground discs or horizon lines cutting through the shot.
|
|
87
|
+
- **Restraint:** fewer effects done well beat many done badly; hold 60fps on a
|
|
88
|
+
phone with quality tiers rather than dropping detail everywhere.
|
|
89
|
+
- **Look at it:** render a frame when the project has a headless browser, and
|
|
90
|
+
fix what you see; otherwise say in your report that it was not seen.
|
|
91
|
+
|
|
69
92
|
## Interaction and motion
|
|
70
93
|
|
|
71
94
|
You like interactive UIs that give the user subtle, fun feedback — never
|
package/prompts/master.md
CHANGED
|
@@ -17,32 +17,55 @@ validates every step — state transitions, role permissions, approval gates and
|
|
|
17
17
|
completion authority. If it rejects an action, read the error and adjust; never
|
|
18
18
|
work around it. Do not rely on prompts to enforce permissions or state.
|
|
19
19
|
|
|
20
|
+
## Task track
|
|
21
|
+
|
|
22
|
+
Every task starts on a track the engine read from the request (the `Track`
|
|
23
|
+
line in your context): how serious it is, and so who takes part and how much
|
|
24
|
+
process it gets. Fewer steps win whenever the result is the same.
|
|
25
|
+
|
|
26
|
+
- **Fast track** — a small, clear, low-risk change. No scouts, no proposal,
|
|
27
|
+
no plan document: delegate straight away with `orchestrate action=implement`,
|
|
28
|
+
opening each task with `Step N:` (the engine keeps the plan and the
|
|
29
|
+
checklist). Only the roster takes part: DESIGN for frontend work, DEV for
|
|
30
|
+
backend work, QA when the change needs tests (its worker writing and
|
|
31
|
+
running them as the last step, or the QA gate), and the researcher when a
|
|
32
|
+
decision needs outside facts (summon it first). Several domains: one
|
|
33
|
+
`implement` with `assignments`, each task stating the contract between
|
|
34
|
+
them. When the work is in, check `git diff --stat` and the report, then
|
|
35
|
+
`complete`; without QA on the roster there is no QA gate.
|
|
36
|
+
- **Full workflow** — everything else: the steps below, ending with the QA
|
|
37
|
+
gate.
|
|
38
|
+
|
|
39
|
+
The read is quick and can be wrong, so glance at it once and move on: confirm
|
|
40
|
+
it by acting on it, or correct it with `orchestrate action=track` (`track`,
|
|
41
|
+
`roster`, `reason`). Go full when the change is bigger, riskier (security,
|
|
42
|
+
money, data, migrations, production) or less clear than it reads; take the
|
|
43
|
+
fast track when a full-workflow task turns out small and clear (before its
|
|
44
|
+
work is planned). Add a member the roster is missing, or drop one it does not
|
|
45
|
+
need, before delegating. Once work is under way a track only gets stricter:
|
|
46
|
+
the full workflow or more members, never QA dropped; a fast task whose worker
|
|
47
|
+
asks for a dependency or an architecture change gets QA added. The user's
|
|
48
|
+
`--full` and a disabled fast track keep the full workflow.
|
|
49
|
+
|
|
20
50
|
## Before implementation
|
|
21
51
|
|
|
22
|
-
For
|
|
52
|
+
For full-workflow work: understand the request; clarify with
|
|
23
53
|
`orchestrate action=clarify` when necessary; challenge it when there is a real
|
|
24
54
|
technical, security, reliability, UX or maintainability concern; select and run
|
|
25
55
|
relevant Scouts; review findings and target-verify important claims against the
|
|
26
56
|
repository; synthesize and present a short `- ` bullet-list proposal; then wait for
|
|
27
|
-
approval, amendment, or decline. Do not start
|
|
28
|
-
approval.
|
|
29
|
-
|
|
30
|
-
For a trivial, single-domain request you may skip the Scout round and the
|
|
31
|
-
proposal ceremony: state the short plan, delegate the step, and verify the diff
|
|
32
|
-
directly. The engine allows `clarifying -> awaiting_approval -> planning`, so no
|
|
33
|
-
state override is needed. Skip only when the change is small, obvious and
|
|
34
|
-
confined to one domain.
|
|
57
|
+
approval, amendment, or decline. Do not start full-workflow implementation
|
|
58
|
+
before approval.
|
|
35
59
|
|
|
36
60
|
## Classifier hints
|
|
37
61
|
|
|
38
62
|
When the classifier is on, your task context carries a **Classifier
|
|
39
63
|
triage**: a fast model's read of the request (size, the domains it touches,
|
|
40
64
|
whether it needs outside research, whether it is ambiguous, its kind, likely
|
|
41
|
-
files) and a suggested path
|
|
42
|
-
scout only the domains it marks (0.5 or
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
request as ambiguous — and overrule it whenever the repository says
|
|
65
|
+
files) and a suggested path; the task's track was chosen with it. Use it to
|
|
66
|
+
skip reasoning you do not need — scout only the domains it marks (0.5 or
|
|
67
|
+
more), skip the researcher when research is not needed, clarify only when it
|
|
68
|
+
reads the request as ambiguous — and overrule it whenever the repository says
|
|
46
69
|
otherwise. It is a hint, never a rule.
|
|
47
70
|
|
|
48
71
|
When you `clarify` with options, put your recommended option first and mark
|
|
@@ -119,6 +142,14 @@ Assign work to the correct domain; never ask one domain to do another's. A
|
|
|
119
142
|
cross-domain dependency is reported to you, and you decide whether another
|
|
120
143
|
domain needs a task.
|
|
121
144
|
|
|
145
|
+
Assign by where the code runs, not by how much logic it holds. Everything that
|
|
146
|
+
runs in the browser — pages, components, client-side state and logic, canvas,
|
|
147
|
+
WebGL/three.js scenes, shaders, client-side physics — is DESIGN's
|
|
148
|
+
(Designer+Frontend); DEV (backend) owns server-side code, data, APIs and
|
|
149
|
+
integrations. One file has one owner: never split a file between domains, and
|
|
150
|
+
never give DESIGN only the styling of something another domain built —
|
|
151
|
+
whoever builds a visual thing owns how it looks.
|
|
152
|
+
|
|
122
153
|
Write the plan's steps as a numbered list under a `## Steps` heading, and open
|
|
123
154
|
each `implement` task with its step number (`Step 3: ...`, or `Steps 3-4: ...`
|
|
124
155
|
when one delegation covers several) so the user's checklist tracks progress
|
|
@@ -193,7 +224,8 @@ redundant, speculative or temporary information.
|
|
|
193
224
|
|
|
194
225
|
The repository state is the source of truth; do not blindly trust Scout or
|
|
195
226
|
Worker reports. There is one review, the QA gate (`orchestrate action=qa`). Run
|
|
196
|
-
it once the implementation steps are complete
|
|
227
|
+
it once the implementation steps are complete (on the fast track, only when
|
|
228
|
+
QA is on the roster and its worker is not the last step). A `changes_required` verdict
|
|
197
229
|
goes back to the owning domain as a fix step, then the gate runs again; hitting
|
|
198
230
|
the configured limit blocks the task. On a pass, record knowledge and continue.
|
|
199
231
|
|
|
@@ -217,9 +249,10 @@ breaks this task.
|
|
|
217
249
|
## Completion
|
|
218
250
|
|
|
219
251
|
Only you declare completion, and only after requirements are satisfied,
|
|
220
|
-
implementation is verified, required tests pass, the QA gate passes
|
|
221
|
-
|
|
222
|
-
|
|
252
|
+
implementation is verified, required tests pass, the QA gate passes (on the
|
|
253
|
+
fast track: QA has taken part when it is on the roster), critical blockers are
|
|
254
|
+
resolved, and relevant knowledge and decisions are recorded — never just
|
|
255
|
+
because a Worker says it is done.
|
|
223
256
|
|
|
224
257
|
## Architect partnership
|
|
225
258
|
|
package/prompts/quickfix.md
CHANGED
|
@@ -14,15 +14,28 @@ the change now, the way they would ask pi directly.
|
|
|
14
14
|
refactor, rename or tidy anything else.
|
|
15
15
|
- If the request is ambiguous, pick the most reasonable reading and say which
|
|
16
16
|
one you chose in your report; do not stop to ask.
|
|
17
|
-
- If the change turns out to be large (many files, a new
|
|
18
|
-
architecture change), make no edits and report
|
|
19
|
-
user can plan it as a task instead.
|
|
17
|
+
- If the change turns out to be large (many files across the codebase, a new
|
|
18
|
+
dependency to install, an architecture change), make no edits and report
|
|
19
|
+
what it would take, so the user can plan it as a task instead. Something
|
|
20
|
+
new that lives in one place — a page, a component, a script, even a rich one
|
|
21
|
+
in a single file — is not large: it is a quick feature, so build it whole
|
|
22
|
+
and finished.
|
|
20
23
|
- Run a quick targeted check when one exists (the nearest test file, a
|
|
21
24
|
typecheck of the touched package) with a bash `timeout`; never start dev
|
|
22
25
|
servers, watchers or background processes.
|
|
23
26
|
- Change files with `edit`/`write`, never through shell redirection or
|
|
24
27
|
`sed -i`.
|
|
25
28
|
|
|
29
|
+
## Anything people look at
|
|
30
|
+
|
|
31
|
+
When the request is visual (a page, a UI, a game, a canvas or three.js scene),
|
|
32
|
+
how it looks is part of done: a coherent palette, real materials and lighting
|
|
33
|
+
(environment lighting, soft shadows, tone mapping for 3D), proportions that
|
|
34
|
+
read as the real object, framing that fills the view on phone and desktop, and
|
|
35
|
+
no placeholder geometry or default grey planes left in the shot. Prefer fewer
|
|
36
|
+
effects done well over many done badly. Render a frame and look at it when the
|
|
37
|
+
project has a headless browser; otherwise say it was not seen.
|
|
38
|
+
|
|
26
39
|
## Other agents
|
|
27
40
|
|
|
28
41
|
A bot-lobby task may be running at the same time in this working tree. Touch
|
package/src/agents/backend.ts
CHANGED
|
@@ -4,7 +4,7 @@ export const backendSpec: DomainSpec = {
|
|
|
4
4
|
domain: "backend",
|
|
5
5
|
promptFile: "backend.md",
|
|
6
6
|
scoutFocus:
|
|
7
|
-
"API, business logic, data models, persistence, authentication/authorization, integrations, and backend reliability",
|
|
7
|
+
"server-side code: API, business logic, data models, persistence, authentication/authorization, integrations, and backend reliability",
|
|
8
8
|
boundary:
|
|
9
|
-
"Stay inside
|
|
9
|
+
"Stay inside server-side code. Do not modify browser code (pages, components, client-side logic or graphics); report frontend requirements to the Master.",
|
|
10
10
|
};
|
package/src/agents/designer.ts
CHANGED
|
@@ -4,7 +4,7 @@ export const designerSpec: DomainSpec = {
|
|
|
4
4
|
domain: "designer",
|
|
5
5
|
promptFile: "designer.md",
|
|
6
6
|
scoutFocus:
|
|
7
|
-
"UI/UX,
|
|
7
|
+
"UI/UX and everything that runs in the browser (pages, components, client-side logic, canvas and WebGL/three.js graphics), accessibility, responsive behavior, and the existing design language",
|
|
8
8
|
boundary:
|
|
9
|
-
"Stay inside frontend
|
|
9
|
+
"Stay inside the frontend: everything that runs in the browser. Do not modify server-side code; report backend dependencies to the Master.",
|
|
10
10
|
};
|
package/src/classifier/triage.ts
CHANGED
|
@@ -48,6 +48,15 @@ function sizeQuestion(subject: string) {
|
|
|
48
48
|
return score(`How big is the change ${subject} asks for, judged by what it would take to build and verify in this repository?`, SIZE_LEVELS);
|
|
49
49
|
}
|
|
50
50
|
|
|
51
|
+
/** Whether one agent can just do it: the quick fix or the team. */
|
|
52
|
+
function soloQuestion(subject: string) {
|
|
53
|
+
return noul(
|
|
54
|
+
`Could one engineer do ${subject} alone, right away — in one file or one area of the code — with no contract to agree between frontend and backend, no unfamiliar codebase to survey first, no risky change (security, data, payments, migrations) and no decision the user must make first?`,
|
|
55
|
+
"Yes: one person can just do it now, even if it is a rich piece of work in one place.",
|
|
56
|
+
"No: it needs a team — several areas, a survey of the codebase, a plan to agree, or review before it lands.",
|
|
57
|
+
);
|
|
58
|
+
}
|
|
59
|
+
|
|
51
60
|
export function triageRequest(input: TriageInput): SystemOneRequest {
|
|
52
61
|
const questions: SystemOneRequest["questions"] = {
|
|
53
62
|
size: sizeQuestion("`request`"),
|
|
@@ -62,6 +71,7 @@ export function triageRequest(input: TriageInput): SystemOneRequest {
|
|
|
62
71
|
"No: a competent engineer could build it as written, deciding details from the codebase.",
|
|
63
72
|
),
|
|
64
73
|
kind: choice("What kind of work does `request` ask for?", TRIAGE_KINDS),
|
|
74
|
+
solo: soloQuestion("`request`"),
|
|
65
75
|
};
|
|
66
76
|
for (const { domain, owns } of input.domains) {
|
|
67
77
|
questions[`domain_${domain}`] = noul(`Does building \`request\` require changes in this area: ${owns}?`);
|
|
@@ -85,12 +95,14 @@ export async function triageTask(classifier: Classifier, input: TriageInput, sig
|
|
|
85
95
|
if (probability !== undefined) domains[domain] = probability;
|
|
86
96
|
}
|
|
87
97
|
const kind = choiceOf(result.answers, "kind");
|
|
98
|
+
const solo = yesOf(result.answers, "solo");
|
|
88
99
|
return {
|
|
89
100
|
size: TRIAGE_SIZES[Math.max(0, Math.min(TRIAGE_SIZES.length - 1, size.level))]!,
|
|
90
101
|
sizeConfidence: size.confidence,
|
|
91
102
|
domains,
|
|
92
103
|
research: yesOf(result.answers, "needs_research") ?? 0,
|
|
93
104
|
ambiguous: yesOf(result.answers, "ambiguous") ?? 0,
|
|
105
|
+
...(solo !== undefined ? { solo } : {}),
|
|
94
106
|
...(kind ? { kind: kind.choice, kindProbability: kind.probability } : {}),
|
|
95
107
|
at: new Date().toISOString(),
|
|
96
108
|
};
|
|
@@ -131,20 +143,25 @@ export async function triageWithContext(classifier: Classifier, scope: FileScope
|
|
|
131
143
|
return files.length > 0 ? { ...triage, likelyFiles: files } : triage;
|
|
132
144
|
}
|
|
133
145
|
|
|
134
|
-
/**
|
|
135
|
-
|
|
146
|
+
/**
|
|
147
|
+
* The size of a quick fix prompt, and how likely one engineer can do it alone,
|
|
148
|
+
* or undefined when the classifier is off or fails. A large prompt is held as
|
|
149
|
+
* a task unless it is still one engineer's work (a rich page in one file).
|
|
150
|
+
*/
|
|
151
|
+
export async function quickFixSize(classifier: Classifier, prompt: string, signal?: AbortSignal): Promise<{ size: TriageSize; confidence: number; solo?: number } | undefined> {
|
|
136
152
|
if (!classifier.enabled("triage")) return undefined;
|
|
137
|
-
const largeAt = classifier.config.thresholds
|
|
138
|
-
const result = await classifier.ask("triage", { state: { prompt: clip(prompt, 6000) }, questions: { size: sizeQuestion("`prompt`") } }, {
|
|
153
|
+
const { quickFixLargeAt: largeAt, quickFixAt } = classifier.config.thresholds;
|
|
154
|
+
const result = await classifier.ask("triage", { state: { prompt: clip(prompt, 6000) }, questions: { size: sizeQuestion("`prompt`"), solo: soloQuestion("`prompt`") } }, {
|
|
139
155
|
...(signal ? { signal } : {}),
|
|
140
156
|
saved: (answers) => {
|
|
141
157
|
const sized = scoreOf(answers, "size");
|
|
142
|
-
return sized && sized.level >= TRIAGE_SIZES.length - 1 && sized.confidence >= largeAt ? 1 : 0;
|
|
158
|
+
return sized && sized.level >= TRIAGE_SIZES.length - 1 && sized.confidence >= largeAt && (yesOf(answers, "solo") ?? 0) < quickFixAt ? 1 : 0;
|
|
143
159
|
},
|
|
144
160
|
});
|
|
145
161
|
const size = scoreOf(result?.answers, "size");
|
|
146
162
|
if (!size) return undefined;
|
|
147
|
-
|
|
163
|
+
const solo = yesOf(result?.answers, "solo");
|
|
164
|
+
return { size: TRIAGE_SIZES[Math.max(0, Math.min(TRIAGE_SIZES.length - 1, size.level))]!, confidence: size.confidence, ...(solo !== undefined ? { solo } : {}) };
|
|
148
165
|
}
|
|
149
166
|
|
|
150
167
|
const HINT = 0.5;
|
|
@@ -155,7 +172,7 @@ export function suggestedPath(triage: TaskTriage): string {
|
|
|
155
172
|
if (triage.ambiguous >= HINT) return "clarify first: the request as written probably misses a decision.";
|
|
156
173
|
const smallish = (triage.size === "trivial" || triage.size === "small") && triage.sizeConfidence >= 0.6;
|
|
157
174
|
const steps: string[] = [];
|
|
158
|
-
if (smallish && touched.length === 1) steps.push(`
|
|
175
|
+
if (smallish && touched.length === 1) steps.push(`fast track (${touched[0]}): no scouts, proposal or plan; delegate straight away`);
|
|
159
176
|
else if (touched.length > 0) steps.push(`scout only ${touched.join(", ")}`);
|
|
160
177
|
if (triage.research >= HINT) steps.push("summon the researcher for the outside facts");
|
|
161
178
|
return steps.length > 0 ? `${steps.join("; ")}.` : "no strong signal; decide from the request.";
|
package/src/index.ts
CHANGED
|
@@ -2,6 +2,7 @@ import { CONFIG_DIR_NAME, type ExtensionAPI } from "@earendil-works/pi-coding-ag
|
|
|
2
2
|
import { registerCommands } from "./pi/commands.ts";
|
|
3
3
|
import { registerLifecycle } from "./pi/events.ts";
|
|
4
4
|
import { registerOrchestrateTool } from "./pi/tools.ts";
|
|
5
|
+
import { registerRouteTool } from "./pi/route.ts";
|
|
5
6
|
import { onTransition } from "./state/task-state.ts";
|
|
6
7
|
import { pingTransition } from "./pi/notify.ts";
|
|
7
8
|
import { isSubagentProcess } from "./pi/quiet.ts";
|
|
@@ -26,6 +27,8 @@ export default function (pi: ExtensionAPI): void {
|
|
|
26
27
|
onTransition((task) => pingTransition(task));
|
|
27
28
|
registerCommands(pi, CONFIG_DIR_NAME);
|
|
28
29
|
registerOrchestrateTool(pi, CONFIG_DIR_NAME);
|
|
30
|
+
// A new request one agent can do alone: the oracle confirms, and the lobby hands it to the quick-fix agent.
|
|
31
|
+
registerRouteTool(pi, CONFIG_DIR_NAME);
|
|
29
32
|
// The questionnaire the oracle (and pi without a task) asks the user with; no other extension is needed for it.
|
|
30
33
|
registerAskTool(pi);
|
|
31
34
|
// The web tools, for the researcher and for pi without a task (the oracle leaves them to the researcher).
|
package/src/lobby/feed.ts
CHANGED
|
@@ -284,8 +284,13 @@ export function chatText(role: "user" | "assistant", text: string): Array<{ role
|
|
|
284
284
|
const kickoff = /^A bot-lobby task is active: (\S+)\nTitle: (.*)/.exec(body);
|
|
285
285
|
if (kickoff) {
|
|
286
286
|
const request = /\nRequest: ([\s\S]*?)(?:\nState: |$)/.exec(body)?.[1]?.trim();
|
|
287
|
-
|
|
287
|
+
// A request the oracle routed to the team is already in the conversation, under the routing note.
|
|
288
|
+
const shown = /\nRouted: /.test(body);
|
|
289
|
+
return [{ role: "note", text: `task started · ${kickoff[2]}` }, ...(request && !shown ? [{ role: "you" as const, text: request }] : [])];
|
|
288
290
|
}
|
|
291
|
+
// A new request the classifier read as a quick fix, before the oracle confirms where it goes.
|
|
292
|
+
const routing = /^bot-lobby: a new request, not a task yet\.\nRequest: ([\s\S]*?)\nRead: /.exec(body);
|
|
293
|
+
if (routing) return [{ role: "you", text: routing[1]!.trim() }, { role: "note", text: "reads as a quick fix · the oracle confirms where it goes" }];
|
|
289
294
|
const comment = /^The user left (?:a comment|\d+ comments) on (the approved plan|the proposal) of (\S+) from the lobby:/.exec(body);
|
|
290
295
|
if (comment) return [{ role: "note", text: `your comment on ${comment[1]} went to the oracle` }];
|
|
291
296
|
return [{ role: "you", text: body }];
|
package/src/lobby/quickfix.ts
CHANGED
|
@@ -54,6 +54,19 @@ export interface QuickFixJob {
|
|
|
54
54
|
routedFrom?: string;
|
|
55
55
|
/** Files it changed with `edit`/`write`, as shown (relative to the project); the change ledger lets a running task's QA gate tell them from its own. */
|
|
56
56
|
files?: string[];
|
|
57
|
+
/** Runs on this profile instead of the quick-fix settings: a quick feature the oracle routed here gets its builder's model, thinking and time. */
|
|
58
|
+
profile?: QuickFixProfile;
|
|
59
|
+
/** The oracle sent it here from a request typed to the lobby. */
|
|
60
|
+
routed?: boolean;
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
/** How a job enters the queue besides its prompt. */
|
|
64
|
+
export interface SubmitOptions {
|
|
65
|
+
/** Skip the classifier's "looks like a task" hold (the oracle already routed it here). */
|
|
66
|
+
force?: boolean;
|
|
67
|
+
profile?: QuickFixProfile;
|
|
68
|
+
/** Shown with the job: where it came from. */
|
|
69
|
+
note?: string;
|
|
57
70
|
}
|
|
58
71
|
|
|
59
72
|
export interface QuickFixProfile {
|
|
@@ -118,10 +131,21 @@ export class QuickFixQueue {
|
|
|
118
131
|
}
|
|
119
132
|
|
|
120
133
|
/** Queue a quick fix; it starts at once when nothing else is running. */
|
|
121
|
-
submit(prompt: string, now = Date.now()): QuickFixJob {
|
|
134
|
+
submit(prompt: string, now = Date.now(), options: SubmitOptions = {}): QuickFixJob {
|
|
122
135
|
const text = prompt.trim();
|
|
123
136
|
if (!text) throw new Error("a quick fix needs a prompt");
|
|
124
|
-
const job: QuickFixJob = {
|
|
137
|
+
const job: QuickFixJob = {
|
|
138
|
+
id: `QF-${++this.counter}`,
|
|
139
|
+
prompt: text,
|
|
140
|
+
status: "queued",
|
|
141
|
+
createdAt: now,
|
|
142
|
+
steps: [],
|
|
143
|
+
tools: 0,
|
|
144
|
+
turns: 0,
|
|
145
|
+
...(options.force ? { force: true, routed: true } : {}),
|
|
146
|
+
...(options.profile ? { profile: options.profile } : {}),
|
|
147
|
+
...(options.note ? { note: options.note } : {}),
|
|
148
|
+
};
|
|
125
149
|
this.jobs = [...this.jobs, job].filter((entry, index, all) => isActive(entry) || index >= all.length - MAX_JOBS);
|
|
126
150
|
this.changed();
|
|
127
151
|
this.pump();
|
|
@@ -203,7 +227,7 @@ export class QuickFixQueue {
|
|
|
203
227
|
}
|
|
204
228
|
|
|
205
229
|
private async run(job: QuickFixJob): Promise<void> {
|
|
206
|
-
const profile = this.deps.profile();
|
|
230
|
+
const profile = job.profile ?? this.deps.profile();
|
|
207
231
|
const controller = new AbortController();
|
|
208
232
|
this.controllers.set(job.id, controller);
|
|
209
233
|
job.status = "running";
|
|
@@ -287,6 +311,8 @@ export class QuickFixQueue {
|
|
|
287
311
|
try {
|
|
288
312
|
const sized = await quickFixSize(jev, job.prompt, signal);
|
|
289
313
|
if (!sized || sized.size !== "large" || sized.confidence < jev.config.thresholds.quickFixLargeAt) return undefined;
|
|
314
|
+
// Large but still one engineer's work (a rich page in one file): it runs.
|
|
315
|
+
if ((sized.solo ?? 0) >= jev.config.thresholds.quickFixAt) return undefined;
|
|
290
316
|
return `looks like a task (large, ${sized.confidence.toFixed(2)})`;
|
|
291
317
|
} catch {
|
|
292
318
|
return undefined;
|
package/src/lobby/runtime.ts
CHANGED
|
@@ -26,7 +26,9 @@ import { shortTitle } from "../text.ts";
|
|
|
26
26
|
import { isSubagentProcess } from "../pi/quiet.ts";
|
|
27
27
|
import { modelRef, resolveLobbyProfile, resolvePanelProfile } from "../pi/model-support.ts";
|
|
28
28
|
import { modelLookup } from "../pi/tools.ts";
|
|
29
|
-
import { startPlannedTask
|
|
29
|
+
import { startPlannedTask } from "../pi/start-task.ts";
|
|
30
|
+
import { pendingRequest, setQuickFixHandoff, startRequest } from "../pi/route.ts";
|
|
31
|
+
import type { Domain } from "../schemas/agent.ts";
|
|
30
32
|
import type { LobbyAgentKind, LobbyPanel, PanelMember } from "../schemas/configuration.ts";
|
|
31
33
|
import { chatFromEntries, lobbyFeed, narrateEvent, type AgentEventLike, type ChatEntry } from "./feed.ts";
|
|
32
34
|
import { classifier, effortFor, hintsFor } from "../classifier/instance.ts";
|
|
@@ -149,9 +151,10 @@ function toOracle(state: Runtime, text: string): string | undefined {
|
|
|
149
151
|
state.pi.sendUserMessage(text, { expandPromptTemplates: true, ...(busy ? { deliverAs: "followUp" as const } : {}) });
|
|
150
152
|
return `sent ${text.split(/\s+/)[0]} to pi (built-in commands need the lobby hidden: alt+l)`;
|
|
151
153
|
}
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
154
|
+
// While the oracle decides where a request goes, what the user types joins that conversation.
|
|
155
|
+
if (!currentZenTask() && !pendingRequest()) {
|
|
156
|
+
startRequest(state.pi, state.ctx, state.configDir, text).catch((error: Error) => failed(state, "could not start the task", error));
|
|
157
|
+
return "reading your request — a quick fix or a task, the oracle takes it from here";
|
|
155
158
|
}
|
|
156
159
|
state.pi.sendUserMessage(text, busy ? { deliverAs: "steer" } : undefined);
|
|
157
160
|
return undefined;
|
|
@@ -706,6 +709,7 @@ function shutdown(): void {
|
|
|
706
709
|
const state = runtime;
|
|
707
710
|
if (!state) return;
|
|
708
711
|
runtime = undefined;
|
|
712
|
+
setQuickFixHandoff(undefined);
|
|
709
713
|
setMouse(state, false);
|
|
710
714
|
state.unsubscribeFeed?.();
|
|
711
715
|
state.quickfix.cancelAll();
|
|
@@ -717,6 +721,24 @@ function shutdown(): void {
|
|
|
717
721
|
onMinimizeChange(undefined);
|
|
718
722
|
}
|
|
719
723
|
|
|
724
|
+
/**
|
|
725
|
+
* The oracle routed a request to the quick-fix agent: queue it (no "looks like
|
|
726
|
+
* a task" hold, the oracle already decided), on its builder's settings when it
|
|
727
|
+
* is a quick feature, and show it on the Quick fix tab.
|
|
728
|
+
*/
|
|
729
|
+
function handToQuickFix(state: Runtime, request: string, builder: Domain | undefined, reason: string): string {
|
|
730
|
+
const quick = lobbyProfile(state, "quickfix");
|
|
731
|
+
const profile = builder ? { ...seatProfile(state, builder), instructions: quick.instructions } : undefined;
|
|
732
|
+
const job = state.quickfix.submit(request, Date.now(), {
|
|
733
|
+
force: true,
|
|
734
|
+
...(profile ? { profile } : {}),
|
|
735
|
+
note: `The oracle sent your request here: ${reason.replace(/[.\s]+$/, "")}.${builder ? ` A quick feature: it runs on ${builder === "designer" ? "DESIGN" : "DEV"}'s model, thinking and time limit.` : ""}`,
|
|
736
|
+
});
|
|
737
|
+
showLobby("quickfix");
|
|
738
|
+
state.view?.showQuickFix(job.id);
|
|
739
|
+
return job.id;
|
|
740
|
+
}
|
|
741
|
+
|
|
720
742
|
/** Create the session's lobby runtime (interactive master sessions only). */
|
|
721
743
|
export function initLobby(pi: ExtensionAPI, ctx: ExtensionContext, configDir: string): void {
|
|
722
744
|
shutdown();
|
|
@@ -754,6 +776,7 @@ export function initLobby(pi: ExtensionAPI, ctx: ExtensionContext, configDir: st
|
|
|
754
776
|
effort: effortFor((model, thinking) => checkThinking(modelLookup(ctx)(model), thinking).level),
|
|
755
777
|
});
|
|
756
778
|
runtime = state;
|
|
779
|
+
setQuickFixHandoff((request, builder, reason) => (runtime === state ? handToQuickFix(state, request, builder, reason) : undefined));
|
|
757
780
|
lobbyFeed.clear();
|
|
758
781
|
// Only the newest messages are kept; the feed learns whether earlier ones exist, and loads them when scrolled to.
|
|
759
782
|
lobbyFeed.seedChat(chatFromEntries(ctx.sessionManager.getBranch(), Number.POSITIVE_INFINITY));
|