@a-t-h-i/bot-lobby 0.6.14 → 0.6.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +70 -79
- package/package.json +4 -2
- package/prompts/master.md +3 -2
- package/prompts/qa.md +33 -11
- package/prompts/reviewer.md +5 -2
- package/prompts/worker.md +5 -2
- package/src/ask/relay.ts +1 -2
- package/src/ask/tool.ts +4 -6
- package/src/ask/types.ts +16 -0
- package/src/ask/web.ts +63 -0
- package/src/index.ts +7 -1
- package/src/lobby/ask.ts +4 -5
- package/src/lobby/blocks.ts +57 -0
- package/src/lobby/feed.ts +1 -1
- package/src/lobby/host.ts +178 -0
- package/src/lobby/keys.ts +10 -41
- package/src/lobby/prompt-hub.ts +30 -77
- package/src/lobby/prompts.ts +8 -12
- package/src/lobby/runtime.ts +49 -310
- package/src/lobby/service.ts +54 -134
- package/src/lobby/task-rows.ts +28 -0
- package/src/pi/commands.ts +14 -20
- package/src/pi/model-settings.ts +112 -0
- package/src/pi/start-task.ts +1 -1
- package/src/pi/tools.ts +6 -7
- package/src/pi/ui.ts +0 -9
- package/src/schemas/configuration.ts +9 -33
- package/src/state/attachments.ts +82 -0
- package/src/webui/api/index.ts +17 -15
- package/src/webui/api/lobby.ts +4 -3
- package/src/webui/api/planner.ts +6 -4
- package/src/webui/api/quickfix.ts +4 -3
- package/src/webui/api/sessions.ts +9 -7
- package/src/webui/api/settings.ts +20 -8
- package/src/webui/api/status.ts +4 -15
- package/src/webui/api/tasks.ts +11 -15
- package/src/webui/command.ts +35 -19
- package/src/webui/dev/fake-service.ts +1 -2
- package/src/webui/dev/fixtures.ts +2 -5
- package/src/webui/events.ts +1 -1
- package/src/webui/protocol.ts +22 -11
- package/src/webui/server.ts +42 -1
- package/src/webui/uploads.ts +68 -0
- package/webui/dist/assets/index-B1PnCyl2.css +2 -0
- package/webui/dist/assets/index-DI9Yq9IC.js +90 -0
- package/webui/dist/assets/inter-latin-wght-normal-Dx4kXJAl.woff2 +0 -0
- package/webui/dist/build.json +138 -110
- package/webui/dist/index.html +4 -4
- package/src/ask/dialog.ts +0 -193
- package/src/ask/image.ts +0 -202
- package/src/ask/png.ts +0 -179
- package/src/ask/state.ts +0 -175
- package/src/ask/view.ts +0 -163
- package/src/lobby/layout.ts +0 -496
- package/src/lobby/markdown.ts +0 -121
- package/src/lobby/mini.ts +0 -167
- package/src/lobby/tabs/excalidraw.ts +0 -95
- package/src/lobby/tabs/git.ts +0 -162
- package/src/lobby/tabs/home.ts +0 -544
- package/src/lobby/tabs/issues.ts +0 -73
- package/src/lobby/tabs/knowledge.ts +0 -135
- package/src/lobby/tabs/metrics.ts +0 -300
- package/src/lobby/tabs/plan.ts +0 -262
- package/src/lobby/tabs/quickfix.ts +0 -136
- package/src/lobby/tabs/tasks.ts +0 -412
- package/src/lobby/theme.ts +0 -30
- package/src/lobby/view.ts +0 -3018
- package/src/pi/settings-ui.ts +0 -708
- package/src/width.ts +0 -102
- package/webui/dist/assets/index-CMn1bsps.css +0 -2
- package/webui/dist/assets/index-Dj1XZA3O.js +0 -80
package/README.md
CHANGED
|
@@ -13,7 +13,9 @@ For a task, your Pi session becomes the **Master**
|
|
|
13
13
|
(the "oracle"): it scouts the codebase, proposes a plan, and delegates the
|
|
14
14
|
work to three domain agents — **Designer+Frontend**, **Backend** and **QA** —
|
|
15
15
|
each running in its own isolated `pi` process. QA's reviewer is the quality
|
|
16
|
-
gate before anything is marked done.
|
|
16
|
+
gate before anything is marked done. QA reviews adversarially (it tries to
|
|
17
|
+
break the change) and writes as few tests as it can: only for a breaking
|
|
18
|
+
change, for behavior that could turn out unpredictable, or when asked.
|
|
17
19
|
|
|
18
20
|
The rule: **you decide with the LLMs, and the engine enforces.** Agents
|
|
19
21
|
propose; you shape and approve the plan with the oracle; the extension
|
|
@@ -43,31 +45,31 @@ tool names.
|
|
|
43
45
|
|
|
44
46
|
1. `/bot-lobby add a login page` — starts a task; the lobby opens.
|
|
45
47
|
2. Answer the Master's questions, then approve its proposal.
|
|
46
|
-
3. Follow the agents in the lobby
|
|
48
|
+
3. Follow the agents in the lobby, a web page that opens in your browser when Pi starts: what they do, and what they think.
|
|
47
49
|
|
|
48
|
-
|
|
49
|
-
extra instructions.
|
|
50
|
+
The lobby's Settings page sets each agent's model, effort (a slider that skips
|
|
51
|
+
what the model cannot do), time limit and extra instructions.
|
|
50
52
|
|
|
51
53
|
## Commands
|
|
52
54
|
|
|
53
55
|
| Command | Does |
|
|
54
56
|
| --- | --- |
|
|
55
|
-
| `/bot-lobby` | Open the lobby (
|
|
57
|
+
| `/bot-lobby` | Open the lobby in your browser (it already runs: the page starts with Pi) |
|
|
56
58
|
| `/bot-lobby <request>` | Start a request: a [quick fix](#quick-fix-or-the-team) when one agent can do it alone, else a task (`--task` to always make it a task, also when it begins with a command word; `--auto` to run unattended, `--budget 90m` to give it a time budget, `--fast` / `--full` to pick its [track](#fast-track-or-full-workflow), `--branch` / `--worktree` / `--no-branch` to give it its own [git branch or worktree](#a-branch-or-worktree-per-task) or none) |
|
|
57
59
|
| `/bot-lobby budget [90m\|off]` | Show or set this session's task time budget |
|
|
58
60
|
| `/bot-lobby status \| tasks \| runs [id]` | Current task, all tasks, recent agent runs |
|
|
59
61
|
| `/bot-lobby approve \| amend <text> \| decline` | Answer the proposal |
|
|
60
62
|
| `/bot-lobby accept [id]` | Accept a task's work as it is, without a QA pass; the oracle then completes it |
|
|
61
63
|
| `/bot-lobby pause \| resume \| cancel [id]` | Control a task |
|
|
62
|
-
| `/bot-lobby auto [on\|off]` | Auto mode: the oracle finishes the task without asking (`alt+g`) |
|
|
64
|
+
| `/bot-lobby auto [on\|off]` | Auto mode: the oracle finishes the task without asking (`alt+g` in Pi) |
|
|
63
65
|
| `/bot-lobby claim <id>` | Take over a task another session owned |
|
|
64
66
|
| `/bot-lobby start-plan PLAN-… [auto]` | Start a plan saved from the Plan tab |
|
|
65
67
|
| `/bot-lobby settings \| config` | Edit settings / show the effective config |
|
|
66
68
|
| `/bot-lobby knowledge` | Knowledge file sizes |
|
|
67
69
|
| `/bot-lobby minimize \| restore` | Hide bot-lobby in this session (`ctrl+shift+m`) |
|
|
68
|
-
| `/bot-lobby web` |
|
|
70
|
+
| `/bot-lobby web` | Open the lobby page again (it starts with every interactive Pi session, on this machine only) and print its link |
|
|
69
71
|
| `/bot-lobby web link` | Print the browser UI's link |
|
|
70
|
-
| `/bot-lobby web stop` | Stop the
|
|
72
|
+
| `/bot-lobby web stop` | Stop the page until the next session |
|
|
71
73
|
| `/bot-lobby web reset` | Reset the browser UI's link (its old cookies stop working) |
|
|
72
74
|
|
|
73
75
|
## Quick fix or the team
|
|
@@ -250,19 +252,29 @@ everything. `workflow.freshContext: false` in the config turns this off.
|
|
|
250
252
|
|
|
251
253
|
## The lobby
|
|
252
254
|
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
line
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
255
|
+
bot-lobby is a web-only plugin. When Pi starts an interactive session it also
|
|
256
|
+
starts the lobby, a page served on `127.0.0.1` (this machine only, behind a
|
|
257
|
+
secret link), opens it in your browser and shows its address in Pi's status
|
|
258
|
+
line. Nothing is drawn in the terminal but Pi itself.
|
|
259
|
+
|
|
260
|
+
The page is a calm, glassy window in a light or a dark theme (the sun/moon
|
|
261
|
+
button in its top row). The tabs are numbered pills joined by dotted lines; the
|
|
262
|
+
lit pill glides to the tab you pick like a drop of water. Under the tabs, each
|
|
263
|
+
tab is a pair of cards, a list on the left and the chosen item on the right.
|
|
264
|
+
At the bottom floats **one text box for everything**: it grows as you type (or
|
|
265
|
+
opens up to a tall editor), takes Markdown, and takes images, PDFs and other
|
|
266
|
+
files (pick, paste or drop them, up to 20 MB each, eight per message). Who it
|
|
267
|
+
talks to follows the tab: the oracle everywhere, the planning panel on Plan, a
|
|
268
|
+
quick fix on Quick fix, a comment on the open task (or a message to its
|
|
269
|
+
oracle) on Tasks, the picked session on Sessions. The oracle's questions pop
|
|
270
|
+
up in the middle of the window over a blurred backdrop; there is only ever one
|
|
271
|
+
pop-up, and one toast, on screen at a time. Activity and Thinking can be
|
|
272
|
+
minimized to their title bar. `alt+h` lists every key.
|
|
261
273
|
|
|
262
274
|
| Tab | What it is |
|
|
263
275
|
| --- | --- |
|
|
264
276
|
| **1 Lobby** | Your conversation with the oracle, an activity log of every agent's steps, and each agent's latest thought |
|
|
265
|
-
| **2 Tasks** | Every task and saved plan as a checklist.
|
|
277
|
+
| **2 Tasks** | Every task and saved plan as a checklist. Start a plan here or in a new session; comment on a plan (images welcome); archive or delete |
|
|
266
278
|
| **3 Plan** | Plan a task with a panel of agents before building it (below) |
|
|
267
279
|
| **4 Quick fix** | One agent makes a change right away, beside any running task; requests the oracle [routes here](#quick-fix-or-the-team) show up too |
|
|
268
280
|
| **5 Metrics** | Run time, success rate, tokens and cost per model and agent |
|
|
@@ -275,8 +287,9 @@ footer.
|
|
|
275
287
|

|
|
276
288
|
|
|
277
289
|
Your conversation with the oracle on the left, the activity log of every
|
|
278
|
-
agent's steps on the right, and the agents' latest thoughts below.
|
|
279
|
-
|
|
290
|
+
agent's steps on the right, and the agents' latest thoughts below. Activity and
|
|
291
|
+
Thinking fold down to their title bar; the box at the bottom steers the
|
|
292
|
+
running turn.
|
|
280
293
|
|
|
281
294
|
### Tasks
|
|
282
295
|
|
|
@@ -309,17 +322,17 @@ which cheaper models hold up.
|
|
|
309
322
|
### Git
|
|
310
323
|
|
|
311
324
|
The repository's open pull requests through the GitHub CLI (`gh` owns sign-in;
|
|
312
|
-
bot-lobby holds no token): the list with checks
|
|
325
|
+
bot-lobby holds no token): the list with checks and size, and the
|
|
313
326
|
selected one with its facts, files, description, reviews and comments.
|
|
314
327
|
|
|
315
|
-
-
|
|
328
|
+
- **Review it with an agent**: a read-only agent on QA's model, thinking
|
|
316
329
|
and time limit (and its custom instructions) gets the description, changed
|
|
317
330
|
files and diff, may read the repository for context, and writes a review:
|
|
318
331
|
verdict, summary, findings by severity (`file:line`), tests, questions. It
|
|
319
332
|
never edits, and never follows instructions written inside the pull request.
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
-
|
|
333
|
+
A focus can be typed first (*is the migration reversible?*), and a review
|
|
334
|
+
can be stopped.
|
|
335
|
+
- **Jev's quick read** ([the classifier](#the-classifier-jev)): size, and
|
|
323
336
|
how likely the change is risky, security-relevant, breaking or untested, in a
|
|
324
337
|
moment, with whether a full review is worth its tokens.
|
|
325
338
|
- Reviews are kept per pull request (`.pi/bot-lobby/reviews/`), marked stale
|
|
@@ -418,37 +431,18 @@ deletion), and one call draws at most 100 shapes and removes at most 50.
|
|
|
418
431
|
through it, and the message names it. A network that blocks websockets but not
|
|
419
432
|
HTTPS still works: the seat falls back to long-polling, as a browser does.
|
|
420
433
|
|
|
421
|
-
Common keys: `
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
426
|
-
**Several sessions from one window.**
|
|
427
|
-
Pi session. The
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
**Paging.** A pane with more lines than rows shows a pager on its bottom
|
|
435
|
-
edge, `▲ prev · page 2/5 · next ▼`: click *prev* or *next* to move a page (its
|
|
436
|
-
rows less one, so a line carries over), and read where you are from the page
|
|
437
|
-
count. The top is page 1 and the bottom the last. A button dims when the pane
|
|
438
|
-
is already at that end, and the words shorten (`▲ prev · 2/5 · next ▼`, then
|
|
439
|
-
`▲ 2/5 ▼`) as the pane narrows. The wheel, the arrows and PageUp/PageDown
|
|
440
|
-
still work. It applies to every scrolling pane: the conversation, activity and
|
|
441
|
-
thinking, the plan draft, the Tasks and Quick fix lists and details, and the
|
|
442
|
-
metrics table.
|
|
443
|
-
|
|
444
|
-
**Status line when hidden.** With the lobby hidden (`alt+l`), one line under
|
|
445
|
-
Pi's editor shows where things stand: a bar of the task's plan steps (or its
|
|
446
|
-
stage before there is a plan) with who is working, the planning round and the
|
|
447
|
-
questions waiting for you, the quick fix in hand, or `idle`. It costs nothing
|
|
448
|
-
while nothing changes. Turn it off with `lobby.miniLine: false` (or in
|
|
449
|
-
`/bot-lobby settings` → Lobby).
|
|
450
|
-
|
|
451
|
-

|
|
434
|
+
Common keys: `alt+1`…`alt+9` jump to a tab, `alt+[` and `alt+]` cycle them,
|
|
435
|
+
`ctrl+s` saves the plan on Plan, `alt+o` browses sessions, `alt+s` opens
|
|
436
|
+
settings, `alt+h` shows them all. Rebind any key under `lobby.keys` in the
|
|
437
|
+
config.
|
|
438
|
+
|
|
439
|
+
**Several sessions from one window.** The Sessions page starts a task in a
|
|
440
|
+
background Pi session (the box's New session target). The page can show any
|
|
441
|
+
session, and your messages steer it; a badge on its row means it has a
|
|
442
|
+
question for you.
|
|
443
|
+
|
|
444
|
+
**A question put away.** Esc (or the cross) puts a question pop-up away
|
|
445
|
+
without losing a word; the *waiting* button in the top row brings it back.
|
|
452
446
|
|
|
453
447
|
The conversation keeps its newest 100 messages in memory; scroll to the top
|
|
454
448
|
to load the rest.
|
|
@@ -502,28 +496,23 @@ has them as well.
|
|
|
502
496
|
|
|
503
497
|
### The questionnaire
|
|
504
498
|
|
|
505
|
-
`ask_user_question` puts up to four questions to you in one
|
|
506
|
-
two to four options (the recommended one first).
|
|
507
|
-
descriptions and **previews** are Markdown: an option's
|
|
508
|
-
sketch, a component mockup, a code snippet, a config) shows
|
|
509
|
-
while that option is focused, under it in a narrow
|
|
510
|
-
can be compared by looking at them.
|
|
511
|
-
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
run Pi in RPC mode get the same questions through Pi's own dialogs.
|
|
499
|
+
`ask_user_question` puts up to four questions to you in one pop-up in the
|
|
500
|
+
lobby page, each with two to four options (the recommended one first).
|
|
501
|
+
Questions, option descriptions and **previews** are Markdown: an option's
|
|
502
|
+
preview (a layout sketch, a component mockup, a code snippet, a config) shows
|
|
503
|
+
beside the list while that option is focused, under it in a narrow window, so
|
|
504
|
+
design choices can be compared by looking at them.
|
|
505
|
+
|
|
506
|
+
Click an option, or press `1`–`4`; several can be picked in a multi-select;
|
|
507
|
+
the last field takes an answer in your own words. *Later* (Esc) puts the
|
|
508
|
+
pop-up away without losing anything, *Cancel* asks whether to leave, so a
|
|
509
|
+
stray click does nothing. Questions you leave are never answered for you: the
|
|
510
|
+
oracle waits and asks again when you next write, and the designer asks again
|
|
511
|
+
before it may decide. A question with nobody to answer it (a one-shot
|
|
512
|
+
`pi -p` run) is put away at once.
|
|
520
513
|
|
|
521
514
|
**Images.** An option can also carry an `image`: a PNG, JPEG, GIF or WebP
|
|
522
515
|
file (a screenshot, a rendered mockup), shown above its preview text.
|
|
523
|
-
Terminals with the Kitty graphics protocol (Kitty, Ghostty, WezTerm) show the
|
|
524
|
-
image itself; other terminals draw PNGs as coloured half-blocks, and name the
|
|
525
|
-
file for other formats. `BOT_LOBBY_IMAGES=blocks` always uses blocks, `off`
|
|
526
|
-
never draws images.
|
|
527
516
|
|
|
528
517
|
**The designer asks you directly.** During a task (not in auto mode) the
|
|
529
518
|
designer worker can put its visual choices to you: its questions reach you
|
|
@@ -601,8 +590,8 @@ switch on its own.
|
|
|
601
590
|
## Configuration
|
|
602
591
|
|
|
603
592
|
Settings live in `~/.pi/bot-lobby/config.json` (`BOT_LOBBY_CONFIG_DIR`
|
|
604
|
-
overrides). Edit them
|
|
605
|
-
the result.
|
|
593
|
+
overrides). Edit them on the lobby's Settings page (the cog in its top row;
|
|
594
|
+
it saves as you go); `/bot-lobby config` shows the result.
|
|
606
595
|
|
|
607
596
|
```json
|
|
608
597
|
{
|
|
@@ -612,7 +601,7 @@ the result.
|
|
|
612
601
|
},
|
|
613
602
|
"scout": { "model": "anthropic/claude-haiku-4-5-20251001", "timeoutMs": 480000 },
|
|
614
603
|
"planner": { "thinking": "high", "timeoutMs": 300000 },
|
|
615
|
-
"lobby": { "planningPanel": ["backend", "designer", "qa", "researcher"], "maxPlanningRounds": 5, "splitPlanAbove": 8 },
|
|
604
|
+
"lobby": { "planningPanel": ["backend", "designer", "qa", "researcher"], "maxPlanningRounds": 5, "splitPlanAbove": 8, "web": { "port": 7347, "openBrowser": true } },
|
|
616
605
|
"workflow": { "maxReviewIterations": 2, "maxParallelWorkers": 3, "stallTimeoutMs": 300000, "wrapUpAt": 0.75, "taskBudgetMinutes": 0, "fastTrack": true, "briefCheck": true, "routeQuickFixes": true, "gitIsolation": "off" },
|
|
617
606
|
"classifier": { "enabled": false, "provider": "auto", "effort": { "cheapModel": "inherit" } }
|
|
618
607
|
}
|
|
@@ -622,7 +611,8 @@ the result.
|
|
|
622
611
|
`quickFix`, `planner`, `lobby`, `workflow`, `knowledge`, `classifier`.
|
|
623
612
|
- An agent without a model runs on the session's model. Thinking is one of
|
|
624
613
|
`off, minimal, low, medium, high, xhigh, max`, limited to what the model
|
|
625
|
-
supports
|
|
614
|
+
supports: on the Settings page it is a slider whose unsupported stops are
|
|
615
|
+
struck through and cannot be chosen. Scouts always think at `low`.
|
|
626
616
|
- `instructions` adds your own text to an agent's built-in prompt.
|
|
627
617
|
- `fallbackModel` and `fallbackThinking` on any agent (and the master): see
|
|
628
618
|
[Fallback models](#fallback-models).
|
|
@@ -709,7 +699,8 @@ BOT_LOBBY_LIVE_EXCALIDRAW=1 node --test test/excalidraw-live.test.ts
|
|
|
709
699
|
```
|
|
710
700
|
|
|
711
701
|
Source layout: `src/workflow` (engine), `src/master` (delegation),
|
|
712
|
-
`src/execution` (subagent processes), `src/lobby` (the
|
|
702
|
+
`src/execution` (subagent processes), `src/lobby` (the lobby's state and service), `src/webui` and `webui/` (the
|
|
703
|
+
server and the page; rebuild with `npm run web:build`),
|
|
713
704
|
`src/classifier` (Jev), `src/state` (persistence), `src/ask` (the
|
|
714
705
|
questionnaire), `src/web` (the web tools), `src/excalidraw` (shared
|
|
715
706
|
Excalidraw sessions: the room protocol, the sessions, the agents' tools),
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@a-t-h-i/bot-lobby",
|
|
3
|
-
"version": "0.6.
|
|
3
|
+
"version": "0.6.15",
|
|
4
4
|
"description": "Structured multi-agent software engineering orchestrator for Pi",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "Apache-2.0",
|
|
@@ -52,10 +52,11 @@
|
|
|
52
52
|
"typebox": "*"
|
|
53
53
|
},
|
|
54
54
|
"devDependencies": {
|
|
55
|
+
"@axe-core/playwright": "4.13.0",
|
|
55
56
|
"@earendil-works/pi-ai": "0.87.0",
|
|
56
57
|
"@earendil-works/pi-coding-agent": "0.87.0",
|
|
57
58
|
"@earendil-works/pi-tui": "0.87.0",
|
|
58
|
-
"@
|
|
59
|
+
"@fontsource-variable/inter": "5.3.0",
|
|
59
60
|
"@playwright/test": "1.62.0",
|
|
60
61
|
"@shadcn/react": "0.3.1",
|
|
61
62
|
"@tailwindcss/vite": "4.3.3",
|
|
@@ -66,6 +67,7 @@
|
|
|
66
67
|
"class-variance-authority": "0.7.1",
|
|
67
68
|
"clsx": "2.1.1",
|
|
68
69
|
"lucide-react": "1.49.0",
|
|
70
|
+
"motion": "14.0.0",
|
|
69
71
|
"radix-ui": "1.6.7",
|
|
70
72
|
"react": "19.3.0",
|
|
71
73
|
"react-dom": "19.3.0",
|
package/prompts/master.md
CHANGED
|
@@ -27,8 +27,9 @@ process it gets. Fewer steps win whenever the result is the same.
|
|
|
27
27
|
no plan document: delegate straight away with `orchestrate action=implement`,
|
|
28
28
|
opening each task with `Step N:` (the engine keeps the plan and the
|
|
29
29
|
checklist). Only the roster takes part: DESIGN for frontend work, DEV for
|
|
30
|
-
backend work, QA when the change needs tests
|
|
31
|
-
|
|
30
|
+
backend work, QA when the change needs tests or an adversarial review
|
|
31
|
+
(its worker writing the few tests that are needed and running them as the
|
|
32
|
+
last step, or the QA gate), and the researcher when a
|
|
32
33
|
decision needs outside facts (summon it first). Several domains: one
|
|
33
34
|
`implement` with `assignments`, each task stating the contract between
|
|
34
35
|
them. When the work is in, check `git diff --stat` and the report, then
|
package/prompts/qa.md
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
You are responsible for quality assurance and quality gates, not merely a test
|
|
4
4
|
runner. Evaluate requirements, acceptance criteria, correctness, regression
|
|
5
5
|
risk, edge cases, security, accessibility, UX, reliability, performance where
|
|
6
|
-
relevant, and
|
|
6
|
+
relevant, and the evidence the change works.
|
|
7
7
|
|
|
8
8
|
## Independence
|
|
9
9
|
|
|
@@ -12,10 +12,11 @@ reproduce important claims where possible. Passing automated tests does not
|
|
|
12
12
|
automatically make a feature acceptable — tests are evidence, not the whole
|
|
13
13
|
quality judgment.
|
|
14
14
|
|
|
15
|
-
##
|
|
15
|
+
## Review adversarially first
|
|
16
16
|
|
|
17
|
-
|
|
18
|
-
|
|
17
|
+
Your main tool is an adversarial review, not a test suite. Read the actual
|
|
18
|
+
diff and the code around it as someone trying to break it, and look for how it
|
|
19
|
+
fails:
|
|
19
20
|
|
|
20
21
|
- invalid, malformed, boundary and empty input (zero, one, many, max, unicode)
|
|
21
22
|
- error, timeout and retry paths; dependencies that are down or slow
|
|
@@ -23,14 +24,34 @@ those failure modes:
|
|
|
23
24
|
- authorization: the wrong user, no user, an expired session
|
|
24
25
|
- concurrency and ordering: double submits, races, out-of-order responses
|
|
25
26
|
- regressions in the callers and consumers the change touches
|
|
27
|
+
- security, accessibility and UX problems the author is unlikely to have tried
|
|
26
28
|
|
|
27
|
-
|
|
29
|
+
Probe suspicions by reading, tracing and running the code (a one-off command or
|
|
30
|
+
script is fine), and report what you found as findings with evidence. A finding
|
|
31
|
+
you confirmed by running something is stronger than a test you added.
|
|
28
32
|
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
33
|
+
## Write as few tests as possible
|
|
34
|
+
|
|
35
|
+
Tests cost tokens and add upkeep, so the default is to write none. Run the
|
|
36
|
+
project's existing tests for what the change touches and judge them; do not add
|
|
37
|
+
to them to look thorough.
|
|
38
|
+
|
|
39
|
+
Write a new test only when one of these holds, and say which in your report:
|
|
40
|
+
|
|
41
|
+
- the change breaks existing behavior on purpose (a breaking change to an API,
|
|
42
|
+
a format, a contract or a default), so the old tests must change or a new one
|
|
43
|
+
must pin the new behavior
|
|
44
|
+
- the change can introduce unpredictable behavior that review cannot settle:
|
|
45
|
+
concurrency, ordering, retries, time, randomness, parsing of untrusted input,
|
|
46
|
+
or state that outlives a request
|
|
47
|
+
- the task, or the user, asked for tests
|
|
48
|
+
|
|
49
|
+
When you do write one, it names the failure it guards against; if you cannot
|
|
50
|
+
say what bug it would catch, do not write it. One sharp test on observable
|
|
51
|
+
behavior beats several shallow ones. Never write duplicate tests, tests that
|
|
52
|
+
mirror the implementation line by line, snapshot spam, tests of framework or
|
|
53
|
+
library behavior, or tests for trivial getters. Do not rewrite tests that
|
|
54
|
+
still pass.
|
|
34
55
|
|
|
35
56
|
## Never pass by default
|
|
36
57
|
|
|
@@ -49,7 +70,8 @@ A PASS is a claim backed by evidence, not an absence of complaints.
|
|
|
49
70
|
|
|
50
71
|
Use the project's existing test runner and conventions. Do not introduce a new
|
|
51
72
|
testing framework without approval. Prefer tests that validate observable
|
|
52
|
-
behavior; cover private helpers through public behavior.
|
|
73
|
+
behavior; cover private helpers through public behavior. This section is only
|
|
74
|
+
about how to run and, rarely, write tests; the rules above decide whether to. Always pass a bash
|
|
53
75
|
`timeout` for test runs, and never start watch mode or long-running servers.
|
|
54
76
|
|
|
55
77
|
## Domain boundary
|
package/prompts/reviewer.md
CHANGED
|
@@ -70,8 +70,11 @@ If implementation changes are required, report them to the Master.
|
|
|
70
70
|
- Verify each acceptance criterion; anything you could not verify is a
|
|
71
71
|
finding, and an important unverifiable claim means CHANGES_REQUIRED or
|
|
72
72
|
BLOCKED, never PASS.
|
|
73
|
-
-
|
|
74
|
-
|
|
73
|
+
- Review adversarially: try to break the change by reading and running it
|
|
74
|
+
rather than by asking for more tests. Flag a missing test only for a breaking
|
|
75
|
+
change or for behavior that could turn out unpredictable, and flag bloated,
|
|
76
|
+
duplicate or implementation-mirroring tests as findings: fewer tests is the
|
|
77
|
+
goal.
|
|
75
78
|
- Always pass a bash `timeout` to test and build commands; never start watch
|
|
76
79
|
mode or servers.
|
|
77
80
|
|
package/prompts/worker.md
CHANGED
|
@@ -31,8 +31,11 @@ section of your output instead and continue with the rest of the work.
|
|
|
31
31
|
## Testing
|
|
32
32
|
|
|
33
33
|
Run the targeted tests for what you changed (the files and behavior you
|
|
34
|
-
touched); the QA gate runs the full suite afterwards.
|
|
35
|
-
|
|
34
|
+
touched); the QA gate runs the full suite afterwards. Write as few tests as
|
|
35
|
+
possible: add one only for a breaking change, for a bug fix whose regression
|
|
36
|
+
would be silent, or for behavior that could turn out unpredictable (concurrency,
|
|
37
|
+
ordering, time, untrusted input), or when your brief asks for tests. Never pad
|
|
38
|
+
a change with tests for the sake of coverage.
|
|
36
39
|
|
|
37
40
|
- Always pass a bash `timeout` to tests and builds (for example 300 seconds).
|
|
38
41
|
- Never start dev servers, watch mode or other long-running processes, and
|
package/src/ask/relay.ts
CHANGED
|
@@ -9,8 +9,7 @@
|
|
|
9
9
|
import { tmpdir } from "node:os";
|
|
10
10
|
import { isAbsolute, join, resolve } from "node:path";
|
|
11
11
|
import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
12
|
-
import type
|
|
13
|
-
import { MAX_OPTIONS, MAX_QUESTIONS, type AskQuestion, type AskResult } from "./types.ts";
|
|
12
|
+
import { MAX_OPTIONS, MAX_QUESTIONS, type AskQuestion, type AskResult, type Asker } from "./types.ts";
|
|
14
13
|
|
|
15
14
|
/** Env flag the master sets for a subagent that may ask the user. */
|
|
16
15
|
export const ASK_ENV = "BOT_LOBBY_ASK";
|
package/src/ask/tool.ts
CHANGED
|
@@ -10,11 +10,9 @@ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
|
10
10
|
import { Container, Text } from "@earendil-works/pi-tui";
|
|
11
11
|
import { Type } from "typebox";
|
|
12
12
|
import { isSubagentProcess } from "../pi/quiet.ts";
|
|
13
|
-
import { askUser
|
|
14
|
-
import { promptHub } from "../lobby/prompt-hub.ts";
|
|
15
|
-
import { isImagePath } from "./image.ts";
|
|
13
|
+
import { askUser } from "./web.ts";
|
|
16
14
|
import { relayAsker, relayEnabled } from "./relay.ts";
|
|
17
|
-
import { ASK_TOOL, MAX_HEADER, MAX_LABEL, MAX_OPTIONS, MAX_QUESTIONS, MIN_OPTIONS, RESERVED, type AskQuestion, type AskResult } from "./types.ts";
|
|
15
|
+
import { ASK_TOOL, isImagePath, MAX_HEADER, MAX_LABEL, MAX_OPTIONS, MAX_QUESTIONS, MIN_OPTIONS, RESERVED, type AskQuestion, type AskResult, type Asker } from "./types.ts";
|
|
18
16
|
|
|
19
17
|
export { ASK_TOOL };
|
|
20
18
|
|
|
@@ -22,7 +20,7 @@ const OptionSchema = Type.Object({
|
|
|
22
20
|
label: Type.String({ maxLength: MAX_LABEL, description: `The option as the user sees and picks it: 1-5 words, at most ${MAX_LABEL} characters.` }),
|
|
23
21
|
description: Type.Optional(Type.String({ description: "What choosing it means: its trade-offs or consequences. Markdown." })),
|
|
24
22
|
preview: Type.Optional(Type.String({ description: "Markdown shown beside the options while this one is focused: a mockup, a code snippet, a diagram, a config. Only when seeing it helps the user compare." })),
|
|
25
|
-
image: Type.Optional(Type.String({ description: "Path to a PNG (or JPEG, GIF, WebP) shown with the preview: a screenshot or a rendered mockup of this option.
|
|
23
|
+
image: Type.Optional(Type.String({ description: "Path to a PNG (or JPEG, GIF, WebP) shown with the preview: a screenshot or a rendered mockup of this option." })),
|
|
26
24
|
});
|
|
27
25
|
|
|
28
26
|
const QuestionSchema = Type.Object({
|
|
@@ -115,7 +113,7 @@ export function registerAskTool(pi: ExtensionAPI, ask?: Asker): void {
|
|
|
115
113
|
const questions = (params as { questions: AskQuestion[] }).questions;
|
|
116
114
|
const invalid = invalidQuestions(questions);
|
|
117
115
|
if (invalid) throw new Error(`${invalid}.`);
|
|
118
|
-
const result = await
|
|
116
|
+
const result = await asker(questions, ctx, signal, "oracle");
|
|
119
117
|
return { content: [{ type: "text", text: answerSummary(questions, result) }], details: result };
|
|
120
118
|
},
|
|
121
119
|
renderCall(args, theme) {
|
package/src/ask/types.ts
CHANGED
|
@@ -6,6 +6,8 @@
|
|
|
6
6
|
* diagram) shown beside the options.
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
|
+
import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
10
|
+
|
|
9
11
|
/** The questionnaire tool's name, in this session and in the agents that may ask. */
|
|
10
12
|
export const ASK_TOOL = "ask_user_question";
|
|
11
13
|
|
|
@@ -53,3 +55,17 @@ export interface AskResult {
|
|
|
53
55
|
cancelled: boolean;
|
|
54
56
|
globalNote?: string;
|
|
55
57
|
}
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Puts up to `MAX_QUESTIONS` questions to the user and returns what they
|
|
61
|
+
* chose. `from` names who asks when it is not this session's model (an agent
|
|
62
|
+
* whose questions the oracle relays).
|
|
63
|
+
*/
|
|
64
|
+
export type Asker = (questions: readonly AskQuestion[], ctx: ExtensionContext, signal?: AbortSignal, from?: string) => Promise<AskResult>;
|
|
65
|
+
|
|
66
|
+
const IMAGE_EXTENSIONS = /\.(png|jpe?g|gif|webp)$/i;
|
|
67
|
+
|
|
68
|
+
/** Whether a path names an image the page can show. */
|
|
69
|
+
export function isImagePath(path: string): boolean {
|
|
70
|
+
return IMAGE_EXTENSIONS.test(path.trim());
|
|
71
|
+
}
|
package/src/ask/web.ts
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Putting questions to the user: every question goes to the web page, through
|
|
3
|
+
* the prompt hub, and waits there for the answer. Nothing is drawn in the
|
|
4
|
+
* terminal. A question nobody can answer (no page is served, as in a print
|
|
5
|
+
* run) is put away at once, so a model never waits on a screen that is not
|
|
6
|
+
* there.
|
|
7
|
+
*/
|
|
8
|
+
import { isAbsolute, resolve } from "node:path";
|
|
9
|
+
import { currentWebServer } from "../webui/server.ts";
|
|
10
|
+
import { promptHub } from "../lobby/prompt-hub.ts";
|
|
11
|
+
import type { AskAnswer, AskQuestion, AskResult, Asker } from "./types.ts";
|
|
12
|
+
|
|
13
|
+
let availability: (() => boolean) | undefined;
|
|
14
|
+
|
|
15
|
+
/** Whether the page is up to answer. */
|
|
16
|
+
export function canAsk(): boolean {
|
|
17
|
+
return availability ? availability() : currentWebServer() !== undefined;
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/** Tests say whether a page is there to answer. */
|
|
21
|
+
export function setCanAsk(next: (() => boolean) | undefined): void {
|
|
22
|
+
availability = next;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/** Image paths made absolute, so the page's preview route finds them whatever the cwd. */
|
|
26
|
+
function withAbsoluteImages(questions: readonly AskQuestion[], cwd: string): AskQuestion[] {
|
|
27
|
+
return questions.map((question) => ({
|
|
28
|
+
...question,
|
|
29
|
+
options: question.options.map((option) => (option.image?.trim() && !isAbsolute(option.image.trim()) ? { ...option, image: resolve(cwd, option.image.trim()) } : option)),
|
|
30
|
+
}));
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/** What the page sent back, kept to the shape of a result; anything else reads as put away. */
|
|
34
|
+
function readResult(value: unknown): AskResult | undefined {
|
|
35
|
+
const raw = value as Partial<AskResult> | null;
|
|
36
|
+
if (!raw || !Array.isArray(raw.answers)) return undefined;
|
|
37
|
+
return {
|
|
38
|
+
answers: raw.answers.filter((answer): answer is AskAnswer => typeof answer === "object" && answer !== null && typeof (answer as AskAnswer).questionIndex === "number"),
|
|
39
|
+
cancelled: raw.cancelled === true,
|
|
40
|
+
...(typeof raw.globalNote === "string" ? { globalNote: raw.globalNote } : {}),
|
|
41
|
+
};
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
/** The questionnaire in the page. */
|
|
45
|
+
export const askUser: Asker = async (questions, ctx, signal, from) => {
|
|
46
|
+
if (questions.length === 0 || !canAsk()) return { answers: [], cancelled: true };
|
|
47
|
+
const outcome = await promptHub.ask("questionnaire", from ?? "oracle", { questions: withAbsoluteImages(questions, ctx.cwd) }, signal ? { signal } : undefined);
|
|
48
|
+
return (outcome.how === "answered" ? readResult(outcome.value) : undefined) ?? { answers: [], cancelled: true };
|
|
49
|
+
};
|
|
50
|
+
|
|
51
|
+
/** A free-text answer (possibly several lines) in the page; undefined when put away. */
|
|
52
|
+
export async function askText(question: string, from = "orchestrate"): Promise<string | undefined> {
|
|
53
|
+
if (!canAsk()) return undefined;
|
|
54
|
+
const outcome = await promptHub.ask("text", from, { question });
|
|
55
|
+
return outcome.how === "answered" && typeof outcome.value === "string" ? outcome.value : undefined;
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/** One of `options` picked in the page; undefined when put away. */
|
|
59
|
+
export async function askChoice(title: string, options: string[], from = "orchestrate"): Promise<string | undefined> {
|
|
60
|
+
if (!canAsk()) return undefined;
|
|
61
|
+
const outcome = await promptHub.ask("choose", from, { title, options });
|
|
62
|
+
return outcome.how === "answered" && typeof outcome.value === "string" ? outcome.value : undefined;
|
|
63
|
+
}
|
package/src/index.ts
CHANGED
|
@@ -4,6 +4,8 @@ import { registerLifecycle } from "./pi/events.ts";
|
|
|
4
4
|
import { registerOrchestrateTool } from "./pi/tools.ts";
|
|
5
5
|
import { registerRouteTool } from "./pi/route.ts";
|
|
6
6
|
import { onTransition } from "./state/task-state.ts";
|
|
7
|
+
import { releaseAttachments } from "./state/attachments.ts";
|
|
8
|
+
import { TERMINAL_STATES } from "./schemas/task.ts";
|
|
7
9
|
import { pingTransition } from "./pi/notify.ts";
|
|
8
10
|
import { isSubagentProcess } from "./pi/quiet.ts";
|
|
9
11
|
import { registerDeskClient } from "./desk/client-extension.ts";
|
|
@@ -25,7 +27,11 @@ export default function (pi: ExtensionAPI): void {
|
|
|
25
27
|
// After the lifecycle, so the lobby opens over a task the status already knows.
|
|
26
28
|
registerLobbyEvents(pi, CONFIG_DIR_NAME);
|
|
27
29
|
registerOwner(pi, CONFIG_DIR_NAME);
|
|
28
|
-
|
|
30
|
+
// A finished task takes the files attached during it along.
|
|
31
|
+
onTransition((task) => {
|
|
32
|
+
pingTransition(task);
|
|
33
|
+
if (TERMINAL_STATES.includes(task.state)) releaseAttachments(task.id);
|
|
34
|
+
});
|
|
29
35
|
registerCommands(pi, CONFIG_DIR_NAME);
|
|
30
36
|
registerOrchestrateTool(pi, CONFIG_DIR_NAME);
|
|
31
37
|
// A new request one agent can do alone: the oracle confirms, and the lobby hands it to the quick-fix agent.
|
package/src/lobby/ask.ts
CHANGED
|
@@ -1,17 +1,16 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* The oracle puts the planning panel's questions to the user through the
|
|
3
|
-
* questionnaire (`../ask`): a
|
|
4
|
-
* recommended one first, and a
|
|
3
|
+
* questionnaire (`../ask`): a card per question with each seat's options, the
|
|
4
|
+
* recommended one first, and a field for an answer in the user's own words.
|
|
5
5
|
* The panel's questions become questionnaires of at most four, and the
|
|
6
6
|
* answers become the user's turn for the next round.
|
|
7
7
|
*/
|
|
8
8
|
import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
9
9
|
import type { PanelQuestion, SettledQuestion } from "./planner.ts";
|
|
10
|
-
import { MAX_HEADER, MAX_LABEL, MAX_OPTIONS, MAX_QUESTIONS, MIN_OPTIONS, RESERVED, type AskAnswer, type AskOption, type AskQuestion, type AskResult } from "../ask/types.ts";
|
|
11
|
-
import type { Asker } from "../ask/dialog.ts";
|
|
10
|
+
import { MAX_HEADER, MAX_LABEL, MAX_OPTIONS, MAX_QUESTIONS, MIN_OPTIONS, RESERVED, type AskAnswer, type AskOption, type AskQuestion, type AskResult, type Asker } from "../ask/types.ts";
|
|
12
11
|
|
|
13
12
|
export { MAX_QUESTIONS, MIN_OPTIONS, MAX_OPTIONS, type AskAnswer, type AskOption, type AskQuestion, type AskResult, type Asker };
|
|
14
|
-
export { askUser
|
|
13
|
+
export { askUser } from "../ask/web.ts";
|
|
15
14
|
|
|
16
15
|
function clip(text: string, max: number): string {
|
|
17
16
|
const flat = text.replace(/\s+/g, " ").trim();
|