@a-t-h-i/bot-lobby 0.6.14 → 0.6.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/README.md +72 -81
  2. package/package.json +4 -2
  3. package/prompts/master.md +11 -7
  4. package/prompts/panel.md +10 -6
  5. package/prompts/planner.md +13 -10
  6. package/prompts/qa.md +33 -11
  7. package/prompts/reviewer.md +5 -2
  8. package/prompts/worker.md +5 -2
  9. package/src/ask/relay.ts +1 -2
  10. package/src/ask/tool.ts +6 -8
  11. package/src/ask/types.ts +16 -0
  12. package/src/ask/web.ts +63 -0
  13. package/src/classifier/triage.ts +4 -3
  14. package/src/index.ts +7 -1
  15. package/src/lobby/ask.ts +4 -5
  16. package/src/lobby/blocks.ts +57 -0
  17. package/src/lobby/feed.ts +1 -1
  18. package/src/lobby/host.ts +178 -0
  19. package/src/lobby/keys.ts +10 -41
  20. package/src/lobby/planner.ts +4 -4
  21. package/src/lobby/prompt-hub.ts +30 -77
  22. package/src/lobby/prompts.ts +8 -12
  23. package/src/lobby/runtime.ts +49 -310
  24. package/src/lobby/service.ts +54 -134
  25. package/src/lobby/split.ts +1 -1
  26. package/src/lobby/task-rows.ts +28 -0
  27. package/src/pi/commands.ts +14 -20
  28. package/src/pi/model-settings.ts +112 -0
  29. package/src/pi/start-task.ts +1 -1
  30. package/src/pi/tools.ts +7 -8
  31. package/src/pi/ui.ts +0 -9
  32. package/src/schemas/configuration.ts +9 -33
  33. package/src/state/attachments.ts +82 -0
  34. package/src/webui/api/index.ts +17 -15
  35. package/src/webui/api/lobby.ts +4 -3
  36. package/src/webui/api/planner.ts +6 -4
  37. package/src/webui/api/quickfix.ts +4 -3
  38. package/src/webui/api/sessions.ts +9 -7
  39. package/src/webui/api/settings.ts +20 -8
  40. package/src/webui/api/status.ts +4 -15
  41. package/src/webui/api/tasks.ts +11 -15
  42. package/src/webui/command.ts +35 -19
  43. package/src/webui/dev/fake-service.ts +1 -2
  44. package/src/webui/dev/fixtures.ts +2 -5
  45. package/src/webui/events.ts +1 -1
  46. package/src/webui/protocol.ts +22 -11
  47. package/src/webui/server.ts +42 -1
  48. package/src/webui/uploads.ts +68 -0
  49. package/webui/dist/assets/index-h1uo5XHE.js +92 -0
  50. package/webui/dist/assets/index-kCj0-8z_.css +2 -0
  51. package/webui/dist/assets/inter-latin-wght-normal-Dx4kXJAl.woff2 +0 -0
  52. package/webui/dist/build.json +159 -115
  53. package/webui/dist/index.html +8 -5
  54. package/webui/dist/manifest.webmanifest +9 -4
  55. package/src/ask/dialog.ts +0 -193
  56. package/src/ask/image.ts +0 -202
  57. package/src/ask/png.ts +0 -179
  58. package/src/ask/state.ts +0 -175
  59. package/src/ask/view.ts +0 -163
  60. package/src/lobby/layout.ts +0 -496
  61. package/src/lobby/markdown.ts +0 -121
  62. package/src/lobby/mini.ts +0 -167
  63. package/src/lobby/tabs/excalidraw.ts +0 -95
  64. package/src/lobby/tabs/git.ts +0 -162
  65. package/src/lobby/tabs/home.ts +0 -544
  66. package/src/lobby/tabs/issues.ts +0 -73
  67. package/src/lobby/tabs/knowledge.ts +0 -135
  68. package/src/lobby/tabs/metrics.ts +0 -300
  69. package/src/lobby/tabs/plan.ts +0 -262
  70. package/src/lobby/tabs/quickfix.ts +0 -136
  71. package/src/lobby/tabs/tasks.ts +0 -412
  72. package/src/lobby/theme.ts +0 -30
  73. package/src/lobby/view.ts +0 -3018
  74. package/src/pi/settings-ui.ts +0 -708
  75. package/src/width.ts +0 -102
  76. package/webui/dist/assets/index-CMn1bsps.css +0 -2
  77. package/webui/dist/assets/index-Dj1XZA3O.js +0 -80
package/README.md CHANGED
@@ -13,7 +13,9 @@ For a task, your Pi session becomes the **Master**
13
13
  (the "oracle"): it scouts the codebase, proposes a plan, and delegates the
14
14
  work to three domain agents — **Designer+Frontend**, **Backend** and **QA** —
15
15
  each running in its own isolated `pi` process. QA's reviewer is the quality
16
- gate before anything is marked done.
16
+ gate before anything is marked done. QA reviews adversarially (it tries to
17
+ break the change) and writes as few tests as it can: only for a breaking
18
+ change, for behavior that could turn out unpredictable, or when asked.
17
19
 
18
20
  The rule: **you decide with the LLMs, and the engine enforces.** Agents
19
21
  propose; you shape and approve the plan with the oracle; the extension
@@ -43,31 +45,31 @@ tool names.
43
45
 
44
46
  1. `/bot-lobby add a login page` — starts a task; the lobby opens.
45
47
  2. Answer the Master's questions, then approve its proposal.
46
- 3. Follow the agents in the lobby (`alt+l` shows or hides it): what they do, and what they think.
48
+ 3. Follow the agents in the lobby, a web page that opens in your browser when Pi starts: what they do, and what they think.
47
49
 
48
- `/bot-lobby settings` sets each agent's model, thinking level, time limit and
49
- extra instructions.
50
+ The lobby's Settings page sets each agent's model, effort (a slider that skips
51
+ what the model cannot do), time limit and extra instructions.
50
52
 
51
53
  ## Commands
52
54
 
53
55
  | Command | Does |
54
56
  | --- | --- |
55
- | `/bot-lobby` | Open the lobby (`alt+l`) |
57
+ | `/bot-lobby` | Open the lobby in your browser (it already runs: the page starts with Pi) |
56
58
  | `/bot-lobby <request>` | Start a request: a [quick fix](#quick-fix-or-the-team) when one agent can do it alone, else a task (`--task` to always make it a task, also when it begins with a command word; `--auto` to run unattended, `--budget 90m` to give it a time budget, `--fast` / `--full` to pick its [track](#fast-track-or-full-workflow), `--branch` / `--worktree` / `--no-branch` to give it its own [git branch or worktree](#a-branch-or-worktree-per-task) or none) |
57
59
  | `/bot-lobby budget [90m\|off]` | Show or set this session's task time budget |
58
60
  | `/bot-lobby status \| tasks \| runs [id]` | Current task, all tasks, recent agent runs |
59
61
  | `/bot-lobby approve \| amend <text> \| decline` | Answer the proposal |
60
62
  | `/bot-lobby accept [id]` | Accept a task's work as it is, without a QA pass; the oracle then completes it |
61
63
  | `/bot-lobby pause \| resume \| cancel [id]` | Control a task |
62
- | `/bot-lobby auto [on\|off]` | Auto mode: the oracle finishes the task without asking (`alt+g`) |
64
+ | `/bot-lobby auto [on\|off]` | Auto mode: the oracle finishes the task without asking (`alt+g` in Pi) |
63
65
  | `/bot-lobby claim <id>` | Take over a task another session owned |
64
66
  | `/bot-lobby start-plan PLAN-… [auto]` | Start a plan saved from the Plan tab |
65
67
  | `/bot-lobby settings \| config` | Edit settings / show the effective config |
66
68
  | `/bot-lobby knowledge` | Knowledge file sizes |
67
69
  | `/bot-lobby minimize \| restore` | Hide bot-lobby in this session (`ctrl+shift+m`) |
68
- | `/bot-lobby web` | Start the browser UI on this machine (loopback only) and open its link |
70
+ | `/bot-lobby web` | Open the lobby page again (it starts with every interactive Pi session, on this machine only) and print its link |
69
71
  | `/bot-lobby web link` | Print the browser UI's link |
70
- | `/bot-lobby web stop` | Stop the browser UI |
72
+ | `/bot-lobby web stop` | Stop the page until the next session |
71
73
  | `/bot-lobby web reset` | Reset the browser UI's link (its old cookies stop working) |
72
74
 
73
75
  ## Quick fix or the team
@@ -250,19 +252,29 @@ everything. `workflow.freshContext: false` in the config turns this off.
250
252
 
251
253
  ## The lobby
252
254
 
253
- A full-screen view with a prompt at the bottom that talks to whatever tab is
254
- open. Its title names the repository (or folder) you work from and its branch,
255
- `◆ my-repo (⎇ main)`. `alt+h` lists every key. **Shift+Enter** starts a new
256
- line in every text field: the prompt (also `ctrl+j`, or `\` before Enter in a
257
- terminal that cannot tell Shift+Enter apart), the questionnaire's own-answer
258
- row, and the dialogs for free-text answers. The search bar is one line by nature. It is text only: no animations, just the
259
- conversation, the activity log and the thoughts, and a one-line status in Pi's
260
- footer.
255
+ bot-lobby is a web-only plugin. When Pi starts an interactive session it also
256
+ starts the lobby, a page served on `127.0.0.1` (this machine only, behind a
257
+ secret link), opens it in your browser and shows its address in Pi's status
258
+ line. Nothing is drawn in the terminal but Pi itself.
259
+
260
+ The page is a calm, glassy window in a light or a dark theme (the sun/moon
261
+ button in its top row). The tabs are numbered pills joined by dotted lines; the
262
+ lit pill glides to the tab you pick like a drop of water. Under the tabs, each
263
+ tab is a pair of cards, a list on the left and the chosen item on the right.
264
+ At the bottom floats **one text box for everything**: it grows as you type (or
265
+ opens up to a tall editor), takes Markdown, and takes images, PDFs and other
266
+ files (pick, paste or drop them, up to 20 MB each, eight per message). Who it
267
+ talks to follows the tab: the oracle everywhere, the planning panel on Plan, a
268
+ quick fix on Quick fix, a comment on the open task (or a message to its
269
+ oracle) on Tasks, the picked session on Sessions. The oracle's questions pop
270
+ up in the middle of the window over a blurred backdrop; there is only ever one
271
+ pop-up, and one toast, on screen at a time. Activity and Thinking can be
272
+ minimized to their title bar. `alt+h` lists every key.
261
273
 
262
274
  | Tab | What it is |
263
275
  | --- | --- |
264
276
  | **1 Lobby** | Your conversation with the oracle, an activity log of every agent's steps, and each agent's latest thought |
265
- | **2 Tasks** | Every task and saved plan as a checklist. `s` starts a plan in a new session, `h` here; `c` comments on a plan; `a` archives, `d` deletes |
277
+ | **2 Tasks** | Every task and saved plan as a checklist. Start a plan here or in a new session; comment on a plan (images welcome); archive or delete |
266
278
  | **3 Plan** | Plan a task with a panel of agents before building it (below) |
267
279
  | **4 Quick fix** | One agent makes a change right away, beside any running task; requests the oracle [routes here](#quick-fix-or-the-team) show up too |
268
280
  | **5 Metrics** | Run time, success rate, tokens and cost per model and agent |
@@ -275,8 +287,9 @@ footer.
275
287
  ![The Lobby tab](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/gallery.png)
276
288
 
277
289
  Your conversation with the oracle on the left, the activity log of every
278
- agent's steps on the right, and the agents' latest thoughts below. `alt+c`,
279
- `alt+a` and `alt+k` hide any of the three; the prompt steers the running turn.
290
+ agent's steps on the right, and the agents' latest thoughts below. Activity and
291
+ Thinking fold down to their title bar; the box at the bottom steers the
292
+ running turn.
280
293
 
281
294
  ### Tasks
282
295
 
@@ -289,7 +302,7 @@ and your comments on it.
289
302
 
290
303
  ![The Plan tab](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/lobby-plan.png)
291
304
 
292
- The panel's questions, with recommended options, on the left; the draft plan
305
+ The panel's questions, with their options, on the left; the draft plan
293
306
  on the right. See [Planning](#planning).
294
307
 
295
308
  ### Quick fix
@@ -309,17 +322,17 @@ which cheaper models hold up.
309
322
  ### Git
310
323
 
311
324
  The repository's open pull requests through the GitHub CLI (`gh` owns sign-in;
312
- bot-lobby holds no token): the list with checks (`✓ ✗ ●`) and size, and the
325
+ bot-lobby holds no token): the list with checks and size, and the
313
326
  selected one with its facts, files, description, reviews and comments.
314
327
 
315
- - `v` **reviews it with an agent**: a read-only agent on QA's model, thinking
328
+ - **Review it with an agent**: a read-only agent on QA's model, thinking
316
329
  and time limit (and its custom instructions) gets the description, changed
317
330
  files and diff, may read the repository for context, and writes a review:
318
331
  verdict, summary, findings by severity (`file:line`), tests, questions. It
319
332
  never edits, and never follows instructions written inside the pull request.
320
- `f` takes a focus first (*is the migration reversible?*, over several lines
321
- with Shift+Enter). `x` stops it.
322
- - `t` is **Jev's quick read** ([the classifier](#the-classifier-jev)): size, and
333
+ A focus can be typed first (*is the migration reversible?*), and a review
334
+ can be stopped.
335
+ - **Jev's quick read** ([the classifier](#the-classifier-jev)): size, and
323
336
  how likely the change is risky, security-relevant, breaking or untested, in a
324
337
  moment, with whether a full review is worth its tokens.
325
338
  - Reviews are kept per pull request (`.pi/bot-lobby/reviews/`), marked stale
@@ -418,37 +431,18 @@ deletion), and one call draws at most 100 shapes and removes at most 50.
418
431
  through it, and the message names it. A network that blocks websockets but not
419
432
  HTTPS still works: the seat falls back to long-polling, as a browser does.
420
433
 
421
- Common keys: `tab` switches tabs, `esc` browses (arrows, single-key
422
- commands), `ctrl+f` searches, `ctrl+s` saves the plan, `alt+o` browses
423
- sessions, `alt+n` starts a task in a new session, `alt+s` opens settings.
424
- Rebind any key under `lobby.keys` in the config.
425
-
426
- **Several sessions from one window.** `alt+n` starts a task in a background
427
- Pi session. The Lobby tab can show any session, and your prompt steers it;
428
- `● waiting` in the tab bar means one has a question for you.
429
-
430
- **Agents at work.** The bottom line of the lobby shows the subagents running
431
- right now at its right end (`◐ DESIGN editing 2m · DEV 40s`, a running quick
432
- fix too), shrinking to names and then a count when the keys leave little room.
433
-
434
- **Paging.** A pane with more lines than rows shows a pager on its bottom
435
- edge, `▲ prev · page 2/5 · next ▼`: click *prev* or *next* to move a page (its
436
- rows less one, so a line carries over), and read where you are from the page
437
- count. The top is page 1 and the bottom the last. A button dims when the pane
438
- is already at that end, and the words shorten (`▲ prev · 2/5 · next ▼`, then
439
- `▲ 2/5 ▼`) as the pane narrows. The wheel, the arrows and PageUp/PageDown
440
- still work. It applies to every scrolling pane: the conversation, activity and
441
- thinking, the plan draft, the Tasks and Quick fix lists and details, and the
442
- metrics table.
443
-
444
- **Status line when hidden.** With the lobby hidden (`alt+l`), one line under
445
- Pi's editor shows where things stand: a bar of the task's plan steps (or its
446
- stage before there is a plan) with who is working, the planning round and the
447
- questions waiting for you, the quick fix in hand, or `idle`. It costs nothing
448
- while nothing changes. Turn it off with `lobby.miniLine: false` (or in
449
- `/bot-lobby settings` → Lobby).
450
-
451
- ![The status line under the editor while the lobby is hidden: a task, planning, idle](https://raw.githubusercontent.com/a-t-h-i/bot-lobby/main/docs/lobby-status-line.png)
434
+ Common keys: `alt+1`…`alt+9` jump to a tab, `alt+[` and `alt+]` cycle them,
435
+ `ctrl+s` saves the plan on Plan, `alt+o` browses sessions, `alt+s` opens
436
+ settings, `alt+h` shows them all. Rebind any key under `lobby.keys` in the
437
+ config.
438
+
439
+ **Several sessions from one window.** The Sessions page starts a task in a
440
+ background Pi session (the box's New session target). The page can show any
441
+ session, and your messages steer it; a badge on its row means it has a
442
+ question for you.
443
+
444
+ **A question put away.** Esc (or the cross) puts a question pop-up away
445
+ without losing a word; the *waiting* button in the top row brings it back.
452
446
 
453
447
  The conversation keeps its newest 100 messages in memory; scroll to the top
454
448
  to load the rest.
@@ -502,28 +496,23 @@ has them as well.
502
496
 
503
497
  ### The questionnaire
504
498
 
505
- `ask_user_question` puts up to four questions to you in one overlay, each with
506
- two to four options (the recommended one first). Questions, option
507
- descriptions and **previews** are Markdown: an option's preview (a layout
508
- sketch, a component mockup, a code snippet, a config) shows beside the list
509
- while that option is focused, under it in a narrow terminal, so design choices
510
- can be compared by looking at them.
511
-
512
- `↑↓` move · `enter` choose · `space` pick several (multi-select) · `1`–`4`
513
- pick · `←→` between questions · the last row takes an answer in your own words
514
- (`shift+enter` for a new line, pasted lines stay lines)
515
- · `esc` asks whether to leave (a second `enter`
516
- leaves, anything else keeps you answering), so a stray press does nothing.
517
- Questions you leave are never answered for you: the oracle waits and asks again
518
- when you next write, and the designer asks again before it may decide. Editor hosts that
519
- run Pi in RPC mode get the same questions through Pi's own dialogs.
499
+ `ask_user_question` puts up to four questions to you in one pop-up in the
500
+ lobby page, each with two to four options. Agents do not recommend an answer, so you think it through; only a quite obvious one is marked `(Recommended)`.
501
+ Questions, option descriptions and **previews** are Markdown: an option's
502
+ preview (a layout sketch, a component mockup, a code snippet, a config) shows
503
+ beside the list while that option is focused, under it in a narrow window, so
504
+ design choices can be compared by looking at them.
505
+
506
+ Click an option, or press `1`–`4`; several can be picked in a multi-select;
507
+ the last field takes an answer in your own words. *Later* (Esc) puts the
508
+ pop-up away without losing anything, *Cancel* asks whether to leave, so a
509
+ stray click does nothing. Questions you leave are never answered for you: the
510
+ oracle waits and asks again when you next write, and the designer asks again
511
+ before it may decide. A question with nobody to answer it (a one-shot
512
+ `pi -p` run) is put away at once.
520
513
 
521
514
  **Images.** An option can also carry an `image`: a PNG, JPEG, GIF or WebP
522
515
  file (a screenshot, a rendered mockup), shown above its preview text.
523
- Terminals with the Kitty graphics protocol (Kitty, Ghostty, WezTerm) show the
524
- image itself; other terminals draw PNGs as coloured half-blocks, and name the
525
- file for other formats. `BOT_LOBBY_IMAGES=blocks` always uses blocks, `off`
526
- never draws images.
527
516
 
528
517
  **The designer asks you directly.** During a task (not in auto mode) the
529
518
  designer worker can put its visual choices to you: its questions reach you
@@ -579,7 +568,7 @@ else TypeSafe.
579
568
  | Decision | Effect |
580
569
  | --- | --- |
581
570
  | Planning seats | Each round, only the seats the idea or your latest answers touch sit; `1`–`4` pins a seat |
582
- | Obvious answers | Answers a question itself when the conversation already makes the recommended option clearly right (≥ 0.9); listed under Assumptions |
571
+ | Obvious answers | Answers a question itself when the conversation already makes an option marked `(Recommended)` clearly right (≥ 0.9); listed under Assumptions |
583
572
  | File hints | Agents start with a short list of the files they most likely need, and get a `find_relevant_files` tool |
584
573
  | Relevant knowledge | When an agent's knowledge, standards or decisions file is too long for its prompt (over 4,000 characters), Jev keeps the sections that bear on the step, and the prompt says how many it left out and where the whole file is, so the agent can read the rest. A file that fits goes in whole, untouched; standards are never left empty |
585
574
  | Quick fix or task | Whether one engineer can do a new request alone decides whether it goes to the [quick-fix agent](#quick-fix-or-the-team) (the oracle confirms) |
@@ -601,8 +590,8 @@ switch on its own.
601
590
  ## Configuration
602
591
 
603
592
  Settings live in `~/.pi/bot-lobby/config.json` (`BOT_LOBBY_CONFIG_DIR`
604
- overrides). Edit them with `/bot-lobby settings`; `/bot-lobby config` shows
605
- the result.
593
+ overrides). Edit them on the lobby's Settings page (the cog in its top row;
594
+ it saves as you go); `/bot-lobby config` shows the result.
606
595
 
607
596
  ```json
608
597
  {
@@ -612,7 +601,7 @@ the result.
612
601
  },
613
602
  "scout": { "model": "anthropic/claude-haiku-4-5-20251001", "timeoutMs": 480000 },
614
603
  "planner": { "thinking": "high", "timeoutMs": 300000 },
615
- "lobby": { "planningPanel": ["backend", "designer", "qa", "researcher"], "maxPlanningRounds": 5, "splitPlanAbove": 8 },
604
+ "lobby": { "planningPanel": ["backend", "designer", "qa", "researcher"], "maxPlanningRounds": 5, "splitPlanAbove": 8, "web": { "port": 7347, "openBrowser": true } },
616
605
  "workflow": { "maxReviewIterations": 2, "maxParallelWorkers": 3, "stallTimeoutMs": 300000, "wrapUpAt": 0.75, "taskBudgetMinutes": 0, "fastTrack": true, "briefCheck": true, "routeQuickFixes": true, "gitIsolation": "off" },
617
606
  "classifier": { "enabled": false, "provider": "auto", "effort": { "cheapModel": "inherit" } }
618
607
  }
@@ -622,7 +611,8 @@ the result.
622
611
  `quickFix`, `planner`, `lobby`, `workflow`, `knowledge`, `classifier`.
623
612
  - An agent without a model runs on the session's model. Thinking is one of
624
613
  `off, minimal, low, medium, high, xhigh, max`, limited to what the model
625
- supports. Scouts always think at `low`.
614
+ supports: on the Settings page it is a slider whose unsupported stops are
615
+ struck through and cannot be chosen. Scouts always think at `low`.
626
616
  - `instructions` adds your own text to an agent's built-in prompt.
627
617
  - `fallbackModel` and `fallbackThinking` on any agent (and the master): see
628
618
  [Fallback models](#fallback-models).
@@ -709,7 +699,8 @@ BOT_LOBBY_LIVE_EXCALIDRAW=1 node --test test/excalidraw-live.test.ts
709
699
  ```
710
700
 
711
701
  Source layout: `src/workflow` (engine), `src/master` (delegation),
712
- `src/execution` (subagent processes), `src/lobby` (the UI),
702
+ `src/execution` (subagent processes), `src/lobby` (the lobby's state and service), `src/webui` and `webui/` (the
703
+ server and the page; rebuild with `npm run web:build`),
713
704
  `src/classifier` (Jev), `src/state` (persistence), `src/ask` (the
714
705
  questionnaire), `src/web` (the web tools), `src/excalidraw` (shared
715
706
  Excalidraw sessions: the room protocol, the sessions, the agents' tools),
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@a-t-h-i/bot-lobby",
3
- "version": "0.6.14",
3
+ "version": "0.6.16",
4
4
  "description": "Structured multi-agent software engineering orchestrator for Pi",
5
5
  "type": "module",
6
6
  "license": "Apache-2.0",
@@ -52,10 +52,11 @@
52
52
  "typebox": "*"
53
53
  },
54
54
  "devDependencies": {
55
+ "@axe-core/playwright": "4.13.0",
55
56
  "@earendil-works/pi-ai": "0.87.0",
56
57
  "@earendil-works/pi-coding-agent": "0.87.0",
57
58
  "@earendil-works/pi-tui": "0.87.0",
58
- "@axe-core/playwright": "4.13.0",
59
+ "@fontsource-variable/inter": "5.3.0",
59
60
  "@playwright/test": "1.62.0",
60
61
  "@shadcn/react": "0.3.1",
61
62
  "@tailwindcss/vite": "4.3.3",
@@ -66,6 +67,7 @@
66
67
  "class-variance-authority": "0.7.1",
67
68
  "clsx": "2.1.1",
68
69
  "lucide-react": "1.49.0",
70
+ "motion": "14.0.0",
69
71
  "radix-ui": "1.6.7",
70
72
  "react": "19.3.0",
71
73
  "react-dom": "19.3.0",
package/prompts/master.md CHANGED
@@ -27,8 +27,9 @@ process it gets. Fewer steps win whenever the result is the same.
27
27
  no plan document: delegate straight away with `orchestrate action=implement`,
28
28
  opening each task with `Step N:` (the engine keeps the plan and the
29
29
  checklist). Only the roster takes part: DESIGN for frontend work, DEV for
30
- backend work, QA when the change needs tests (its worker writing and
31
- running them as the last step, or the QA gate), and the researcher when a
30
+ backend work, QA when the change needs tests or an adversarial review
31
+ (its worker writing the few tests that are needed and running them as the
32
+ last step, or the QA gate), and the researcher when a
32
33
  decision needs outside facts (summon it first). Several domains: one
33
34
  `implement` with `assignments`, each task stating the contract between
34
35
  them. When the work is in, check `git diff --stat` and the report, then
@@ -70,14 +71,17 @@ otherwise. It is a hint, never a rule.
70
71
 
71
72
  When the user leaves your questions unanswered (they put them away, or
72
73
  `ask_user_question` says so), the decision is still theirs: never assume the
73
- answers, never fall back on the recommended options, and never carry on with
74
+ answers, never fall back on a marked option, and never carry on with
74
75
  work that depends on them. Say in one short line that the questions are
75
76
  waiting, end your turn, and ask again when they next write.
76
77
 
77
- When you `clarify` with options, put your recommended option first and mark
78
- it `(Recommended)`. When the request already makes it clearly right, the
79
- classifier answers for you: the reply says so, the decision is recorded, and
80
- you mention it in the proposal so the user can amend it.
78
+ When you `clarify`, do not recommend an answer: the point is that the user
79
+ thinks the decision through. List the options neutrally (no `(Recommended)`
80
+ marker, none first because you prefer it, no hint of your own pick) and let
81
+ the user decide. Mark an option `(Recommended)` only when the answer is quite
82
+ obvious from the request or the repository; then the classifier may answer for
83
+ you: the reply says so, the decision is recorded, and you mention it in the
84
+ proposal so the user can amend it.
81
85
 
82
86
  The designer worker may ask the user itself (outside auto mode): visual
83
87
  choices it cannot settle alone, shown with Markdown wireframes or rendered
package/prompts/panel.md CHANGED
@@ -25,9 +25,12 @@ questions and the user's answers, and the oracle's current draft plan.
25
25
  - Ask at most two questions, the most important first. They go to the
26
26
  oracle, who picks at most four for the user each round across the whole
27
27
  panel and decides the rest with your recommendation, so make each one
28
- short, plain and specific, and give it two to four options, your
29
- recommendation first with `(Recommended)` after its label. The user can
30
- always type their own answer instead, so do not add an "Other" option.
28
+ short, plain and specific, and give it two to four options in a neutral
29
+ order. Do not recommend one: the user should think the decision through.
30
+ Only when the answer is quite obvious from the conversation or the
31
+ repository, put that option first with `(Recommended)` after its label. The
32
+ user can always type their own answer instead, so do not add an "Other"
33
+ option.
31
34
  - Read the draft's Assumptions: if one the oracle made for your seat is
32
35
  wrong, say so under Notes and ask about it again.
33
36
  - If an answer from the user is vague or conflicts with what you see in the
@@ -44,11 +47,12 @@ OPEN or READY
44
47
 
45
48
  ## Questions
46
49
  1. The question, ending with a question mark?
47
- - Short label (Recommended) — what choosing it means
50
+ - Short label — what choosing it means
48
51
  - Another label — what choosing it means
49
52
 
50
- (Two to four options per question, labels of one to five words. Omit
51
- Questions when READY.)
53
+ (Two to four options per question, labels of one to five words, no
54
+ recommendation unless the answer is quite obvious, and then only that option
55
+ carries `(Recommended)`. Omit Questions when READY.)
52
56
 
53
57
  ## Notes
54
58
  - …
@@ -6,8 +6,8 @@ without guessing. The panel's domain members — DEV, DESIGN, QA and RESEARCH
6
6
  bring you their questions each round, and you decide which ones reach the
7
7
  user: you own the plan and the questions. You are thorough but you spare the
8
8
  user: every decision that changes the implementation gets made, either by the
9
- user or by you with the recommended option, written down as an assumption the
10
- user can overrule. You never
9
+ user or, when the answer is quite obvious, by you, written down as an
10
+ assumption the user can overrule. You never
11
11
  write code and never change files; you may read the repository to ask
12
12
  informed questions and to ground the plan in what exists.
13
13
 
@@ -30,23 +30,26 @@ below.
30
30
  one of the questions.
31
31
  - Keep each question short and plain: one line the user can answer at a
32
32
  glance. Give it two to four options — labels of one to five words and a
33
- short clause on what each means — your recommendation first with
34
- `(Recommended)` after its label. The user answers all of them together in
35
- one dialog and can type their own answer, so never add an "Other" option.
33
+ short clause on what each means — in a neutral order. Do not recommend an
34
+ option: the user should think each decision through. Only when the answer is
35
+ quite obvious from the conversation or the repository, put that option first
36
+ with `(Recommended)` after its label. The user answers all of them together
37
+ in one dialog and can type their own answer, so never add an "Other" option.
36
38
  - The conversation may carry an **Already settled with the user** list:
37
39
  questions the user answered (or left for you to decide). They are closed in
38
40
  any wording, so never ask one again, not even rephrased; fold the answer
39
41
  into the plan. The engine drops a repeat before the user sees it, so asking
40
42
  again only wastes a round. Ask about something new, or ask nothing.
41
43
  - Decide every question you do not ask, and any the user leaves unanswered,
42
- with its recommended option, and list those decisions under
44
+ with the marked option where there is one and otherwise your best call, and
45
+ list those decisions under
43
46
  `### Assumptions` in the plan, one line each, so the user can see and
44
47
  overrule them.
45
48
  - Challenge answers that are vague, contradictory or risky, and ask again.
46
49
  Do not accept "whatever you think" for a decision with real trade-offs:
47
50
  propose one and ask the user to confirm it.
48
51
  - The conversation may show questions **decided by the classifier**: a
49
- fast model answered them with their recommended option because the
52
+ fast model answered them with their marked option because the
50
53
  conversation already made it clearly right. Treat them as answered, list
51
54
  each under `### Assumptions` (the user can overrule it), and do not ask
52
55
  them again.
@@ -62,7 +65,7 @@ GRILLING.
62
65
  Planning may be limited to a number of rounds; your task says which round
63
66
  this is. Ask the questions that change the most early. In the final round,
64
67
  and in any round after it, no member runs and nothing more is asked: fold the
65
- answers into the plan, decide every open point with its recommended option,
68
+ answers into the plan, decide every open point (the marked option where there is one, otherwise your best call),
66
69
  list each under `### Assumptions`, omit the Questions section and set the
67
70
  status READY.
68
71
 
@@ -76,12 +79,12 @@ Three to six words naming the task.
76
79
 
77
80
  ## Questions
78
81
  1. [DEV] The most important open question?
79
- - Short label (Recommended) — what choosing it means
82
+ - Short label — what choosing it means
80
83
  - Another label — what choosing it means
81
84
  2. …
82
85
 
83
86
  (At most four questions, each tagged with its seat; two to four options per
84
- question, labels of one to five words. Omit the Questions section when READY
87
+ question, labels of one to five words, `(Recommended)` only on an obvious one. Omit the Questions section when READY
85
88
  or when you decided everything yourself.)
86
89
 
87
90
  ## Plan
package/prompts/qa.md CHANGED
@@ -3,7 +3,7 @@
3
3
  You are responsible for quality assurance and quality gates, not merely a test
4
4
  runner. Evaluate requirements, acceptance criteria, correctness, regression
5
5
  risk, edge cases, security, accessibility, UX, reliability, performance where
6
- relevant, and test coverage where applicable.
6
+ relevant, and the evidence the change works.
7
7
 
8
8
  ## Independence
9
9
 
@@ -12,10 +12,11 @@ reproduce important claims where possible. Passing automated tests does not
12
12
  automatically make a feature acceptable — tests are evidence, not the whole
13
13
  quality judgment.
14
14
 
15
- ## Test what can break
15
+ ## Review adversarially first
16
16
 
17
- Start from risk, not from coverage. For each change, ask how it fails and test
18
- those failure modes:
17
+ Your main tool is an adversarial review, not a test suite. Read the actual
18
+ diff and the code around it as someone trying to break it, and look for how it
19
+ fails:
19
20
 
20
21
  - invalid, malformed, boundary and empty input (zero, one, many, max, unicode)
21
22
  - error, timeout and retry paths; dependencies that are down or slow
@@ -23,14 +24,34 @@ those failure modes:
23
24
  - authorization: the wrong user, no user, an expired session
24
25
  - concurrency and ordering: double submits, races, out-of-order responses
25
26
  - regressions in the callers and consumers the change touches
27
+ - security, accessibility and UX problems the author is unlikely to have tried
26
28
 
27
- ## Do not overtest
29
+ Probe suspicions by reading, tracing and running the code (a one-off command or
30
+ script is fine), and report what you found as findings with evidence. A finding
31
+ you confirmed by running something is stronger than a test you added.
28
32
 
29
- Tests are proportional to risk. Every test names the failure it guards against;
30
- if you cannot say what bug it would catch, do not write it. No duplicate tests,
31
- no tests that mirror the implementation line by line, no snapshot spam, no
32
- testing of framework or library behavior, and no trivial getters. Prefer a few
33
- sharp tests on observable behavior over many shallow ones.
33
+ ## Write as few tests as possible
34
+
35
+ Tests cost tokens and add upkeep, so the default is to write none. Run the
36
+ project's existing tests for what the change touches and judge them; do not add
37
+ to them to look thorough.
38
+
39
+ Write a new test only when one of these holds, and say which in your report:
40
+
41
+ - the change breaks existing behavior on purpose (a breaking change to an API,
42
+ a format, a contract or a default), so the old tests must change or a new one
43
+ must pin the new behavior
44
+ - the change can introduce unpredictable behavior that review cannot settle:
45
+ concurrency, ordering, retries, time, randomness, parsing of untrusted input,
46
+ or state that outlives a request
47
+ - the task, or the user, asked for tests
48
+
49
+ When you do write one, it names the failure it guards against; if you cannot
50
+ say what bug it would catch, do not write it. One sharp test on observable
51
+ behavior beats several shallow ones. Never write duplicate tests, tests that
52
+ mirror the implementation line by line, snapshot spam, tests of framework or
53
+ library behavior, or tests for trivial getters. Do not rewrite tests that
54
+ still pass.
34
55
 
35
56
  ## Never pass by default
36
57
 
@@ -49,7 +70,8 @@ A PASS is a claim backed by evidence, not an absence of complaints.
49
70
 
50
71
  Use the project's existing test runner and conventions. Do not introduce a new
51
72
  testing framework without approval. Prefer tests that validate observable
52
- behavior; cover private helpers through public behavior. Always pass a bash
73
+ behavior; cover private helpers through public behavior. This section is only
74
+ about how to run and, rarely, write tests; the rules above decide whether to. Always pass a bash
53
75
  `timeout` for test runs, and never start watch mode or long-running servers.
54
76
 
55
77
  ## Domain boundary
@@ -70,8 +70,11 @@ If implementation changes are required, report them to the Master.
70
70
  - Verify each acceptance criterion; anything you could not verify is a
71
71
  finding, and an important unverifiable claim means CHANGES_REQUIRED or
72
72
  BLOCKED, never PASS.
73
- - Flag missing tests for real failure modes, and equally flag bloated,
74
- duplicate or implementation-mirroring tests.
73
+ - Review adversarially: try to break the change by reading and running it
74
+ rather than by asking for more tests. Flag a missing test only for a breaking
75
+ change or for behavior that could turn out unpredictable, and flag bloated,
76
+ duplicate or implementation-mirroring tests as findings: fewer tests is the
77
+ goal.
75
78
  - Always pass a bash `timeout` to test and build commands; never start watch
76
79
  mode or servers.
77
80
 
package/prompts/worker.md CHANGED
@@ -31,8 +31,11 @@ section of your output instead and continue with the rest of the work.
31
31
  ## Testing
32
32
 
33
33
  Run the targeted tests for what you changed (the files and behavior you
34
- touched); the QA gate runs the full suite afterwards. New public behavior,
35
- endpoints, and bug fixes require appropriate tests before claiming completion.
34
+ touched); the QA gate runs the full suite afterwards. Write as few tests as
35
+ possible: add one only for a breaking change, for a bug fix whose regression
36
+ would be silent, or for behavior that could turn out unpredictable (concurrency,
37
+ ordering, time, untrusted input), or when your brief asks for tests. Never pad
38
+ a change with tests for the sake of coverage.
36
39
 
37
40
  - Always pass a bash `timeout` to tests and builds (for example 300 seconds).
38
41
  - Never start dev servers, watch mode or other long-running processes, and
package/src/ask/relay.ts CHANGED
@@ -9,8 +9,7 @@
9
9
  import { tmpdir } from "node:os";
10
10
  import { isAbsolute, join, resolve } from "node:path";
11
11
  import type { ExtensionContext } from "@earendil-works/pi-coding-agent";
12
- import type { Asker } from "./dialog.ts";
13
- import { MAX_OPTIONS, MAX_QUESTIONS, type AskQuestion, type AskResult } from "./types.ts";
12
+ import { MAX_OPTIONS, MAX_QUESTIONS, type AskQuestion, type AskResult, type Asker } from "./types.ts";
14
13
 
15
14
  /** Env flag the master sets for a subagent that may ask the user. */
16
15
  export const ASK_ENV = "BOT_LOBBY_ASK";
package/src/ask/tool.ts CHANGED
@@ -10,11 +10,9 @@ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
10
10
  import { Container, Text } from "@earendil-works/pi-tui";
11
11
  import { Type } from "typebox";
12
12
  import { isSubagentProcess } from "../pi/quiet.ts";
13
- import { askUser, type Asker } from "./dialog.ts";
14
- import { promptHub } from "../lobby/prompt-hub.ts";
15
- import { isImagePath } from "./image.ts";
13
+ import { askUser } from "./web.ts";
16
14
  import { relayAsker, relayEnabled } from "./relay.ts";
17
- import { ASK_TOOL, MAX_HEADER, MAX_LABEL, MAX_OPTIONS, MAX_QUESTIONS, MIN_OPTIONS, RESERVED, type AskQuestion, type AskResult } from "./types.ts";
15
+ import { ASK_TOOL, isImagePath, MAX_HEADER, MAX_LABEL, MAX_OPTIONS, MAX_QUESTIONS, MIN_OPTIONS, RESERVED, type AskQuestion, type AskResult, type Asker } from "./types.ts";
18
16
 
19
17
  export { ASK_TOOL };
20
18
 
@@ -22,13 +20,13 @@ const OptionSchema = Type.Object({
22
20
  label: Type.String({ maxLength: MAX_LABEL, description: `The option as the user sees and picks it: 1-5 words, at most ${MAX_LABEL} characters.` }),
23
21
  description: Type.Optional(Type.String({ description: "What choosing it means: its trade-offs or consequences. Markdown." })),
24
22
  preview: Type.Optional(Type.String({ description: "Markdown shown beside the options while this one is focused: a mockup, a code snippet, a diagram, a config. Only when seeing it helps the user compare." })),
25
- image: Type.Optional(Type.String({ description: "Path to a PNG (or JPEG, GIF, WebP) shown with the preview: a screenshot or a rendered mockup of this option. Drawn as the image in terminals that can, as coloured blocks (PNG) elsewhere." })),
23
+ image: Type.Optional(Type.String({ description: "Path to a PNG (or JPEG, GIF, WebP) shown with the preview: a screenshot or a rendered mockup of this option." })),
26
24
  });
27
25
 
28
26
  const QuestionSchema = Type.Object({
29
27
  question: Type.String({ description: "The whole question, clear and specific, ending with a question mark. Markdown." }),
30
28
  header: Type.String({ maxLength: MAX_HEADER, description: `A short chip naming the question (at most ${MAX_HEADER} characters), e.g. "Auth" or "Layout".` }),
31
- options: Type.Array(OptionSchema, { minItems: MIN_OPTIONS, maxItems: MAX_OPTIONS, description: `${MIN_OPTIONS}-${MAX_OPTIONS} distinct options. Put the one you recommend first and end its label with "(Recommended)". The user can always answer in their own words instead.` }),
29
+ options: Type.Array(OptionSchema, { minItems: MIN_OPTIONS, maxItems: MAX_OPTIONS, description: `${MIN_OPTIONS}-${MAX_OPTIONS} distinct options in a neutral order. Do not recommend one: the user should think the decision through. Only when the answer is quite obvious, put that option first and end its label with "(Recommended)". The user can always answer in their own words instead.` }),
32
30
  multiSelect: Type.Optional(Type.Boolean({ description: "True when several answers can apply together." })),
33
31
  });
34
32
 
@@ -38,7 +36,7 @@ export const AskParams = Type.Object({
38
36
 
39
37
  const DESCRIPTION = [
40
38
  "Ask the user one to four questions with options to pick from, when the answer would change what you do and you would otherwise guess.",
41
- "Each question has 2-4 options (the one you recommend first, its label ending in \"(Recommended)\"); the user can pick one (or several with multiSelect), or answer in their own words.",
39
+ "Each question has 2-4 options in a neutral order, with no recommendation (mark one \"(Recommended)\", first, only when the answer is quite obvious); the user can pick one (or several with multiSelect), or answer in their own words.",
42
40
  "Questions, descriptions and previews are Markdown. Give options a `preview` when the user needs to see them to choose: a UI mockup, a layout sketch, a code snippet, a config; the focused option's preview shows beside the list. An option can also carry an `image` file (a screenshot, a rendered mockup).",
43
41
  "Do not use it for yes/no confirmations of what you were already told to do, or for questions the conversation already answers.",
44
42
  ].join(" ");
@@ -115,7 +113,7 @@ export function registerAskTool(pi: ExtensionAPI, ask?: Asker): void {
115
113
  const questions = (params as { questions: AskQuestion[] }).questions;
116
114
  const invalid = invalidQuestions(questions);
117
115
  if (invalid) throw new Error(`${invalid}.`);
118
- const result = await promptHub.run("questionnaire", "oracle", { questions }, () => asker(questions, ctx, signal), { signal });
116
+ const result = await asker(questions, ctx, signal, "oracle");
119
117
  return { content: [{ type: "text", text: answerSummary(questions, result) }], details: result };
120
118
  },
121
119
  renderCall(args, theme) {