@a-t-h-i/bot-lobby 0.4.0 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +140 -7
- package/package.json +1 -1
- package/prompts/master.md +12 -0
- package/prompts/panel.md +39 -0
- package/prompts/planner.md +64 -0
- package/prompts/quickfix.md +41 -0
- package/src/execution/agent-runner.ts +7 -2
- package/src/execution/pi-runner.ts +25 -0
- package/src/index.ts +3 -0
- package/src/lobby/feed.ts +253 -0
- package/src/lobby/issues.ts +227 -0
- package/src/lobby/layout.ts +174 -0
- package/src/lobby/planner.ts +474 -0
- package/src/lobby/quickfix.ts +227 -0
- package/src/lobby/runtime.ts +440 -0
- package/src/lobby/tabs/home.ts +164 -0
- package/src/lobby/tabs/issues.ts +72 -0
- package/src/lobby/tabs/metrics.ts +162 -0
- package/src/lobby/tabs/plan.ts +160 -0
- package/src/lobby/tabs/quickfix.ts +101 -0
- package/src/lobby/tabs/tasks.ts +209 -0
- package/src/lobby/view.ts +855 -0
- package/src/pi/activity.ts +120 -0
- package/src/pi/commands.ts +19 -57
- package/src/pi/events.ts +5 -2
- package/src/pi/model-support.ts +34 -0
- package/src/pi/run-summary.ts +2 -0
- package/src/pi/settings-ui.ts +4 -0
- package/src/pi/start-task.ts +63 -0
- package/src/pi/tools.ts +2 -2
- package/src/pi/ui.ts +116 -55
- package/src/schemas/configuration.ts +68 -3
- package/src/schemas/findings.ts +6 -0
- package/src/schemas/task.ts +1 -0
- package/src/state/backlog.ts +106 -0
- package/src/state/comments.ts +136 -0
- package/src/state/metrics.ts +305 -0
- package/src/workflow/workflow.ts +39 -7
package/README.md
CHANGED
|
@@ -74,6 +74,7 @@ structure.
|
|
|
74
74
|
## Usage
|
|
75
75
|
|
|
76
76
|
```
|
|
77
|
+
/bot-lobby Open the lobby (alt+l): tasks, planning, quick fixes, issues, metrics
|
|
77
78
|
/bot-lobby <request> Start a task and hand it to the Master
|
|
78
79
|
/bot-lobby status [taskId] Active task, state, approvals, blockers, legal next states
|
|
79
80
|
/bot-lobby tasks Task list (plus any unreadable task state)
|
|
@@ -89,8 +90,111 @@ structure.
|
|
|
89
90
|
/bot-lobby-settings Same as the settings subcommand
|
|
90
91
|
/bot-lobby minimize|restore Hide or restore bot-lobby for this session (ctrl+shift+m)
|
|
91
92
|
/bot-lobby claim <taskId> Take ownership of an orphaned task
|
|
93
|
+
/bot-lobby lobby | help Open the lobby, or show this list
|
|
92
94
|
```
|
|
93
95
|
|
|
96
|
+
## The lobby
|
|
97
|
+
|
|
98
|
+
The lobby is bot-lobby's full-screen home: a tabbed view over every task in the
|
|
99
|
+
project, your planning, quick fixes, GitHub issues and model performance, with
|
|
100
|
+
one prompt at the bottom whose target follows the tab. It opens by itself when
|
|
101
|
+
this session starts (or resumes) a task — the small zen widget returns whenever
|
|
102
|
+
you hide it — and `alt+l` or `/bot-lobby` opens and hides it at any time, with
|
|
103
|
+
or without a task.
|
|
104
|
+
|
|
105
|
+
```
|
|
106
|
+
◆ bot-lobby │ 1 Lobby 2 Tasks 2 3 Plan 4 Quick fix ⠋ 5 Issues 6 Metrics ⠋ TASK-add-login implementing
|
|
107
|
+
───────────────────────────────────────────────────────────────────────────────────────────────────────────
|
|
108
|
+
(the zen scene: the oracle, DEV · DESIGN · RESEARCH · QA, the plan checklist)
|
|
109
|
+
── Conversation · TASK-add-login ──────────────────── ┬ ── Activity ───────────────────────────────────────
|
|
110
|
+
you ▸ add a login page with email + password │ 12:04 MASTER ✓ scouting designer, backend
|
|
111
|
+
oracle ▸ Proposal: │ 12:06 DEV ⠋ reading auth.ts…
|
|
112
|
+
- LoginForm component │ 12:06 DESIGN ⠋ editing LoginForm.tsx…
|
|
113
|
+
- POST /api/login with rate limiting │ 12:06 QUICK FIX ✓ done: rename getUser
|
|
114
|
+
── Thinking ───────────────────────────────────────────────────────────────────────── DEV · 12s ago ──
|
|
115
|
+
The auth module already exposes a session helper; reuse it rather than adding a new one.
|
|
116
|
+
── message the oracle ─────────────────────────────────────────────────────────────────────────────────
|
|
117
|
+
_
|
|
118
|
+
TYPE enter send · shift+enter newline · esc browse · tab next tab · alt+l hide lobby
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
- **1 Lobby** — the task's zen scene, then the conversation with the oracle
|
|
122
|
+
(its text only: no tool rows, no thinking), an activity log that narrates
|
|
123
|
+
every tool call in plain words (`reading index.html…`, `searching for
|
|
124
|
+
"router" in src`, `running npm test`, `delegating to backend: Step 2 …`) from
|
|
125
|
+
the Master and every subagent, and a single **Thinking** pane — the one place
|
|
126
|
+
thoughts show up: the oracle's live thought as it streams, and each finished
|
|
127
|
+
thought from a subagent, quick fix or the planner (pi's own transcript,
|
|
128
|
+
behind the lobby, still carries the oracle's thinking blocks; `ctrl+t`
|
|
129
|
+
collapses them there). The prompt talks to the
|
|
130
|
+
oracle (while it works, enter steers the running turn; `esc` stops it); with
|
|
131
|
+
no task, it starts one.
|
|
132
|
+
- **2 Tasks** — every task in the project: this session's, the ones other pi
|
|
133
|
+
sessions are driving, pending plans saved from the planner, and recently
|
|
134
|
+
finished ones. The detail pane shows the request, the approved plan with its
|
|
135
|
+
step checklist, your comments on it, amendments, what the task waits on and
|
|
136
|
+
its recent runs. `c` comments on the selected task's plan (see below), `s`
|
|
137
|
+
starts a pending plan as a task in this session, `d` twice discards one.
|
|
138
|
+
- **3 Plan** — task planning mode with a planning panel. Describe what you
|
|
139
|
+
want and every seat grills you from its own domain, on the model and
|
|
140
|
+
thinking level its settings name: **DEV** (APIs, data, errors, security,
|
|
141
|
+
performance), **DESIGN** (flows, states, copy, visual language,
|
|
142
|
+
accessibility), **QA** (acceptance criteria, test strategy, edge cases,
|
|
143
|
+
definition of done) and **RESEARCH** (libraries, versions, docs and prior
|
|
144
|
+
art, with the web tools when `pi-web-access` is installed). The **oracle**
|
|
145
|
+
chairs on the Planner model: it reads the seats' questions and notes, folds
|
|
146
|
+
every answer into the draft plan (with a *Decisions by domain* section) and
|
|
147
|
+
asks only what no single seat owns. Each round the seats run in parallel,
|
|
148
|
+
read-only, then the oracle; the questions arrive numbered and attributed
|
|
149
|
+
(`3. QA Which browsers must pass?`), you answer them all in one message,
|
|
150
|
+
and every seat reads every answer the next round — so the agents that later
|
|
151
|
+
build the task start aligned. A roster shows what each seat is doing and
|
|
152
|
+
whether it is READY; the plan is READY only when every seat and the oracle
|
|
153
|
+
agree, and the draft pane lists what each seat said the plan must respect.
|
|
154
|
+
While browsing, `1`–`4` seat or unseat DEV, DESIGN, QA and RESEARCH for the
|
|
155
|
+
next round, `enter` switches between the conversation and the draft, `s`
|
|
156
|
+
saves the plan to the pending tasks list, `n` starts over, `r` retries a
|
|
157
|
+
round that failed or lost a seat, and `x` stops one.
|
|
158
|
+
- **4 Quick fix** — a direct prompt, the way you would ask pi, that skips the
|
|
159
|
+
whole workflow: one coding agent (full tools) makes the change right away
|
|
160
|
+
while any task keeps running. Quick fixes run one at a time in the order you
|
|
161
|
+
send them; each shows its steps and final report, and `x` cancels one. A
|
|
162
|
+
request that turns out to be large is reported back instead of attempted.
|
|
163
|
+
- **5 Issues** — the repository's open GitHub issues through the `gh` CLI (it
|
|
164
|
+
owns sign-in; bot-lobby stores no token). `enter` reads one with its
|
|
165
|
+
comments, `n` files a new one (first line is the title), `r` refreshes, and
|
|
166
|
+
`p` plans it: the Plan tab opens seeded with the issue, and the saved plan
|
|
167
|
+
keeps a link to it, so an issue becomes a task only after it has been
|
|
168
|
+
planned.
|
|
169
|
+
- **6 Metrics** — model performance across every Master turn, subagent run,
|
|
170
|
+
quick fix, planning seat and oracle planning turn: per model and thinking level, the number of runs,
|
|
171
|
+
success rate, mean/median/p90 time, turns, tools, tokens, output tokens per
|
|
172
|
+
second and cost (columns drop from the right on narrow terminals); how long a
|
|
173
|
+
task takes from request to done by the oracle's model and thinking level; and
|
|
174
|
+
where the time goes by agent. `g` splits the table by agent, `s` cycles the
|
|
175
|
+
sort (runs, average time, success, cost).
|
|
176
|
+
|
|
177
|
+
**Keys.** Like a modal editor, the lobby has a typing mode (keys go to the
|
|
178
|
+
prompt) and a browsing mode (`esc`; arrows move through lists, single keys run
|
|
179
|
+
the tab's commands, and on Lobby, Plan and Quick fix any other key resumes
|
|
180
|
+
typing). Everywhere: `tab`/`shift+tab` or `alt+1`…`alt+6` switch tabs,
|
|
181
|
+
`pageup`/`pagedown` scroll, `ctrl+c` clears the prompt (or hides the lobby
|
|
182
|
+
when it is empty) and `alt+l` hides the lobby. Anything that needs pi itself —
|
|
183
|
+
built-in slash commands, `/model`, the tool-row toggle — works with the lobby
|
|
184
|
+
hidden; bot-lobby's own `/bot-lobby …` commands also work from the Lobby
|
|
185
|
+
prompt. When the Master asks you something (an approval, a clarifying
|
|
186
|
+
question), the lobby steps aside for the dialog and comes back once you answer.
|
|
187
|
+
|
|
188
|
+
**Plan comments.** A comment on a task's plan is saved beside the task
|
|
189
|
+
(`comments.jsonl`) from any session, and the session that owns the task passes
|
|
190
|
+
new comments to its oracle — right away when you comment in that session,
|
|
191
|
+
within a few seconds from another one, held while the task is paused or the
|
|
192
|
+
session is minimized. The oracle treats a comment like an amendment and calls
|
|
193
|
+
`orchestrate action=plan` with the full revised plan, which replaces the
|
|
194
|
+
approved plan while implementing or reviewing, keeps finished steps done and
|
|
195
|
+
marks the comments addressed (`○` waiting, `◐` sent to the oracle, `✓` plan
|
|
196
|
+
amended). Before a plan exists, a comment asks for a revised proposal instead.
|
|
197
|
+
|
|
94
198
|
## Sessions and ownership
|
|
95
199
|
|
|
96
200
|
A task is owned by the pi session that started it (`ctx.sessionManager` id,
|
|
@@ -116,7 +220,8 @@ every in-flight subagent process.
|
|
|
116
220
|
|
|
117
221
|
While the owning session has a task active, its transcript switches to a zen view: `orchestrate` rows
|
|
118
222
|
and the built-in spinner are hidden, and a widget above the editor animates the
|
|
119
|
-
task
|
|
223
|
+
task (the same scene heads the lobby's first tab; the widget shows while the
|
|
224
|
+
lobby is hidden). At 72 columns and wider it draws a large scene: a header box with the task
|
|
120
225
|
title and state in its top border, a progress bar, and a metadata row with
|
|
121
226
|
elapsed time, quiet-mode hint and task id; an oracle tower with a twinkling
|
|
122
227
|
aura (drifting z's while dormant), a radiant orb crown, two window eyes, a
|
|
@@ -214,7 +319,7 @@ One tool, every workflow step. It is the Master's only way to move a task.
|
|
|
214
319
|
| `scout` | created…synthesizing | Run domain reconnaissance in parallel; repeat later to target-verify a claim |
|
|
215
320
|
| `research` | any active | Summon the read-only Researcher (domain + instruction) for cited internet evidence; persists the report for audit |
|
|
216
321
|
| `propose` | created…awaiting_approval | Record the proposal, request approval, handle approve/amend/decline |
|
|
217
|
-
| `plan` | planning | Record the internal plan (all §12 areas required) |
|
|
322
|
+
| `plan` | planning, implementing, reviewing | Record the internal plan (all §12 areas required); later, replace it with an amended plan (addresses lobby comments) |
|
|
218
323
|
| `implement` | planning, implementing, reviewing | Delegate a step to a domain Worker, or several domains at once with `assignments` (parallel, sharing files through the file desk) |
|
|
219
324
|
| `qa` | implementing, reviewing | Run the QA gate — the only review — over the whole feature |
|
|
220
325
|
| `knowledge` | any active | Record Master-approved knowledge or a decision |
|
|
@@ -339,6 +444,9 @@ top-level `/bot-lobby-settings`) and persist globally to
|
|
|
339
444
|
},
|
|
340
445
|
"scout": { "model": "anthropic/claude-haiku-4-5-20251001", "timeoutMs": 480000 },
|
|
341
446
|
"researcher": { "model": "anthropic/claude-sonnet-5", "thinking": "low", "instructions": "", "timeoutMs": 600000 },
|
|
447
|
+
"quickFix": { "model": "anthropic/claude-sonnet-5", "thinking": "low", "instructions": "", "timeoutMs": 600000 },
|
|
448
|
+
"planner": { "model": "anthropic/claude-sonnet-5", "thinking": "high", "instructions": "", "timeoutMs": 300000 },
|
|
449
|
+
"lobby": { "autoOpen": true, "planningPanel": ["backend", "designer", "qa", "researcher"] },
|
|
342
450
|
"workflow": {
|
|
343
451
|
"maxReviewIterations": 2,
|
|
344
452
|
"maxParallelScouts": 3,
|
|
@@ -373,6 +481,19 @@ visible; only the master keeps `inherit`, since it is the session itself.
|
|
|
373
481
|
Each subagent entry has a `timeoutMs` (default 15 min; scouts 8, researcher 10),
|
|
374
482
|
falling back to `workflow.agentTimeoutMs`.
|
|
375
483
|
|
|
484
|
+
The lobby's two agents have entries of their own: `quickFix` (the direct-change
|
|
485
|
+
agent, `low` thinking and 10 minutes by default) and `planner` (the oracle
|
|
486
|
+
chairing the planning panel, `high` thinking; its time limit bounds one round
|
|
487
|
+
for every seat, 5 minutes by default). Both appear in `/bot-lobby settings`,
|
|
488
|
+
take custom instructions, and run on the session's model until you pin one.
|
|
489
|
+
Planning seats reuse their domain's entry — DEV the Backend's, DESIGN the
|
|
490
|
+
Designer's, QA the QA's, RESEARCH the Researcher's model, thinking and
|
|
491
|
+
instructions — so a seat plans on the model that will later build its part.
|
|
492
|
+
`lobby.planningPanel` names the seats a new planning session starts with
|
|
493
|
+
(every seat by default; `[]` lets the oracle plan alone), and
|
|
494
|
+
`lobby.autoOpen` (default `true`) opens the lobby by itself when this session
|
|
495
|
+
starts or resumes a task.
|
|
496
|
+
|
|
376
497
|
`thinking` must be one of `off`, `minimal`, `low`, `medium`, `high`, `xhigh`,
|
|
377
498
|
`max`; a legacy `inherit` or unknown value falls back to `medium`. The thinking
|
|
378
499
|
picker lists only the levels the selected model supports. Switching to a model
|
|
@@ -416,8 +537,11 @@ and the output contract — and an empty layer is dropped.
|
|
|
416
537
|
├── Backend/knowledge/ knowledge.md, engineering-standards.md, decisions.md, completed-tasks.md
|
|
417
538
|
├── QA/knowledge/ knowledge.md, testing-standards.md, decisions.md, completed-tasks.md
|
|
418
539
|
├── archive/<Agent>/ previous knowledge versions (outside all retrieval paths)
|
|
540
|
+
├── backlog/PLAN-<slug>.json pending tasks saved from the planner (optionally linked to an issue)
|
|
541
|
+
├── metrics.jsonl one line per finished run of any agent, for the Metrics tab
|
|
419
542
|
└── tasks/TASK-<stamp>/
|
|
420
543
|
├── state.json the task record (kept after completion)
|
|
544
|
+
├── comments.jsonl your lobby comments on the plan and their delivery (append-only)
|
|
421
545
|
├── proposal.md scratchpads: deleted on completion
|
|
422
546
|
├── plan.md
|
|
423
547
|
├── designer.md backend.md qa.md
|
|
@@ -465,10 +589,19 @@ src/
|
|
|
465
589
|
├── desk/ File desk for parallel workers: checkout table, socket, worker extension
|
|
466
590
|
├── knowledge/ Paths, store (single write path), selector, compactor
|
|
467
591
|
├── prompts/ Layer loader + compiler
|
|
468
|
-
├──
|
|
592
|
+
├── lobby/
|
|
593
|
+
│ ├── runtime.ts Mounts the full-screen lobby on pi's TUI, dialogs hand-off, comment delivery, Master metrics
|
|
594
|
+
│ ├── view.ts The tabbed view: tab bar, per-tab prompt, typing/browsing modes, keys
|
|
595
|
+
│ ├── tabs/ Pure renderers: home, tasks, plan, quickfix, issues, metrics
|
|
596
|
+
│ ├── feed.ts Activity log, thinking pane and conversation store
|
|
597
|
+
│ ├── quickfix.ts Direct-change jobs, one at a time
|
|
598
|
+
│ ├── planner.ts The planning panel: seats and the oracle per round, reply parsing, saving a plan
|
|
599
|
+
│ ├── issues.ts GitHub issues through the gh CLI
|
|
600
|
+
│ └── layout.ts Exact-width columns, rules, wrapping and scroll windows
|
|
601
|
+
├── state/ Project root, config, task persistence, state mutation, comments, backlog, metrics
|
|
469
602
|
├── schemas/ Task, agent, findings, configuration types
|
|
470
603
|
└── pi/ Commands, lifecycle, orchestrate tool, status widget
|
|
471
|
-
prompts/ global, master, designer, backend, qa, scout, worker, reviewer, researcher
|
|
604
|
+
prompts/ global, master, designer, backend, qa, scout, worker, reviewer, researcher, quickfix, planner, panel
|
|
472
605
|
```
|
|
473
606
|
|
|
474
607
|
Prompts are composed, never duplicated: `global + domain + role + task context +
|
|
@@ -518,7 +651,7 @@ compaction, bounded review loops, dependency/architecture approval, retries,
|
|
|
518
651
|
cancellation, corrupted-state detection, and the commands/status UI.
|
|
519
652
|
|
|
520
653
|
Deliberately deferred (matching the build plan): worktree-based isolation for
|
|
521
|
-
parallel Workers (they share one working tree through the file desk)
|
|
522
|
-
dashboard, cost/token analytics beyond per-run usage, and
|
|
654
|
+
parallel Workers (they share one working tree through the file desk) and
|
|
523
655
|
cross-platform runtime abstractions. The internal module boundaries keep those
|
|
524
|
-
extractable.
|
|
656
|
+
extractable. The lobby (see above) has since added the full-screen dashboard
|
|
657
|
+
and per-model performance analytics.
|
package/package.json
CHANGED
package/prompts/master.md
CHANGED
|
@@ -77,6 +77,18 @@ asked. If the user
|
|
|
77
77
|
amends the request, reassess affected assumptions — never silently reinterpret
|
|
78
78
|
an amendment.
|
|
79
79
|
|
|
80
|
+
## Lobby comments
|
|
81
|
+
|
|
82
|
+
The user can comment on the approved plan (or the proposal) from the lobby, in
|
|
83
|
+
this session or another one. Each comment reaches you as a message naming the
|
|
84
|
+
task; open ones are also listed under `Open plan comments` in your task
|
|
85
|
+
context. Treat a comment like an amendment: reassess what it affects, then call
|
|
86
|
+
`orchestrate action=plan` with the full revised plan (it replaces the current
|
|
87
|
+
one while implementing or reviewing, keeps finished steps done, and marks the
|
|
88
|
+
comments addressed) before delegating more work. Before a plan exists, revise
|
|
89
|
+
the proposal and call `action=propose` again. If a comment needs no change,
|
|
90
|
+
say why in one line.
|
|
91
|
+
|
|
80
92
|
## Delegation
|
|
81
93
|
|
|
82
94
|
Assign work to the correct domain; never ask one domain to do another's. A
|
package/prompts/panel.md
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# Planning Panel Member
|
|
2
|
+
|
|
3
|
+
You sit on the planning panel for a task that has not started yet. The panel
|
|
4
|
+
is the oracle plus one member per domain — DEV, DESIGN, QA and RESEARCH — and
|
|
5
|
+
the user answers everyone's questions in one conversation, so every agent that
|
|
6
|
+
later works on the task starts from the same decisions.
|
|
7
|
+
|
|
8
|
+
You never write code or change files. You may read the repository (and, for
|
|
9
|
+
RESEARCH, the web) to ask sharper questions and to state facts.
|
|
10
|
+
|
|
11
|
+
## Each round
|
|
12
|
+
|
|
13
|
+
You receive the conversation so far, including every panel member's earlier
|
|
14
|
+
questions and the user's answers, and the oracle's current draft plan.
|
|
15
|
+
|
|
16
|
+
- Ask only what your seat owns (below), and only what would change how the
|
|
17
|
+
task is built or verified. Never repeat a question that has been answered,
|
|
18
|
+
or one another member already asked this round.
|
|
19
|
+
- Ask at most two questions, the most important first. Make each specific and
|
|
20
|
+
answerable; offer options (`a) … b) …`) and say which you would pick.
|
|
21
|
+
- If an answer from the user is vague or conflicts with what you see in the
|
|
22
|
+
repository, say so and ask again.
|
|
23
|
+
- Report what the plan must respect from your seat under Notes: facts from
|
|
24
|
+
files you read (name them), constraints, risks, what "done" means for you.
|
|
25
|
+
- When nothing in your seat is open any more, set the status to READY and ask
|
|
26
|
+
nothing.
|
|
27
|
+
|
|
28
|
+
## Output format
|
|
29
|
+
|
|
30
|
+
## Status
|
|
31
|
+
OPEN or READY
|
|
32
|
+
|
|
33
|
+
## Questions
|
|
34
|
+
1. …
|
|
35
|
+
|
|
36
|
+
(Omit Questions when READY.)
|
|
37
|
+
|
|
38
|
+
## Notes
|
|
39
|
+
- …
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# Task Planner
|
|
2
|
+
|
|
3
|
+
You are the oracle chairing a planning panel: you help the user turn an idea
|
|
4
|
+
(or a GitHub issue) into a task plan that the team of agents can execute
|
|
5
|
+
without guessing. The panel's domain members — DEV, DESIGN, QA and RESEARCH —
|
|
6
|
+
ask the user their own questions each round; you own the plan and the
|
|
7
|
+
questions no single domain owns. You are relentless: together you grill the
|
|
8
|
+
user until every decision that changes the implementation is made. You never
|
|
9
|
+
write code and never change files; you may read the repository to ask
|
|
10
|
+
informed questions and to ground the plan in what exists.
|
|
11
|
+
|
|
12
|
+
## Each turn
|
|
13
|
+
|
|
14
|
+
You receive the conversation so far (every member's questions and the user's
|
|
15
|
+
answers) and, under `## Panel this round`, each member's status, questions
|
|
16
|
+
and notes. Read the repository when it helps, then reply in the output format
|
|
17
|
+
below.
|
|
18
|
+
|
|
19
|
+
- Fold every member's notes and every answer into the draft plan, so each
|
|
20
|
+
domain's decisions are written down where all agents will read them. When
|
|
21
|
+
members disagree, say so and ask the user to decide.
|
|
22
|
+
- Ask at most three questions of your own, the most important first, and
|
|
23
|
+
only cross-cutting ones the members did not ask: scope and non-goals,
|
|
24
|
+
priorities, trade-offs between domains, sequencing, rollout and rollback.
|
|
25
|
+
Never repeat a member's question. Each one must be specific and answerable.
|
|
26
|
+
- Offer concrete options when they help (`a) … b) …`), and say which you
|
|
27
|
+
would pick and why.
|
|
28
|
+
- Challenge answers that are vague, contradictory or risky, and ask again.
|
|
29
|
+
Do not accept "whatever you think" for a decision with real trade-offs:
|
|
30
|
+
propose one and ask the user to confirm it.
|
|
31
|
+
- Ground every claim about the codebase in files you read; name them.
|
|
32
|
+
- Keep a draft plan updated every turn so the user sees it converge.
|
|
33
|
+
|
|
34
|
+
Declare the plan READY only when every panel member is READY and nothing
|
|
35
|
+
that would change the implementation is still open. Until then the status is
|
|
36
|
+
GRILLING.
|
|
37
|
+
|
|
38
|
+
## Output format
|
|
39
|
+
|
|
40
|
+
## Status
|
|
41
|
+
GRILLING or READY
|
|
42
|
+
|
|
43
|
+
## Title
|
|
44
|
+
Three to six words naming the task.
|
|
45
|
+
|
|
46
|
+
## Questions
|
|
47
|
+
1. The most important open question.
|
|
48
|
+
2. …
|
|
49
|
+
|
|
50
|
+
(Omit the Questions section when READY.)
|
|
51
|
+
|
|
52
|
+
## Plan
|
|
53
|
+
The current draft, in Markdown:
|
|
54
|
+
|
|
55
|
+
### Objective
|
|
56
|
+
### Scope and non-goals
|
|
57
|
+
### Acceptance criteria
|
|
58
|
+
### Affected areas
|
|
59
|
+
(files, modules and domains: designer, backend, qa)
|
|
60
|
+
### Decisions by domain
|
|
61
|
+
(what the user decided for DEV, DESIGN, QA and RESEARCH, one bullet each)
|
|
62
|
+
### Steps
|
|
63
|
+
1. …
|
|
64
|
+
### Risks and open points
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
# Quick Fix Agent
|
|
2
|
+
|
|
3
|
+
You make one small, direct code change the user asked for from the bot-lobby
|
|
4
|
+
lobby. There is no scouting, proposal, plan or review round: the user wants
|
|
5
|
+
the change now, the way they would ask pi directly.
|
|
6
|
+
|
|
7
|
+
## How to work
|
|
8
|
+
|
|
9
|
+
- Read only what you need to make the change safely; follow the file's
|
|
10
|
+
existing conventions.
|
|
11
|
+
- Make the smallest correct change that does exactly what was asked. Do not
|
|
12
|
+
refactor, rename or tidy anything else.
|
|
13
|
+
- If the request is ambiguous, pick the most reasonable reading and say which
|
|
14
|
+
one you chose in your report; do not stop to ask.
|
|
15
|
+
- If the change turns out to be large (many files, a new dependency, an
|
|
16
|
+
architecture change), make no edits and report what it would take, so the
|
|
17
|
+
user can plan it as a task instead.
|
|
18
|
+
- Run a quick targeted check when one exists (the nearest test file, a
|
|
19
|
+
typecheck of the touched package) with a bash `timeout`; never start dev
|
|
20
|
+
servers, watchers or background processes.
|
|
21
|
+
- Change files with `edit`/`write`, never through shell redirection or
|
|
22
|
+
`sed -i`.
|
|
23
|
+
|
|
24
|
+
## Other agents
|
|
25
|
+
|
|
26
|
+
A bot-lobby task may be running at the same time in this working tree. Touch
|
|
27
|
+
only the files the request needs, re-read a file right before editing it, and
|
|
28
|
+
never revert, reformat or "fix" changes you did not make.
|
|
29
|
+
|
|
30
|
+
## Report
|
|
31
|
+
|
|
32
|
+
End with a short report in this shape:
|
|
33
|
+
|
|
34
|
+
## Done
|
|
35
|
+
One or two sentences on what changed.
|
|
36
|
+
|
|
37
|
+
## Files
|
|
38
|
+
- path — what changed
|
|
39
|
+
|
|
40
|
+
## Checked
|
|
41
|
+
What you ran and the result, or "not checked" with the reason.
|
|
@@ -2,7 +2,7 @@ import type { Domain, Role } from "../schemas/agent.ts";
|
|
|
2
2
|
import type { AgentRun } from "../schemas/findings.ts";
|
|
3
3
|
import { roleSpec } from "../roles/registry.ts";
|
|
4
4
|
import { compilePrompt } from "../prompts/compiler.ts";
|
|
5
|
-
import { activityDetail, activityWord } from "../pi/activity.ts";
|
|
5
|
+
import { activityDetail, activityWord, describeToolCall } from "../pi/activity.ts";
|
|
6
6
|
import { shortDuration, truncate } from "../text.ts";
|
|
7
7
|
import { runPiAgent, spawnPiProcess, type PiStreamEvent, type ProcessRunner } from "./pi-runner.ts";
|
|
8
8
|
|
|
@@ -114,6 +114,7 @@ function baseRun(request: AgentRequest, runId: string, startedAt: string, attemp
|
|
|
114
114
|
output: "",
|
|
115
115
|
attempts,
|
|
116
116
|
startedAt,
|
|
117
|
+
...(request.thinking ? { thinking: request.thinking } : {}),
|
|
117
118
|
...(attempts > 1 ? { note: `retry ${attempts - 1} of ${(request.retries ?? 0)}`, noteKind: "warning" as const } : {}),
|
|
118
119
|
};
|
|
119
120
|
}
|
|
@@ -208,9 +209,13 @@ function createLiveRun(base: AgentRun, request: AgentRequest) {
|
|
|
208
209
|
case "tool_execution_start": {
|
|
209
210
|
const activity = activityWord(event.toolName);
|
|
210
211
|
const detail = activityDetail(event.toolName, event.args);
|
|
211
|
-
|
|
212
|
+
const step = describeToolCall(event.toolName, event.args);
|
|
213
|
+
emit({ activity, detail, step, tools: (state.tools ?? 0) + 1 }, changed(activity, detail) || step !== state.step);
|
|
212
214
|
return;
|
|
213
215
|
}
|
|
216
|
+
case "thought":
|
|
217
|
+
emit({ thought: event.text }, true);
|
|
218
|
+
return;
|
|
214
219
|
case "thinking":
|
|
215
220
|
case "writing":
|
|
216
221
|
emit({ activity: event.type, detail: undefined }, changed(event.type, undefined));
|
|
@@ -61,8 +61,13 @@ export type PiStreamEvent =
|
|
|
61
61
|
| { type: "compaction" }
|
|
62
62
|
| { type: "usage"; input: number; output: number; cost: number; model?: string }
|
|
63
63
|
| { type: "wrap_up" }
|
|
64
|
+
/** One finished thinking block, bounded to `MAX_THOUGHT_CHARS`. */
|
|
65
|
+
| { type: "thought"; text: string }
|
|
64
66
|
| { type: "heartbeat" };
|
|
65
67
|
|
|
68
|
+
/** Longest thought forwarded from a subagent stream. */
|
|
69
|
+
export const MAX_THOUGHT_CHARS = 1500;
|
|
70
|
+
|
|
66
71
|
export interface ProcessRunOptions {
|
|
67
72
|
cwd: string;
|
|
68
73
|
signal?: AbortSignal;
|
|
@@ -215,6 +220,11 @@ export function createStreamCollector(
|
|
|
215
220
|
phase = next;
|
|
216
221
|
onEvent?.({ type: next });
|
|
217
222
|
}
|
|
223
|
+
// One parse per finished thinking block (not per token) forwards the thought itself.
|
|
224
|
+
if (onEvent && line.includes('"thinking_end"')) {
|
|
225
|
+
const thought = finishedThought(line);
|
|
226
|
+
if (thought) onEvent({ type: "thought", text: thought });
|
|
227
|
+
}
|
|
218
228
|
};
|
|
219
229
|
const keep = (line: string) => {
|
|
220
230
|
deltaPhase(line);
|
|
@@ -261,6 +271,21 @@ export function createStreamCollector(
|
|
|
261
271
|
};
|
|
262
272
|
}
|
|
263
273
|
|
|
274
|
+
/** The text of a `thinking_end` message update, trimmed and bounded; undefined for anything else. */
|
|
275
|
+
export function finishedThought(line: string): string | undefined {
|
|
276
|
+
let event: { assistantMessageEvent?: { type?: string; content?: unknown } };
|
|
277
|
+
try {
|
|
278
|
+
event = JSON.parse(line) as typeof event;
|
|
279
|
+
} catch {
|
|
280
|
+
return undefined;
|
|
281
|
+
}
|
|
282
|
+
const update = event?.assistantMessageEvent;
|
|
283
|
+
if (update?.type !== "thinking_end" || typeof update.content !== "string") return undefined;
|
|
284
|
+
const text = update.content.trim();
|
|
285
|
+
if (!text) return undefined;
|
|
286
|
+
return text.length > MAX_THOUGHT_CHARS ? `${text.slice(0, MAX_THOUGHT_CHARS - 1)}…` : text;
|
|
287
|
+
}
|
|
288
|
+
|
|
264
289
|
/** Build the `pi` argv for one isolated, headless RPC agent run; the task goes over stdin. */
|
|
265
290
|
export function buildPiArgs(options: Omit<PiRunOptions, "task"> & { systemPromptFile?: string }): string[] {
|
|
266
291
|
const args = ["--mode", "rpc", "--no-session", "--no-prompt-templates", "--no-themes"];
|
package/src/index.ts
CHANGED
|
@@ -6,9 +6,12 @@ import { onTransition } from "./state/task-state.ts";
|
|
|
6
6
|
import { ping } from "./pi/notify.ts";
|
|
7
7
|
import { isSubagentProcess } from "./pi/quiet.ts";
|
|
8
8
|
import { registerDeskClient } from "./desk/client-extension.ts";
|
|
9
|
+
import { registerLobbyEvents } from "./lobby/runtime.ts";
|
|
9
10
|
|
|
10
11
|
export default function (pi: ExtensionAPI): void {
|
|
11
12
|
registerLifecycle(pi, CONFIG_DIR_NAME);
|
|
13
|
+
// After the lifecycle, so the lobby opens over a task the widget state already knows.
|
|
14
|
+
registerLobbyEvents(pi, CONFIG_DIR_NAME);
|
|
12
15
|
onTransition((task) => ping(task.state, task.title));
|
|
13
16
|
registerCommands(pi, CONFIG_DIR_NAME);
|
|
14
17
|
registerOrchestrateTool(pi, CONFIG_DIR_NAME);
|