thinkpool-pair 0.7.107 → 0.7.108

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/account.mjs CHANGED
@@ -420,13 +420,17 @@ export async function runAccount(SUPABASE_URL, SUPABASE_ANON) {
420
420
  // channel is RLS-gated to {me, P}); I only ever answer for rooms I serve AND share
421
421
  // with P (roomPartner === P) — defense in depth behind the channel firewall.
422
422
  const onPeerListReq = (partnerUid, ch, payload) => {
423
- if (process.env.TP_PAIRBUS_OFF === '1') return
423
+ // EN-M4 — the kill-switch must gate the ANSWERING side too. TP_CROSSROOM_OFF on THIS
424
+ // device means "no cross-session reach here", which has to include refusing to EXPOSE
425
+ // this device's rooms to a peer — otherwise a partner still reads in while you've opted
426
+ // out. (TP_PAIRBUS_OFF already disables the bus entirely.)
427
+ if (process.env.TP_PAIRBUS_OFF === '1' || process.env.TP_CROSSROOM_OFF === '1') return
424
428
  const rooms = [...children.keys()].filter((r) => roomPartner.get(r) === partnerUid).map((r) => ({ code: r, name: roomNames.get(r) || null, host: machine }))
425
429
  try { ch.send({ type: 'broadcast', event: 'peer-list-res', payload: { nonce: payload?.nonce, rooms } }) } catch { /* channel down */ }
426
430
  }
427
431
  const onPeerPeekReq = (partnerUid, ch, payload) => {
428
432
  const reply = (extra) => { try { ch.send({ type: 'broadcast', event: 'peer-peek-res', payload: { nonce: payload?.nonce, ...extra } }) } catch { /* channel down */ } }
429
- if (process.env.TP_PAIRBUS_OFF === '1') return reply({ error: 'Cross-session reach is off on this device.' })
433
+ if (process.env.TP_PAIRBUS_OFF === '1' || process.env.TP_CROSSROOM_OFF === '1') return reply({ error: 'Cross-session reach is off on this device.' })
430
434
  const target = String(payload?.targetRoom || '').toUpperCase().trim()
431
435
  const tc = children.get(target)
432
436
  if (!tc || roomPartner.get(target) !== partnerUid) return reply({ error: `Room ${target} is not a room this device shares with you.` })
package/bridge.mjs CHANGED
@@ -40,11 +40,26 @@ import { createSdkMcpServer, tool } from '@anthropic-ai/claude-agent-sdk'
40
40
  import { z } from 'zod'
41
41
  import { startClaudeSession } from './claude-session.mjs'
42
42
  import { FLOW_CONDUCTOR_PROMPT, FLOW_LANE_PROMPT, buildConductorEnv } from './flow-conductor.mjs'
43
+ import { normalizePlanOutput } from './flow-task-graph.mjs' // FL-B1 — validate the conductor's submit_flow_plan task-graph
43
44
  import { createFlowWorktree, worktreeSpec } from './flow-worktree.mjs'
44
- import { startPreview, stopAllPreviews } from './flow-preview.mjs'
45
- import { FLOW_REVIEWER_PROMPT, revertLane } from './flow-review.mjs'
45
+ import { startPreview, stopAllPreviews, previews } from './flow-preview.mjs'
46
+ // FL-M9 — per-lane preview servers leak (one per done lane, never stopped until shutdown).
47
+ // Lane previews are keyed `lane:<flowId>:<laneId>`; stop a whole flow's set when it assembles
48
+ // (the assembled preview supersedes them) or when a lane is reverted.
49
+ function stopFlowPreviews (flowId, laneId = null) {
50
+ for (const [key, h] of previews) {
51
+ if (typeof key !== 'string') continue
52
+ if (laneId ? key === `lane:${flowId}:${laneId}` : key.startsWith(`lane:${flowId}:`)) { try { h.stop() } catch { /* noop */ } }
53
+ }
54
+ }
55
+ import { FLOW_REVIEWER_PROMPT, revertLane, parseReviewVerdict } from './flow-review.mjs'
46
56
  import { mergeWorktrees, inlineSingleHtml, initRepo } from './flow-assembly.mjs'
47
- import { canDispatch, FLOW_LIMITS } from './flow-budget.mjs'
57
+ import { canDispatch, FLOW_LIMITS, makeBudget, recordSpend, killSwitchEnv } from './flow-budget.mjs'
58
+ // FL-B2 — per-flow HARD budget ledger (flowId → budget). Accumulates each lane's
59
+ // completed-turn output tokens (onEvent 'result') so the autopilot cap stops the next
60
+ // wave BEFORE overrun. Lives bridge-side because waves dispatch across separate
61
+ // broadcasts; without persistent state the cap can never bite.
62
+ const flowBudgets = new Map()
48
63
  import { formatPeek, PEEK, siblingsOf, resolveSibling, crossPostDecision, CROSSPOST, spawnDecision, CROSSROOM, formatPairRoster, crossRoomPostDecision } from './cross-terminal.mjs'
49
64
  import { turnInFlight } from './update-gate.mjs'
50
65
  import { saveSession, flushSession, deleteSession, loadAll, canResume, loadPtyId, savePtyId, loadNames, saveNames, appendDurableEvents, seedDurableEvents, readDurablePage, readDurableOldestSeq } from './session-store.mjs'
@@ -1041,6 +1056,48 @@ function openStructured({ id, model, resume, log, commands, mode, spawnedBy, rol
1041
1056
  // an old seq, which is what keeps the between-turns restart dup-free (see
1042
1057
  // event-id.mjs + src/pages/code/seqDedup.js). pushLog() is the single stamp point.
1043
1058
  entry.seq = makeSeqCounter(maxSeq(entry.log))
1059
+ entry.rolePrompt = rolePrompt || null // FL-M6 — persisted so a bridge restart restores the flow role prompt
1060
+ // FL-B1 (lane side) — record a flow lane's slice as done. Shared by the mark_flow_done
1061
+ // MCP tool AND the FLOW_DONE sentinel-Write intercept: like the conductor's submit, the
1062
+ // MCP tool is DEFERRED (needs ToolSearch, which hangs), so lanes signal done by Writing a
1063
+ // FLOW_DONE file — a DIRECT tool — which the PreToolUse hook routes here. Returns a message.
1064
+ const markFlowDone = async () => {
1065
+ if (!entry.flowSessionId || !entry.flowTaskKey) return 'Not a Flow lane — nothing to mark done.'
1066
+ if (entry.flowDone) return 'This slice is already recorded as done.'
1067
+ let commitSha = null
1068
+ try { commitSha = execFileSync('git', ['-C', entry.cwd || process.cwd(), 'rev-parse', 'HEAD'], { encoding: 'utf8' }).trim() } catch { /* no commits yet */ }
1069
+ let previewUrl = null
1070
+ try { const pv = await startPreview({ dir: entry.cwd || process.cwd(), id: `lane:${entry.flowSessionId}:${id}` }); previewUrl = pv.url } catch { /* preview best-effort */ }
1071
+ bcast('flow-task-done', { term: id, flowId: entry.flowSessionId, taskKey: entry.flowTaskKey, laneId: id, commitSha, previewUrl }, flowChannel)
1072
+ process.stderr.write(`\n ${A.cyan}◆ flow slice done — ${entry.flowTaskKey} (${commitSha ? commitSha.slice(0, 8) : 'no commit'})${previewUrl ? ` · preview ${previewUrl}` : ''}${A.rst}\n`)
1073
+ entry.flowDone = true // FL-B3 — retire immediately so it drops from the ≤8 lane cap
1074
+ setTimeout(() => { try { endStructured(id) } catch { /* already gone */ } }, 2500)
1075
+ return `Slice "${entry.flowTaskKey}" recorded as DONE${commitSha ? ` (commit ${commitSha.slice(0, 8)})` : ''}${previewUrl ? `. Live preview: ${previewUrl}` : ''}. The room dispatches dependent slices + assembles when all are done. You are finished — stop here.`
1076
+ }
1077
+ entry.markFlowDone = markFlowDone // FL-M6 — callable from the restore path to reconcile a finished-during-restart lane
1078
+ // FL-B4 — an adversarial REVIEW lane records its verdict by Writing FLOW_REVIEW.json (the
1079
+ // review gate was dead code: reviewers had no working emit path, so a failing slice was
1080
+ // never reverted). On FAIL → broadcast flow-revert for the reviewed slice (the client flips
1081
+ // it to pending → the next wave rebuilds it). Either way the review lane itself is done.
1082
+ const onReviewVerdict = async (raw) => {
1083
+ if (!entry.flowSessionId || !entry.flowTaskKey) return { ok: false, message: 'Not a Flow review lane.' }
1084
+ let v, target = null
1085
+ try {
1086
+ v = parseReviewVerdict(raw)
1087
+ const o = typeof raw === 'string' ? JSON.parse(String(raw).replace(/```(?:json)?|```/g, '').trim()) : raw
1088
+ target = (o && (o.taskKey || o.target)) || (entry.flowReviewTarget || null)
1089
+ } catch (e) {
1090
+ return { ok: false, message: `Review verdict REJECTED: ${e?.message || e}. Re-Write FLOW_REVIEW.json with {"pass":<boolean>,"reasons":["<specific finding>"],"taskKey":"<the slice you reviewed>"}.` }
1091
+ }
1092
+ if (!v.pass && target) {
1093
+ bcast('flow-revert', { term: id, flowId: entry.flowSessionId, taskKey: target }, flowChannel)
1094
+ process.stderr.write(`\n ${A.yel}◆ review FAIL — reverting ${target}: ${v.reasons.join('; ').slice(0, 120)}${A.rst}\n`)
1095
+ } else {
1096
+ process.stderr.write(`\n ${A.cyan}◆ review PASS — ${target || entry.flowTaskKey}${A.rst}\n`)
1097
+ }
1098
+ const doneMsg = await markFlowDone()
1099
+ return { ok: true, message: `Review verdict recorded: ${v.pass ? 'PASS' : `FAIL — reverting ${target || '(no target)'} (${v.reasons.join('; ').slice(0, 160)})`}. ${doneMsg}` }
1100
+ }
1044
1101
  // Identity for the durable archive — pushLog appends every new transcript event to
1045
1102
  // <room>/<id>.events.jsonl keyed off these. Seed the archive once from the restored
1046
1103
  // (≤2000) log so the retained window is immediately pageable; no-op if it already
@@ -1070,7 +1127,12 @@ function openStructured({ id, model, resume, log, commands, mode, spawnedBy, rol
1070
1127
  if (entry.log.length) process.stderr.write(`\n ◆ restored ${entry.log.length} prior events (${id.slice(0, 8)})${resume ? ' + resuming live context' : ''}.\n`)
1071
1128
  // Persist the permission mode alongside the transcript so a bridge restart
1072
1129
  // restores the session in the SAME mode (a bypass room stays bypass on resume).
1073
- const sessionData = () => ({ sessionId: entry.session?.sessionId || resume || null, log: entry.log, commands: entry.commands, mode: entry.mode, spawnedBy: entry.spawnedBy })
1130
+ // FL-M6 — persist the flow context so a bridge restart restores a lane's identity. Without
1131
+ // flowSessionId/flowTaskKey a restored lane can't mark done + the conductor loses its
1132
+ // subagent-block; without cwd it runs outside its worktree; without rolePrompt it loses its
1133
+ // Flow role. blockSubagents / onLaneDone / onReviewVerdict / onSubmitPlan all derive from
1134
+ // these, so restoring them restores the whole behavior.
1135
+ const sessionData = () => ({ sessionId: entry.session?.sessionId || resume || null, log: entry.log, commands: entry.commands, mode: entry.mode, spawnedBy: entry.spawnedBy, flowSessionId: entry.flowSessionId, flowTaskKey: entry.flowTaskKey, cwd: entry.cwd, rolePrompt: entry.rolePrompt, flowReviewTarget: entry.flowReviewTarget || null })
1074
1136
  const persist = () => saveSession(room, id, sessionData())
1075
1137
  // Synchronous flush of this session's record. Used on open (so a brand-new session
1076
1138
  // has a file under its id BEFORE its first event — surviving a restart inside the
@@ -1130,11 +1192,13 @@ function openStructured({ id, model, resume, log, commands, mode, spawnedBy, rol
1130
1192
  },
1131
1193
  async (args) => {
1132
1194
  const okText = (t) => ({ content: [{ type: 'text', text: t }] })
1133
- entry.pairPeekCount = (entry.pairPeekCount || 0) + 1
1134
- if (entry.pairPeekCount > CROSSROOM.peekPerTurnCap) return okText(`Cross-session read limit reached for this turn (${CROSSROOM.peekPerTurnCap}). Continue with what you have.`)
1195
+ // EN-m2 — validate BEFORE spending the per-turn budget so a malformed / self-room
1196
+ // (no-op) call doesn't burn the allowance.
1135
1197
  const target = String(args?.session || '').toUpperCase().trim()
1136
1198
  if (!target) return okText('Name a room code to read (call list_sessions to see them).')
1137
1199
  if (target === room) return okText('That is this room — use read_terminal for terminals in your own room.')
1200
+ entry.pairPeekCount = (entry.pairPeekCount || 0) + 1
1201
+ if (entry.pairPeekCount > CROSSROOM.peekPerTurnCap) return okText(`Cross-session read limit reached for this turn (${CROSSROOM.peekPerTurnCap}). Continue with what you have.`)
1138
1202
  const res = await pairRequest('pair-peek-req', { targetRoom: target, terminal: args?.terminal, lines: args?.lines })
1139
1203
  if (res?.error) return okText(res.error)
1140
1204
  return okText(res?.text || `Room ${target} returned nothing.`)
@@ -1198,6 +1262,9 @@ function openStructured({ id, model, resume, log, commands, mode, spawnedBy, rol
1198
1262
  if (target.kind !== 'agent') return okText(`Terminal ${target.ref} is a shell, not an agent — cross-posting is only allowed to agent terminals.`)
1199
1263
  const te = sessions.get(target.id)
1200
1264
  if (!te?.session) return okText(`Terminal ${target.ref} is no longer live.`)
1265
+ // Guard a missing/blank text (parity with post_to_session) — never inject an empty
1266
+ // turn, and don't burn the per-turn budget on a no-op.
1267
+ if (!args?.text || !String(args.text).trim()) return okText('Nothing to send — provide a non-empty `text` for the post.')
1201
1268
  entry.postCount = (entry.postCount || 0) + 1
1202
1269
  const fromRef = String(id).slice(0, 8)
1203
1270
  const msg = `[From: terminal ${fromRef}'s agent — relayed via ThinkPool cross-terminal, approved by a person in the room]\n${args.text}`
@@ -1253,7 +1320,10 @@ function openStructured({ id, model, resume, log, commands, mode, spawnedBy, rol
1253
1320
  // Inherit the spawner's permission mode by default (so a bypass orchestrator's
1254
1321
  // lanes run autonomously); an explicit `mode` overrides. The PreToolUse gate
1255
1322
  // already confirmed any bypass-escalation from a non-bypass lane before we got here.
1256
- const childMode = args?.mode || entry.mode
1323
+ // PM-m1 — never INHERIT plan mode into a spawned worker lane: a lane spawned from a
1324
+ // plan-mode lane would stall on ExitPlanMode cards. An explicit args.mode still wins;
1325
+ // otherwise inherit, but fall plan → default (an autonomous worker doesn't plan-gate).
1326
+ const childMode = args?.mode || (entry.mode === 'plan' ? 'default' : entry.mode)
1257
1327
  openStructured({ id: newId, model: args?.model, mode: childMode })
1258
1328
  const ne = sessions.get(newId)
1259
1329
  if (!ne) { if (args?.name) { delete termNames[newId]; saveNames(room, termNames) } return okText('Could not open a new lane — the terminal cap may have just been reached. Close one and retry.') }
@@ -1303,19 +1373,27 @@ function openStructured({ id, model, resume, log, commands, mode, spawnedBy, rol
1303
1373
  'mark_flow_done',
1304
1374
  "ThinkPool Flow ONLY — call this ONCE when your slice is built, RUNS, and meets its acceptance criteria. Records your latest commit as the slice's atomic-revert target and signals the room that this slice is done (which unblocks slices that depended on you). No arguments: your commit is read from your worktree. Do not call before your slice actually runs + meets acceptance.",
1305
1375
  {},
1306
- async () => {
1376
+ async () => ({ content: [{ type: 'text', text: await markFlowDone() }] }),
1377
+ ),
1378
+ // FL-B1 — the conductor SUBMITS its decomposition through THIS tool, not the built-in
1379
+ // ExitPlanMode. In the current SDK, ExitPlanMode is a deferred tool the conductor must
1380
+ // ToolSearch for — and it hangs there, wedging every flow in `planning`. A dedicated MCP
1381
+ // tool removes all dependence on plan-mode/ExitPlanMode/deferred-tool discovery: validate
1382
+ // the task-graph here, broadcast flow-plan (the same path the old interception used), and
1383
+ // reject malformed plans back to the conductor so it re-emits.
1384
+ tool(
1385
+ 'submit_flow_plan',
1386
+ 'ThinkPool Flow CONDUCTOR ONLY — submit your decomposition for human approval. Call this ONCE when your task-graph is ready. Pass it as `plan`: a JSON string {"summary":"<one line>","tasks":[{"key":"<kebab>","title":"…","scope":"…","acceptance":"…","deps":["<key>"],"sliceType":"feature|scaffold|review|fix"}]}. The room validates it is an acyclic DAG, persists the tasks, and shows the approval card. This is how a conductor FINISHES — do NOT use ExitPlanMode, do NOT write a plan file.',
1387
+ { plan: z.string().describe('the task-graph as a JSON string (the {summary, tasks:[…]} object)') },
1388
+ async (args) => {
1307
1389
  const okText = (t) => ({ content: [{ type: 'text', text: t }] })
1308
- if (!entry.flowSessionId || !entry.flowTaskKey) return okText('Not a Flow lane — nothing to mark done.')
1309
- let commitSha = null
1310
- try { commitSha = execFileSync('git', ['-C', entry.cwd || process.cwd(), 'rev-parse', 'HEAD'], { encoding: 'utf8' }).trim() } catch { /* no commits yet */ }
1311
- // Step 3 runtime: serve this lane's worktree so the room can iframe a live
1312
- // preview. Static serve — works for the no-build single-file artifact; URL
1313
- // rides the done broadcast.
1314
- let previewUrl = null
1315
- try { const pv = await startPreview({ dir: entry.cwd || process.cwd(), id: `lane:${id}` }); previewUrl = pv.url } catch { /* preview best-effort */ }
1316
- bcast('flow-task-done', { term: id, flowId: entry.flowSessionId, taskKey: entry.flowTaskKey, laneId: id, commitSha, previewUrl }, flowChannel)
1317
- process.stderr.write(`\n ${A.cyan}◆ flow slice done — ${entry.flowTaskKey} (${commitSha ? commitSha.slice(0, 8) : 'no commit'})${previewUrl ? ` · preview ${previewUrl}` : ''}${A.rst}\n`)
1318
- return okText(`Slice "${entry.flowTaskKey}" marked done${commitSha ? ` (commit ${commitSha.slice(0, 8)})` : ' (no commit yet — commit, then re-call)'}${previewUrl ? `. Live preview: ${previewUrl}` : ''}. The room records it + dispatches any slices that depended on you.`)
1390
+ if (!entry.flowSessionId) return okText('Not a Flow conductor — there is no flow to submit a plan for.')
1391
+ let norm
1392
+ try { norm = normalizePlanOutput(args?.plan || '') }
1393
+ catch (e) { return okText(`Plan REJECTED: ${e?.message || e}. Re-call submit_flow_plan with valid JSON — a non-empty "tasks" array, every "deps" entry referencing an existing task "key", and no cycles.`) }
1394
+ bcast('flow-plan', { term: id, flowId: entry.flowSessionId, plan: JSON.stringify(norm) }, flowChannel)
1395
+ process.stderr.write(`\n ${A.mag}◆ flow plan submitted — flow ${String(entry.flowSessionId).slice(0, 8)} (${norm.tasks.length} slice${norm.tasks.length === 1 ? '' : 's'}).${A.rst}\n`)
1396
+ return okText(`Plan submitted (${norm.tasks.length} slice${norm.tasks.length === 1 ? '' : 's'}) — the room is showing the human an approval card. You are DONE; stop here and wait for approval (lane dispatch is the room's job).`)
1319
1397
  },
1320
1398
  ),
1321
1399
  ],
@@ -1323,10 +1401,27 @@ function openStructured({ id, model, resume, log, commands, mode, spawnedBy, rol
1323
1401
  entry.session = startClaudeSession({
1324
1402
  cwd: cwd || process.cwd(), model, resume, mode, rolePrompt,
1325
1403
  // A Flow CONDUCTOR (flowSessionId set, no flowTaskKey) must DECOMPOSE, not explore via
1326
- // a fan-out of Task/Explore subagents — those raise permission cards in plan mode (which
1327
- // now land on the hidden conductor terminal) and the model ignores the prompt rule not
1328
- // to. Hard-block the Task tool for conductors so it reads with Read/Grep/Glob + ExitPlanMode.
1404
+ // a fan-out of Task/Explore subagents, and must not build. Hard-block Task/Bash/Write for
1405
+ // conductors so it reads with Read/Grep/Glob and submits via the FLOW_PLAN.json sentinel.
1329
1406
  blockSubagents: !!flowSessionId && !flowTaskKey,
1407
+ // FL-B1 — the conductor submits its task-graph by Writing FLOW_PLAN.json; the PreToolUse
1408
+ // hook intercepts that Write and calls this. Validate the DAG and broadcast flow-plan (the
1409
+ // client persists it + shows the approval card); a malformed plan comes straight back to
1410
+ // the conductor as the tool result so it re-emits. Returns { ok, message }.
1411
+ onSubmitPlan: (planText) => {
1412
+ if (!entry.flowSessionId) return { ok: false, message: 'Not a Flow conductor — nothing to submit.' }
1413
+ let norm
1414
+ try { norm = normalizePlanOutput(planText || '') }
1415
+ catch (e) { return { ok: false, message: `Plan REJECTED: ${e?.message || e}. Re-Write FLOW_PLAN.json with valid JSON — a non-empty "tasks" array, every "deps" entry referencing an existing task "key", and no cycles.` } }
1416
+ bcast('flow-plan', { term: id, flowId: entry.flowSessionId, plan: JSON.stringify(norm) }, flowChannel)
1417
+ process.stderr.write(`\n ${A.mag}◆ flow plan submitted — flow ${String(entry.flowSessionId).slice(0, 8)} (${norm.tasks.length} slice${norm.tasks.length === 1 ? '' : 's'}).${A.rst}\n`)
1418
+ return { ok: true, message: `Plan SUBMITTED (${norm.tasks.length} slice${norm.tasks.length === 1 ? '' : 's'}) — the room is showing the human an approval card. You are DONE: stop here, do not write anything else, wait for approval (lane dispatch is the room's job).` }
1419
+ },
1420
+ // FL-B1 (lane side) — a flow lane signals slice-done by Writing FLOW_DONE; the hook
1421
+ // routes that Write here (mark_flow_done MCP tool is deferred → ToolSearch → hangs).
1422
+ onLaneDone: flowTaskKey ? (async () => ({ ok: true, message: await markFlowDone() })) : null,
1423
+ // FL-B4 — a review lane records its verdict via FLOW_REVIEW.json (revert-on-fail).
1424
+ onReviewVerdict: flowTaskKey ? onReviewVerdict : null,
1330
1425
  mcpServers: { thinkpool: peekServer },
1331
1426
  // Tier C precheck — the PreToolUse gate calls this BEFORE raising a card, so a
1332
1427
  // disabled/looping/over-cap post never bothers a person. Closes over `entry`.
@@ -1354,6 +1449,21 @@ function openStructured({ id, model, resume, log, commands, mode, spawnedBy, rol
1354
1449
  // id, an event that arrives both live AND in a reconnect replay renders
1355
1450
  // twice (the 2026-06-19 duplicate-message bug). See event-id.mjs.
1356
1451
  stampEvent(evt)
1452
+ // FL-B2 — fold this flow lane's completed-turn output tokens into its budget so the
1453
+ // autopilot cap can halt the next wave before overrun (output_tokens = the billed
1454
+ // reasoning+output spend the indicator already tracks; conservative enough for a guard).
1455
+ if (entry.flowSessionId && entry.flowTaskKey && evt.kind === 'result' && evt.usage) {
1456
+ const b = flowBudgets.get(entry.flowSessionId)
1457
+ if (b) flowBudgets.set(entry.flowSessionId, recordSpend(b, evt.usage.output_tokens || 0))
1458
+ }
1459
+ // FL-B1 (lane) — a flow lane that ENDS ITS TURN is done: a bypass lane runs its slice
1460
+ // to completion in one turn, then narrates "done" and stops. Models often skip the
1461
+ // explicit signal (FLOW_DONE / mark_flow_done), which left the slice stuck 'dispatched'
1462
+ // forever → the flow never assembled. Auto-record on turn-end (idempotent — markFlowDone
1463
+ // no-ops if already done, so the explicit sentinel still works as the fast path).
1464
+ if (entry.flowSessionId && entry.flowTaskKey && evt.kind === 'result' && !entry.flowDone) {
1465
+ Promise.resolve().then(() => markFlowDone()).catch(() => { /* best-effort */ })
1466
+ }
1357
1467
  // The init system event carries the session's slash command list. Stash it
1358
1468
  // on the entry so the ANNOUNCE can hand it to clients that connect/reload
1359
1469
  // AFTER init (the one-time code-event would miss them), then re-announce.
@@ -1458,6 +1568,24 @@ function allowPending(s) {
1458
1568
  }
1459
1569
  }
1460
1570
 
1571
+ // PM-M3 — switching to acceptEdits must ALSO retroactively clear the cards it would
1572
+ // auto-approve (non-destructive writes), same reasoning as allowPending for bypass:
1573
+ // setPermissionMode only applies going forward, so without this a flip to acceptEdits
1574
+ // leaves the user staring at write cards the new mode is meant to auto-allow. Only
1575
+ // write-class, non-high-risk cards clear; Bash/network/destructive/plan/ask stay gated.
1576
+ const ACCEPT_EDIT_TOOLS = new Set(['Write', 'Edit', 'MultiEdit', 'NotebookEdit'])
1577
+ function acceptEditsPending(s) {
1578
+ if (!s?.pending) return
1579
+ for (const [key, p] of s.pending) {
1580
+ const tn = p?.payload?.toolName
1581
+ const risk = p?.payload?.risk
1582
+ if (risk === 'plan' || risk === 'ask' || risk === 'high') continue
1583
+ if (!ACCEPT_EDIT_TOOLS.has(tn)) continue
1584
+ try { p.resolve('allow') } catch { /* noop */ }
1585
+ s.pending.delete(key)
1586
+ }
1587
+ }
1588
+
1461
1589
  function endStructured(id) {
1462
1590
  if (!id) return
1463
1591
  const s = sessions.get(id)
@@ -1468,6 +1596,13 @@ function endStructured(id) {
1468
1596
  sessions.delete(id)
1469
1597
  bcast('term-exit', { id })
1470
1598
  fireImageCleanup(id) // drop this term's screenshots from Storage (COGS)
1599
+ // EN-m3 — closing a spawner ORPHANS the lanes it spawned (only that spawner could
1600
+ // close_terminal them; they can never be closed otherwise + keep occupying the caps).
1601
+ // Cascade-close its direct children (spawnedBy === this id — a real terminal id, not a
1602
+ // `flow:<id>` marker, so flow lanes are untouched; maxHop:1 means no deeper recursion).
1603
+ for (const [cid, ce] of [...sessions]) {
1604
+ if (ce.spawnedBy === id) endStructured(cid)
1605
+ }
1471
1606
  }
1472
1607
  deleteSession(room, id) // remove <id>.json so loadAll() won't bring it back
1473
1608
  if (s) announce()
@@ -1674,6 +1809,7 @@ channel
1674
1809
  s.postCount = 0
1675
1810
  s.pairPeekCount = 0 // cross-room read budget resets with the in-room ones (Tier 1)
1676
1811
  s.crossRoomPostCount = 0 // Tier 3 cross-room post budget resets on a real human turn
1812
+ s.spawnCount = 0 // EN-M3 — spawn budget MUST reset too, else turn 2+ can never spawn a lane
1677
1813
  s.hop = 0
1678
1814
  s.roomHop = 0 // a human turn is room-hop 0 — clears any injected cross-room hop depth
1679
1815
  const text = String(payload.text)
@@ -1741,6 +1877,8 @@ channel
1741
1877
  s.mode = payload.mode; s.session.setMode(payload.mode) // track for persist/restore
1742
1878
  // Flipping to bypass = "stop asking, including the cards this turn already raised."
1743
1879
  if (payload.mode === 'bypassPermissions') allowPending(s)
1880
+ // Flipping to acceptEdits = "auto-approve writes" — clear the write cards already raised.
1881
+ else if (payload.mode === 'acceptEdits') acceptEditsPending(s)
1744
1882
  }
1745
1883
  })
1746
1884
  .on('broadcast', { event: 'code-close' }, ({ payload }) => {
@@ -1788,7 +1926,21 @@ channel
1788
1926
  // Closed sessions are unlinked by deleteSession, so a file existing on disk ==
1789
1927
  // a session that was live at shutdown (never an explicitly closed one).
1790
1928
  const all = loadAll(room)
1791
- if (all.length) for (const rec of all) openStructured({ id: rec.id, resume: canResume(rec) ? rec.sessionId : undefined, log: rec.log, commands: rec.commands, mode: rec.mode, spawnedBy: rec.spawnedBy })
1929
+ if (all.length) for (const rec of all) {
1930
+ // FL-M6 — restore the flow context (id/role/cwd) so an in-flight flow survives a
1931
+ // bridge restart: the conductor keeps its subagent-block + plan interception, and
1932
+ // lanes keep their worktree cwd + the ability to mark done.
1933
+ openStructured({ id: rec.id, resume: canResume(rec) ? rec.sessionId : undefined, log: rec.log, commands: rec.commands, mode: rec.mode, spawnedBy: rec.spawnedBy, flowSessionId: rec.flowSessionId, flowTaskKey: rec.flowTaskKey, cwd: rec.cwd, rolePrompt: rec.rolePrompt })
1934
+ const re = sessions.get(rec.id)
1935
+ if (re && rec.flowReviewTarget) re.flowReviewTarget = rec.flowReviewTarget
1936
+ // FL-M6 — a flow LANE whose turn had already ENDED before the restart finished its
1937
+ // slice, but its done-signal was lost while the bridge was down (resume won't re-run
1938
+ // an ended turn, so it would sit idle → the flow stalls forever). Reconcile: re-record
1939
+ // it done. A lane still MID-turn (restoredTurnOpen) resumes + finishes normally.
1940
+ if (re && rec.flowTaskKey && !restoredTurnOpen(rec.log || [])) {
1941
+ setTimeout(() => { try { re.markFlowDone?.() } catch { /* noop */ } }, 2000)
1942
+ }
1943
+ }
1792
1944
  // Fresh open: ONLY for an explicit `-- <claude>` share. In account mode
1793
1945
  // (autoAgent set, no attachedCmd) the web's `?new=1` flow opens the first
1794
1946
  // terminal; opening one here too raced it into TWO terminals — each a
@@ -1830,9 +1982,11 @@ flowChannel
1830
1982
  .on('broadcast', { event: 'flow-start' }, ({ payload }) => {
1831
1983
  // Ensemble Flow — the room asks this bridge to launch a CONDUCTOR for a Flow run.
1832
1984
  // The room already created the flow_sessions row (flowId); the conductor opens in
1833
- // plan mode (so ExitPlanMode is permitted), seeded with the build prompt. It
1834
- // decomposes → calls ExitPlanMode(planJson) → the requestPermission flow-plan branch
1835
- // broadcasts flow-plan back. `host` routes to one bridge in multi-machine rooms.
1985
+ // DEFAULT mode (NOT plan mode — FL-B1: plan mode triggers the model's ExitPlanMode
1986
+ // reflex, and ExitPlanMode is a deferred tool that hangs ToolSearch in the SDK, wedging
1987
+ // every flow in `planning`). It can ONLY read (Read/Grep/Glob auto-allow) + submit: its
1988
+ // Write/Edit/Bash/Task are hard-blocked (blockSubagents), and it finishes by calling the
1989
+ // submit_flow_plan MCP tool → flow-plan broadcast. `host` routes one bridge in multi-machine rooms.
1836
1990
  if (!payload?.flowId) return
1837
1991
  if (payload.host && payload.host !== name) return
1838
1992
  for (const [, e] of sessions) if (e.flowSessionId === payload.flowId) return // already conducting this flow
@@ -1843,7 +1997,7 @@ flowChannel
1843
1997
  // lanes form ONE Ensemble group (same ensembleRoot) — the drilled flow shows a row of
1844
1998
  // all its terminals to hop between, with a working Back. (No flowTaskKey → it's still
1845
1999
  // the conductor, excluded from the lane cap below.)
1846
- openStructured({ id: cid, mode: 'plan', rolePrompt: FLOW_CONDUCTOR_PROMPT, flowSessionId: payload.flowId, spawnedBy: `flow:${payload.flowId}` })
2000
+ openStructured({ id: cid, mode: 'default', rolePrompt: FLOW_CONDUCTOR_PROMPT, flowSessionId: payload.flowId, spawnedBy: `flow:${payload.flowId}` })
1847
2001
  const ce = sessions.get(cid)
1848
2002
  if (ce?.session) { try { ce.session.sendTurn(payload.prompt || '') } catch { /* session still starting */ } }
1849
2003
  process.stderr.write(`\n ${A.mag}◆ flow conductor launched — flow ${String(payload.flowId).slice(0, 8)} (plan mode).${A.rst}\n`)
@@ -1860,9 +2014,26 @@ flowChannel
1860
2014
  const assignments = []
1861
2015
  // Step 6 — cap concurrent Flow lanes (invariant: ≤ FLOW_LIMITS.maxConcurrentLanes).
1862
2016
  // Overflow tasks stay pending; the room re-dispatches them in the next wave.
1863
- let liveLanes = [...sessions.values()].filter(e => typeof e.spawnedBy === 'string' && e.spawnedBy.startsWith('flow:') && e.flowTaskKey).length // lanes only — conductor has no flowTaskKey
2017
+ // Live lanes = flow lanes still running (NOT retired-on-done — FL-B3). Conductor has
2018
+ // no flowTaskKey so it never counts.
2019
+ let liveLanes = [...sessions.values()].filter(e => typeof e.spawnedBy === 'string' && e.spawnedBy.startsWith('flow:') && e.flowTaskKey && !e.flowDone).length
2020
+ // FL-B2 — real budget: cap from the dispatch payload (flow_sessions.budget_cap, autopilot
2021
+ // only) with the operator env override; kill-switch refuses all dispatch. A per-flow spend
2022
+ // ledger accumulates across waves (flowBudgets) so the cap bites BEFORE overrun.
2023
+ const ks = killSwitchEnv()
2024
+ if (ks.disabled) { process.stderr.write(`\n ${A.yel}◆ flow dispatch refused — TP_FLOW_OFF kill-switch is on.${A.rst}\n`); return }
2025
+ const dispMode = payload.mode || 'guide'
2026
+ const capTokens = ks.capFromEnv != null ? ks.capFromEnv : (payload.budgetCap != null ? Number(payload.budgetCap) : null)
2027
+ let budget = flowBudgets.get(payload.flowId) || makeBudget({ capTokens })
2028
+ if (budget.capTokens !== capTokens) budget = { ...budget, capTokens } // cap can arrive late
2029
+ flowBudgets.set(payload.flowId, budget)
1864
2030
  for (const t of payload.tasks) {
1865
- const gate = canDispatch({ mode: payload.mode || 'guide', liveLanes, budget: { capTokens: null, spentTokens: 0, killed: false } })
2031
+ // FL-M7 — idempotent dispatch: never spawn a 2nd lane into a slice that already has a
2032
+ // live lane (a re-broadcast / resend would otherwise collide in the same worktree).
2033
+ if ([...sessions.values()].some(e => e.flowSessionId === payload.flowId && e.flowTaskKey === t.task_key && !e.flowDone)) {
2034
+ process.stderr.write(`\n ${A.dim}◆ flow dispatch skip ${t.task_key} — lane already live.${A.rst}\n`); continue
2035
+ }
2036
+ const gate = canDispatch({ mode: dispMode, liveLanes, budget })
1866
2037
  if (!gate.ok) { process.stderr.write(`\n ${A.yel}◆ flow dispatch held (${gate.reason}, ${liveLanes}/${FLOW_LIMITS.maxConcurrentLanes} lanes) — ${assignments.length} spawned this wave.${A.rst}\n`); break }
1867
2038
  try {
1868
2039
  // Step 4 — a `review` slice gets the adversarial reviewer prompt (try-to-break),
@@ -1877,16 +2048,30 @@ flowChannel
1877
2048
  // a Flow summoned from a bypass terminal runs hands-off).
1878
2049
  openStructured({ id: laneId, cwd: dir, mode: 'bypassPermissions', rolePrompt: isReview ? FLOW_REVIEWER_PROMPT : FLOW_LANE_PROMPT, flowSessionId: payload.flowId, flowTaskKey: t.task_key, spawnedBy: `flow:${payload.flowId}` })
1879
2050
  const le = sessions.get(laneId)
1880
- if (le?.session) {
1881
- const spec =
1882
- `[Flow lane — slice "${t.task_key}" of flow ${String(payload.flowId).slice(0, 8)}]\n` +
1883
- `TITLE: ${t.title}\n` +
1884
- `SCOPE (files you OWN — edit ONLY these): ${t.scope || '(none stated)'}\n` +
1885
- `ACCEPTANCE (done = this runs + proves it): ${t.acceptance || '(meet the title)'}\n` +
1886
- (t.deps && t.deps.length ? `DEPENDS ON (already built): ${t.deps.join(', ')}\n` : '') +
1887
- `\nProject (context): ${payload.flowPrompt || ''}\n\n` +
1888
- `Build your slice. Own only your files. Done = it runs + meets acceptance. Commit when done.`
1889
- try { le.session.sendTurn(spec) } catch { /* session still starting */ }
2051
+ if (le) {
2052
+ // FL-B4 — a review slice reviews the slices it DEPENDS ON; remember the primary
2053
+ // target so a FAIL verdict knows what to revert even if the model omits taskKey.
2054
+ if (isReview) le.flowReviewTarget = (t.deps && t.deps[0]) || null
2055
+ if (le.session) {
2056
+ const spec = isReview
2057
+ ? `[Flow REVIEW lane — slice "${t.task_key}" of flow ${String(payload.flowId).slice(0, 8)}]\n` +
2058
+ `You are ADVERSARIALLY reviewing the slice(s): ${(t.deps || []).join(', ') || t.title}\n` +
2059
+ (Array.isArray(t.deps) && t.deps.length
2060
+ ? `Reviewed slice worktree(s) — check out + RUN each yourself:\n` +
2061
+ t.deps.map((dep) => { const ws = worktreeSpec({ flowId: payload.flowId, taskKey: dep }); return ` - ${dep}: dir ${ws.dir} (branch ${ws.branch})` }).join('\n') + '\n'
2062
+ : '') +
2063
+ `ACCEPTANCE to INDEPENDENTLY verify: ${t.acceptance || t.title}\n` +
2064
+ `\nProject (context): ${payload.flowPrompt || ''}\n\n` +
2065
+ `Review it — run it, hunt for failure. When done, use the Write tool to write FLOW_REVIEW.json with {"pass":<boolean>,"reasons":["<specific finding>"],"taskKey":"<the slice you reviewed>"}. That Write IS your verdict.`
2066
+ : `[Flow lane — slice "${t.task_key}" of flow ${String(payload.flowId).slice(0, 8)}]\n` +
2067
+ `TITLE: ${t.title}\n` +
2068
+ `SCOPE (files you OWN — edit ONLY these): ${t.scope || '(none stated)'}\n` +
2069
+ `ACCEPTANCE (done = this runs + proves it): ${t.acceptance || '(meet the title)'}\n` +
2070
+ (t.deps && t.deps.length ? `DEPENDS ON (already built): ${t.deps.join(', ')}\n` : '') +
2071
+ `\nProject (context): ${payload.flowPrompt || ''}\n\n` +
2072
+ `Build your slice. Own only your files. Done = it runs + meets acceptance. Commit when done.`
2073
+ try { le.session.sendTurn(spec) } catch { /* session still starting */ }
2074
+ }
1890
2075
  }
1891
2076
  assignments.push({ task_key: t.task_key, laneId })
1892
2077
  liveLanes++
@@ -1923,6 +2108,7 @@ flowChannel
1923
2108
  try { initRepo({ dir: outDir }) } catch { /* best-effort */ }
1924
2109
  let previewUrl = null
1925
2110
  try { const pv = await startPreview({ dir: outDir, id: `assembled:${payload.flowId}` }); previewUrl = pv.url } catch { /* best-effort */ }
2111
+ stopFlowPreviews(payload.flowId) // FL-M9 — the assembled preview supersedes the per-lane ones
1926
2112
  bcast('flow-assembled', { term: 'flow', flowId: payload.flowId, previewUrl, outDir, files: merged.files, conflicts: merged.conflicts }, flowChannel)
1927
2113
  process.stderr.write(`\n ${A.mag}◆ flow assembled — ${merged.files.length} file(s)${merged.conflicts.length ? `, ${merged.conflicts.length} conflict(s)` : ''}${previewUrl ? ` · preview ${previewUrl}` : ''} (flow ${short}).${A.rst}\n`)
1928
2114
  announce()
@@ -1930,13 +2116,26 @@ flowChannel
1930
2116
  process.stderr.write(`\n ${A.yel}◆ flow assemble failed: ${e?.message || e}${A.rst}\n`)
1931
2117
  }
1932
2118
  })
1933
- .on('broadcast', { event: 'flow-revert' }, ({ payload }) => {
2119
+ .on('broadcast', { event: 'flow-revert' }, async ({ payload }) => {
1934
2120
  // Step 4 — atomic per-lane revert: roll back ONE slice (remove its worktree +
1935
2121
  // branch) without touching the others, when adversarial review rejects it. The
1936
2122
  // room re-dispatches the reverted task on the next wave.
1937
2123
  if (!payload?.flowId || !payload?.taskKey) return
1938
2124
  if (payload.host && payload.host !== name) return
1939
2125
  try {
2126
+ // FL-M8 — tear down the lane's live session FIRST. revertLane does `git worktree
2127
+ // remove --force`; pulling the worktree out from under a still-running lane corrupts
2128
+ // its process state. Retire it, let the SDK release the cwd, THEN remove the tree.
2129
+ let killed = false
2130
+ for (const [sid, e] of sessions) {
2131
+ if (e.flowSessionId === payload.flowId && e.flowTaskKey === payload.taskKey && !e.flowDone) {
2132
+ e.flowDone = true
2133
+ try { endStructured(sid) } catch { /* already gone */ }
2134
+ stopFlowPreviews(payload.flowId, sid) // FL-M9 — drop the reverted lane's preview
2135
+ killed = true
2136
+ }
2137
+ }
2138
+ if (killed) await new Promise((r) => setTimeout(r, 400))
1940
2139
  const { branch } = revertLane({ flowId: payload.flowId, taskKey: payload.taskKey })
1941
2140
  bcast('flow-reverted', { term: 'flow', flowId: payload.flowId, taskKey: payload.taskKey, branch }, flowChannel)
1942
2141
  process.stderr.write(`\n ${A.yel}◆ flow slice reverted — ${payload.taskKey} (${branch}).${A.rst}\n`)
@@ -120,7 +120,7 @@ const simplifyBlocks = (blocks = []) => blocks.map((b) => {
120
120
  */
121
121
  const MODES = new Set(['default', 'acceptEdits', 'plan', 'bypassPermissions'])
122
122
 
123
- export function startClaudeSession({ cwd, model, resume, env, mode: initialMode = 'default', onEvent, requestPermission, mcpServers, crossPostGate, crossRoomPostGate, rolePrompt, blockSubagents = false }) {
123
+ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode = 'default', onEvent, requestPermission, mcpServers, crossPostGate, crossRoomPostGate, rolePrompt, blockSubagents = false, onSubmitPlan = null, onLaneDone = null, onReviewVerdict = null }) {
124
124
  const ac = new AbortController()
125
125
  let input = makeInputStream() // `let`: auto-restart swaps in a fresh stream
126
126
  let sessionId = resume || null
@@ -249,23 +249,66 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
249
249
  const lvl = hookInput.effort?.level
250
250
  if (lvl && lvl !== effort) { effort = lvl; emit({ kind: 'effort', level: effort }) }
251
251
  // Flow CONDUCTOR hard-stop: a conductor must DECOMPOSE using ONLY Read/Grep/Glob (which
252
- // are auto-allowed, no card) then ExitPlanMode. Two things otherwise stall it on a
252
+ // are auto-allowed, no card) then ExitPlanMode. FOUR tool classes otherwise stall it on a
253
253
  // permission card on the now-hidden conductor terminal: (1) Task/Agent subagents — the
254
254
  // model's parallelize instinct fans out Explore agents despite the prompt; (2) Bash —
255
255
  // the model reaches for `find`/`grep`/`wc`, and Bash is medium-risk → a card in plan
256
- // mode. Deny BOTH and tell it to use Grep/Glob/Read instead. (blockSubagents is set only
257
- // for the conductor session — lanes build under their own bypass/one-hop rules.)
258
- if (blockSubagents && (toolName === 'Task' || toolName === 'Agent' || toolName === 'Bash')) {
256
+ // mode; (3) Write/Edit — the model's plan-mode habit is to WRITE A PLAN DOC
257
+ // (`~/.claude/plans/*.md`) instead of emitting the task-graph; that write pops a card AND
258
+ // never produces tasks, so the flow wedges in `planning` forever (FL-B1). Deny ALL and
259
+ // redirect to ExitPlanMode. (blockSubagents is set only for the conductor session —
260
+ // lanes build under their own bypass/one-hop rules.)
261
+ // FL-B1 — the conductor SUBMITS its decomposition by Writing the task-graph JSON to the
262
+ // sentinel file FLOW_PLAN.json. Write is a DIRECT (non-deferred) tool, so this avoids the
263
+ // deferred-tool/ToolSearch path that hangs every other submit route (ExitPlanMode AND the
264
+ // submit_flow_plan MCP tool are deferred → ToolSearch → hang). We intercept that Write
265
+ // here (the file is never actually written): validate via onSubmitPlan + broadcast the
266
+ // plan, then feed the result straight back as the tool result so the conductor re-emits on
267
+ // a malformed plan or stops on success.
268
+ const isConductorWrite = toolName === 'Write' || toolName === 'Edit' || toolName === 'MultiEdit' || toolName === 'NotebookEdit'
269
+ const writePath = toolInput?.file_path || toolInput?.notebook_path || ''
270
+ if (blockSubagents && isConductorWrite && /(?:^|[/\\])FLOW_PLAN\.json$/i.test(writePath)) {
271
+ const planText = toolInput?.content ?? toolInput?.new_string ?? ''
272
+ let res = { ok: false, message: 'Plan submission is not wired for this session.' }
273
+ try { res = (onSubmitPlan && (await onSubmitPlan(planText))) || res } catch (e) { res = { ok: false, message: `Plan submission failed: ${e?.message || e}` } }
274
+ return { continue: true, hookSpecificOutput: { hookEventName: 'PreToolUse', permissionDecision: 'deny', permissionDecisionReason: res.message } }
275
+ }
276
+ if (blockSubagents && (toolName === 'Task' || toolName === 'Agent' || toolName === 'Bash' || isConductorWrite)) {
259
277
  const reason = (toolName === 'Bash')
260
- ? 'Flow conductors do NOT run Bash — it pops a permission card and stalls the flow. Use the Grep tool (not `grep`/`rg`), the Glob tool (not `find`/`ls`), and Read (not `cat`/`head`) — those run with no prompt. Then DECOMPOSE and call ExitPlanMode once.'
261
- : 'Flow conductors do NOT spawn subagents — it stalls the flow on a permission card. Inspect the codebase yourself with Read/Grep/Glob (no prompt). Then DECOMPOSE and call ExitPlanMode once. Do not call Task/Agent again.'
278
+ ? 'Flow conductors do NOT run Bash — it pops a permission card and stalls the flow. Use the Grep tool (not `grep`/`rg`), the Glob tool (not `find`/`ls`), and Read (not `cat`/`head`) — those run with no prompt. Then DECOMPOSE and submit by Writing your task-graph JSON to the file FLOW_PLAN.json.'
279
+ : isConductorWrite
280
+ ? 'Flow conductors do NOT write project or plan-document files. Do NOT use ExitPlanMode and do NOT call submit_flow_plan (both hang here). To SUBMIT your decomposition, use the Write tool with file_path "FLOW_PLAN.json" and content = your task-graph JSON ({"summary":"…","tasks":[…]}). That single Write IS how you submit — the room ingests it directly. Do not write any other file.'
281
+ : 'Flow conductors do NOT spawn subagents — it stalls the flow on a permission card. Inspect the codebase yourself with Read/Grep/Glob (no prompt). Then DECOMPOSE and submit by Writing your task-graph JSON to the file FLOW_PLAN.json. Do not call Task/Agent again.'
262
282
  return { continue: true, hookSpecificOutput: { hookEventName: 'PreToolUse', permissionDecision: 'deny', permissionDecisionReason: reason } }
263
283
  }
284
+ // FL-B4 (review lane) — an adversarial reviewer records its verdict by Writing
285
+ // FLOW_REVIEW.json; the hook routes it to the verdict logic (revert-on-fail + mark the
286
+ // review slice done). Same deferred-tool workaround as the builder's FLOW_DONE.
287
+ if (onReviewVerdict && (toolName === 'Write' || toolName === 'Edit') && /(?:^|[/\\])FLOW_REVIEW(\.[a-zA-Z0-9]+)?$/.test(toolInput?.file_path || '')) {
288
+ let res = { ok: false, message: 'verdict signal failed' }
289
+ try { res = (await onReviewVerdict(toolInput?.content ?? toolInput?.new_string ?? '')) || res } catch (e) { res = { ok: false, message: `verdict signal failed: ${e?.message || e}` } }
290
+ return { continue: true, hookSpecificOutput: { hookEventName: 'PreToolUse', permissionDecision: 'deny', permissionDecisionReason: res.message } }
291
+ }
292
+ // FL-B1 (lane) — a flow lane signals its slice is done by Writing a file named FLOW_DONE
293
+ // (the mark_flow_done MCP tool is deferred → ToolSearch → hangs; Write is direct). Route
294
+ // that Write to the done logic and feed the result back; the file is never written.
295
+ if (onLaneDone && (toolName === 'Write' || toolName === 'Edit') && /(?:^|[/\\])FLOW_DONE(\.[a-zA-Z0-9]+)?$/.test(toolInput?.file_path || '')) {
296
+ let res = { ok: false, message: 'done signal failed' }
297
+ try { res = (await onLaneDone()) || res } catch (e) { res = { ok: false, message: `done signal failed: ${e?.message || e}` } }
298
+ return { continue: true, hookSpecificOutput: { hookEventName: 'PreToolUse', permissionDecision: 'deny', permissionDecisionReason: res.message } }
299
+ }
264
300
  // ThinkPool cross-terminal READ (B2) is read-only + within-room — never
265
301
  // prompt, in any mode. It still surfaces as a tool card so the room sees the peek.
266
302
  if (toolName === 'mcp__thinkpool__read_terminal') {
267
303
  return { continue: true, hookSpecificOutput: { hookEventName: 'PreToolUse', permissionDecision: 'allow', permissionDecisionReason: 'Auto-approved (read-only ThinkPool cross-terminal read).' } }
268
304
  }
305
+ // FL-B1 — the Flow conductor submits its task-graph through this tool (replaces the
306
+ // deferred/hanging ExitPlanMode). It only broadcasts a plan for HUMAN approval — no FS or
307
+ // system effect — so auto-allow it (the conductor runs in plan mode, which would otherwise
308
+ // card it and re-wedge the flow).
309
+ if (toolName === 'mcp__thinkpool__submit_flow_plan') {
310
+ return { continue: true, hookSpecificOutput: { hookEventName: 'PreToolUse', permissionDecision: 'allow', permissionDecisionReason: 'Auto-approved (Flow plan submission — broadcasts the task-graph for human approval).' } }
311
+ }
269
312
  // ThinkPool cross-ROOM READ (list_sessions / read_session) is read-only. It reaches
270
313
  // the account's own other rooms on this machine (Tier 1) and the partner's rooms on
271
314
  // the pair bus (Tier 2) — both firewalled server-side (the supervisor only serves
@@ -24,9 +24,9 @@ export { normalizePlanOutput }
24
24
  // moat, partition by file ownership, plan-in-store-not-context, right-size the slice
25
25
  // count (3 focused beats 7 scattered), explicit acyclic deps.
26
26
  export const FLOW_CONDUCTOR_PROMPT = [
27
- 'THINKPOOL FLOW CONDUCTOR — HARD RULES (read first, no exceptions): You are in PLAN MODE. Your ONLY output action is calling the ExitPlanMode tool EXACTLY ONCE with the task-graph JSON. You MUST NOT write or edit files, MUST NOT run bash/git, MUST NOT call AskUserQuestion, and MUST NOT build the project yourself — in plan mode those are blocked and will just stall you on a permission prompt. You do NOT build; you DECOMPOSE so the lanes build. A fully-specified build is STILL decomposed, never built here. If a detail is ambiguous, make a reasonable assumption and note it in the plan summary — do NOT ask. Think, then call ExitPlanMode. That is the whole job.',
27
+ 'THINKPOOL FLOW CONDUCTOR — HARD RULES (read first, no exceptions): You are a DECOMPOSITION-ONLY conductor. Your ONLY output action is using the Write tool ONCE to write your task-graph JSON to the file named exactly FLOW_PLAN.json. That single Write IS how you submit your plan — the room ingests it directly. Do NOT use ExitPlanMode and do NOT call submit_flow_plan (both hang here). Do NOT write any other file and do NOT write a plan-document markdown. You MUST NOT write or edit files, MUST NOT run bash/git, MUST NOT call AskUserQuestion, and MUST NOT build the project yourself — those are blocked and will just stall you. You do NOT build; you DECOMPOSE so the lanes build. A fully-specified build is STILL decomposed, never built here. If a detail is ambiguous, make a reasonable assumption and note it in the plan summary — do NOT ask. Think, then call submit_flow_plan. That is the whole job.',
28
28
 
29
- 'DO NOT SPAWN SUBAGENTS. You MUST NOT use the Task/Agent tool or launch Explore / sub-agents — in plan mode every subagent spawn pops a permission card and stalls the whole flow (this is the #1 way the conductor hangs asking for permission). You almost never need to look at code to decompose: decompose from the build request itself. If you genuinely must inspect something first, read it YOURSELF with Read/Grep/Glob (those are allowed in plan mode without a prompt) — never delegate it to a spawned agent. Decompose, then ExitPlanMode.',
29
+ 'DO NOT SPAWN SUBAGENTS. You MUST NOT use the Task/Agent tool or launch Explore / sub-agents — every subagent spawn pops a permission card and stalls the whole flow (this is the #1 way the conductor hangs asking for permission). You almost never need to look at code to decompose: decompose from the build request itself. If you genuinely must inspect something first, read it YOURSELF with Read/Grep/Glob (those are allowed without a prompt) — never delegate it to a spawned agent. Decompose, then Write FLOW_PLAN.json.',
30
30
 
31
31
  'THINKPOOL FLOW — you are the CONDUCTOR of an ensemble build. Your job RIGHT NOW (Step 1) is to decompose the user\'s build request into a TASK-GRAPH: a set of RUNNABLE slices with explicit dependencies, then submit that plan for human approval. Do NOT start building and do NOT spawn lanes yet — that happens only after the user approves the plan (Step 2+).',
32
32
 
@@ -40,9 +40,9 @@ export const FLOW_CONDUCTOR_PROMPT = [
40
40
 
41
41
  'RIGHT-SIZE THE SLICE COUNT. Bias to FEW slices that each carry real weight, not many tiny ones. For a small app: 2-3 slices. Concurrent lanes are bounded (a low dispatch ceiling), so a flat fan-out of well-chosen slices beats a sprawling one. Three focused slices outperform seven scattered ones.',
42
42
 
43
- 'SUBMIT VIA ExitPlanMode — when the decomposition is ready, CALL the ExitPlanMode tool, passing your plan as ONE JSON object (a JSON string) in its `plan` argument. The room intercepts that call, validates the plan is a real DAG (acyclic, all deps resolve), persists it to the task-graph store, and shows a plan-approval card. A malformed plan is rejected and you will be asked to re-emit. Do not also narrate the plan in prose. The exact shape:',
43
+ 'SUBMIT BY WRITING FLOW_PLAN.json — when the decomposition is ready, use the Write tool with file_path "FLOW_PLAN.json" and content set to your plan as ONE JSON object. The room validates the plan is a real DAG (acyclic, all deps resolve), persists it to the task-graph store, and shows a plan-approval card. A malformed plan is rejected with the reason as the tool result — fix it and Write FLOW_PLAN.json again. Do not also narrate the plan in prose. The exact shape:',
44
44
  '{ "summary": "<one line: what we are building>", "tasks": [ { "key": "<short-unique-kebab-key>", "title": "<imperative title>", "scope": "<files/components this lane owns>", "acceptance": "<runnable proof it is done>", "deps": ["<other-task-key>", ...], "sliceType": "feature|scaffold|review|fix" } ] }',
45
- '`key` values are stable short kebab-case identifiers. `deps` references other tasks\' keys. Put ONLY the JSON in the ExitPlanMode plan argument (no surrounding prose, no markdown fences).',
45
+ '`key` values are stable short kebab-case identifiers. `deps` references other tasks\' keys. The content of FLOW_PLAN.json must be ONLY the JSON object (no surrounding prose, no markdown fences).',
46
46
 
47
47
  'PLAN LIVES IN THE STORE, NOT THE CHAT. Your JSON plan is the artifact — it gets persisted to the task-graph (flow_sessions / flow_tasks) and rendered as an approval card. Do not ALSO narrate a long plan in prose. You may add 1-2 sentences framing the decomposition choice (why these slices, what was coupled), nothing more.',
48
48
 
@@ -77,7 +77,7 @@ export const FLOW_LANE_PROMPT = [
77
77
 
78
78
  'DONE MEANS RUNS. You are not done when you\'ve written code — you\'re done when your slice RUNS and meets the `acceptance` criteria (the observable, runnable proof). Verify it yourself before declaring done. State the proof (the command, the output) when you finish.',
79
79
 
80
- 'COMMIT WHEN DONE, THEN SIGNAL. You are in your own git worktree on your own branch. When your slice meets acceptance — it RUNS and you have verified the proof — commit it atomically (one focused commit); that commit is the revert target if review later rejects it. Then call the `mark_flow_done` tool ONCE: it records your commit as this slice\'s revert target and tells the room the slice is done (which unblocks any slices depending on you). Do not push. Do NOT call mark_flow_done before your slice actually runs + meets acceptance.',
80
+ 'COMMIT WHEN DONE, THEN SIGNAL. You are in your own git worktree on your own branch. When your slice meets acceptance — it RUNS and you have verified the proof — commit it atomically (one focused commit) with `git` via the Bash tool; that commit is the revert target if review later rejects it. THEN, as your VERY LAST action, signal completion by using the Write tool to write a file named exactly FLOW_DONE in your worktree root (any content, e.g. "done"). That single Write IS the done signal — it records your commit + tells the room the slice is done (unblocking dependents). Do NOT use mark_flow_done (it hangs here). Do not push. Do NOT write FLOW_DONE before your slice actually runs, meets acceptance, AND is committed.',
81
81
 
82
82
  'STAY IN YOUR LANE. Do NOT spawn further lanes (one hop only — the fork-bomb breaker). Do NOT edit other worktrees. If you\'re blocked on a dependency that isn\'t ready, say so + stop — another lane is building it.',
83
83
 
package/flow-review.mjs CHANGED
@@ -26,7 +26,7 @@ export const FLOW_REVIEWER_PROMPT = [
26
26
 
27
27
  'BE SPECIFIC. Your reasons must name exactly WHAT failed and HOW you found it — the command you ran, the output you got, the acceptance criterion it violated. "Doesn\'t work" is useless. "GET /api/todos returned 500 with `column todos.user_id does not exist`; acceptance required 200 + []" is a usable verdict.',
28
28
 
29
- 'EMIT A VERDICT, NOT A mark_flow_done. You do NOT call mark_flow_done — that is the builder\'s path. You emit a REVIEW VERDICT: a single JSON object `{ "pass": <boolean>, "reasons": ["<specific finding>", ...], "taskKey": "<the slice you reviewed>" }`. `pass: false` means revert this lane. Put ONLY that JSON object in your verdict (no surrounding prose, no markdown fences). Do not edit the slice — you review, you do not fix.',
29
+ 'EMIT YOUR VERDICT BY WRITING FLOW_REVIEW.json. Do NOT call mark_flow_done or ExitPlanMode (they hang here). As your LAST action, use the Write tool with file_path "FLOW_REVIEW.json" and content = a single JSON object { "pass": <boolean>, "reasons": ["<specific finding>", ...], "taskKey": "<the slice you reviewed>" }. `pass: false` reverts the reviewed slice so it rebuilds. The content of FLOW_REVIEW.json must be ONLY that JSON object (no prose, no markdown fences). Do not edit the slice — you review, you do not fix.',
30
30
  ].join(' ')
31
31
 
32
32
  // Parse a reviewer's raw verdict. Accepts an object OR a JSON string (optionally
@@ -153,14 +153,22 @@ export function normalizePlanOutput (raw) {
153
153
  if (!obj || typeof obj !== 'object') throw new Error('plan output is not an object')
154
154
  const summary = typeof obj.summary === 'string' && obj.summary.trim() ? obj.summary.trim() : ''
155
155
  if (!Array.isArray(obj.tasks) || obj.tasks.length === 0) throw new Error('plan has no tasks')
156
- const tasks = obj.tasks.map((t) => makeTask({
157
- key: String(t.key),
158
- title: String(t.title ?? t.key),
159
- scope: String(t.scope ?? ''),
160
- acceptance: String(t.acceptance ?? ''),
161
- deps: Array.isArray(t.deps) ? t.deps.map(String) : [],
162
- sliceType: Object.values(SLICE_TYPE).includes(t.sliceType) ? t.sliceType : SLICE_TYPE.feature,
163
- }))
156
+ const tasks = obj.tasks.map((t) => {
157
+ // Accept `id` as an alias for `key` and `description` for `scope` — conductors emit
158
+ // either naming. A task with NO key is rejected (was silently String(undefined) →
159
+ // task_key "undefined", which broke worktree/dispatch); throwing makes it re-emit.
160
+ const key = t.key ?? t.id
161
+ if (key == null || !String(key).trim()) throw new Error('a task is missing its "key"')
162
+ const sliceType = t.sliceType ?? t.slice_type
163
+ return makeTask({
164
+ key: String(key),
165
+ title: String(t.title ?? key),
166
+ scope: String(t.scope ?? t.description ?? ''),
167
+ acceptance: String(t.acceptance ?? ''),
168
+ deps: Array.isArray(t.deps) ? t.deps.map(String) : [],
169
+ sliceType: Object.values(SLICE_TYPE).includes(sliceType) ? sliceType : SLICE_TYPE.feature,
170
+ })
171
+ })
164
172
  validateDag(tasks)
165
173
  return { summary, tasks }
166
174
  }
package/flow-worktree.mjs CHANGED
@@ -25,14 +25,27 @@ export function worktreeSpec ({ flowId, taskKey, root = ROOT }) {
25
25
  return { branch, dir, wtRoot }
26
26
  }
27
27
 
28
- // Create a worktree for a task-lane off `base` (default origin/main). Idempotent: if
29
- // the dir already holds a worktree, reuse it. Returns { dir, branch, created }.
30
- export function createFlowWorktree ({ flowId, taskKey, base = 'origin/main', root = ROOT, git = runGit }) {
28
+ // Create a worktree for a task-lane off `base`. Idempotent: if the dir already holds a
29
+ // worktree, reuse it. Returns { dir, branch, created }.
30
+ //
31
+ // FL-M6/dispatch — the base must RESOLVE in this repo. The old hard default `origin/main`
32
+ // threw on any repo without an `origin` remote or a non-`main` default branch (e.g. a
33
+ // fresh `git init` checkout), and the dispatch loop swallowed the throw → NO lane ever
34
+ // spawned. Resolve the first ref that actually exists: caller's base → origin/main →
35
+ // origin/HEAD → main → master → HEAD.
36
+ export function createFlowWorktree ({ flowId, taskKey, base = null, root = ROOT, git = runGit }) {
31
37
  const { branch, dir, wtRoot } = worktreeSpec({ flowId, taskKey, root })
32
38
  if (fs.existsSync(path.join(dir, '.git'))) return { dir, branch, created: false }
33
39
  fs.mkdirSync(wtRoot, { recursive: true })
40
+ let ref = base
41
+ if (!ref) {
42
+ for (const cand of ['origin/main', 'origin/HEAD', 'main', 'master', 'HEAD']) {
43
+ try { git(['rev-parse', '--verify', '--quiet', cand], root); ref = cand; break } catch { /* try next candidate */ }
44
+ }
45
+ ref = ref || 'HEAD'
46
+ }
34
47
  try {
35
- git(['worktree', 'add', '-b', branch, dir, base], root)
48
+ git(['worktree', 'add', '-b', branch, dir, ref], root)
36
49
  } catch {
37
50
  // Branch already exists (a prior dispatch of this task) — check it out instead.
38
51
  git(['worktree', 'add', dir, branch], root)
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "thinkpool-pair",
3
- "version": "0.7.107",
3
+ "version": "0.7.108",
4
4
  "description": "Share a local coding-agent CLI (Claude Code, Codex, Gemini, Aider, …) into a ThinkPool Code room, live.",
5
5
  "type": "module",
6
6
  "bin": {