thinkpool-pair 0.7.221 → 0.7.223

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bridge.mjs CHANGED
@@ -55,9 +55,10 @@ import { canonicalRoomFilePath, waitForNativeImages } from './codex-images.mjs'
55
55
  import { createManagedLaneWorktree, removeManagedLaneWorktree } from './lane-worktree.mjs'
56
56
  import { commandOnPath } from './agent-detect.mjs'
57
57
 
58
- const STRUCTURED_MODES = new Set(['default', 'acceptEdits', 'plan', 'bypassPermissions'])
59
- import { FLOW_CONDUCTOR_PROMPT, FLOW_LANE_PROMPT, buildConductorEnv, assembleCrossWaveContext, buildLanePrompt } from './flow-conductor.mjs'
60
- import { normalizePlanOutput, laneModelFor } from './flow-task-graph.mjs' // FL-B1 — validate the conductor's submit_flow_plan task-graph; laneModelFor — per-slice model tier (2026-07-03-flow-lane-model-tiers)
58
+ const STRUCTURED_MODES = new Set(['default', 'acceptEdits', 'plan', 'review', 'bypassPermissions'])
59
+ import { FLOW_CONDUCTOR_PROMPT, FLOW_LANE_PROMPT, FLOW_CODEX_CONDUCTOR_PROMPT, FLOW_CODEX_LANE_PROMPT, buildConductorEnv, assembleCrossWaveContext, buildLanePrompt } from './flow-conductor.mjs'
60
+ import { legacyBuilderCompletionAllowed, normalizePlanOutput, validatePlanForRuntime, validReviewTargetShape } from './flow-task-graph.mjs'
61
+ import { normalizeFlowRuntime, flowLaneModelFor, flowConductorModelFor, spawnedLaneModelFor, modelCatalogValues, resolveCodexModel, assertRuntimeModelCompatible } from './flow-models.mjs'
61
62
  // S1 (context-offload) — durable digest store. mark_flow_done digests a closed slice in;
62
63
  // the dispatch loop reads the bounded cross-wave context back out (via assembleCrossWaveContext,
63
64
  // which enforces the CEILING). Consume the store — the internals live in flow-context-store.mjs.
@@ -65,7 +66,7 @@ import { writeLaneArtifact, digestSlice, appendDigest, resumeLane } from './flow
65
66
  import { createFlowWorktree, worktreeSpec } from './flow-worktree.mjs'
66
67
  import { startPreview, stopAllPreviews, previews } from './flow-preview.mjs'
67
68
  import { ViewportManager, createViewportTools, sharedViewportBrowser } from './viewport.mjs'
68
- import { designPrompt, resolveDesignSource, resolveManifestDesignSource, restoreDesignSources, syncRealtimeAuth, validateDesignRequest } from './design-edit.mjs'
69
+ import { designPrompt, materializeDesignAsset, resolveDesignSource, resolveManifestDesignSource, restoreDesignSources, syncRealtimeAuth, validateDesignRequest } from './design-edit.mjs'
69
70
  // FL-M9 — per-lane preview servers leak (one per done lane, never stopped until shutdown).
70
71
  // Lane previews are keyed `lane:<flowId>:<laneId>`; stop a whole flow's set when it assembles
71
72
  // (the assembled preview supersedes them) or when a lane is reverted.
@@ -75,7 +76,7 @@ function stopFlowPreviews (flowId, laneId = null) {
75
76
  if (laneId ? key === `lane:${flowId}:${laneId}` : key.startsWith(`lane:${flowId}:`)) { try { h.stop() } catch { /* noop */ } }
76
77
  }
77
78
  }
78
- import { FLOW_REVIEWER_PROMPT, revertLane, parseReviewVerdict, reviewVerdictToReflection } from './flow-review.mjs'
79
+ import { FLOW_REVIEWER_PROMPT, FLOW_CODEX_REVIEWER_PROMPT, revertLane, parseReviewVerdict, reviewVerdictToReflection } from './flow-review.mjs'
79
80
  import { reviewGateDecision } from './flow-review-gate.mjs'
80
81
  import { pairAdjudicationPrompt, reviewReflectionDecision, REVIEW_DEFAULTS } from './flow-review-reflect.mjs'
81
82
  import { mergeWorktrees, inlineSingleHtml, initRepo } from './flow-assembly.mjs'
@@ -89,6 +90,7 @@ import { canDispatch, FLOW_LIMITS, makeBudget, recordSpend, killSwitchEnv } from
89
90
  // Spec: docs/specs/2026-06-30-flow-build-s4-clean-redispatch.md
90
91
  import { prepareRedispatch, redispatchKey } from './flow-redispatch.mjs'
91
92
  import { sanitizeSession } from './transcript-sanitize.mjs'
93
+ import { executeFlowRevert } from './flow-host-revert.mjs'
92
94
  // ACCEPTED LIMITATION — this registry is in-memory only. A bridge restart in the window between
93
95
  // a flow-revert kill and the next dispatch wave loses the pending resume record, so the task
94
96
  // re-dispatches COLD (fresh lane, no resume). That's safe: the healed transcript survives on
@@ -101,7 +103,7 @@ const flowRedispatch = new Map()
101
103
  // broadcasts; without persistent state the cap can never bite.
102
104
  const flowBudgets = new Map()
103
105
  import { formatPeek, PEEK, siblingsOf, resolveSibling, crossPostDecision, CROSSPOST, spawnDecision, SPAWN, CROSSROOM, formatPairRoster, crossRoomPostDecision, formatRoomNow, formatClosableHint, laneStatusOf } from './cross-terminal.mjs'
104
- import { armInterruptedResume, sendInterruptedContinue, supersedeInterruptedResume } from './interrupted-resume.mjs'
106
+ import { dispatchPendingRecap, recoverInterruptedTurn, recoverMissingResumeOnce, armInterruptedResume, sendInterruptedContinue, supersedeInterruptedResume } from './interrupted-resume.mjs'
105
107
  import { turnInFlight } from './update-gate.mjs'
106
108
  import { saveSession, flushSession, deleteSession, loadAll, canResume, loadPtyId, savePtyId, loadNames, saveNames, appendDurableEvents, seedDurableEvents, readDurablePage, readDurableOldestSeq, hasDurableArchive, servesHistoryPage } from './session-store.mjs'
107
109
  import { stampEvent, makeSeqCounter, maxSeq, seqable, capReplayEvents, chunkReplayEvents, boundEventForBroadcast, inlineImageBlocks, usageReportLine, codexUsageReportLine, buildRecapFromLog, RECAP_CAP, trimmedBeforeSeq, firstSeq } from './event-id.mjs'
@@ -1282,7 +1284,11 @@ function pumpDesign(term) {
1282
1284
  designActive.set(term, active)
1283
1285
  designStatus({ previewId: next.record.previewId, requestId: next.request.cid, state: 'applying' })
1284
1286
  // Safe transcript summary: never include host paths or the full locator packet.
1285
- const visible = { kind: 'you', text: `Design edit · ${next.request.target.name || next.request.target.tag}: ${next.request.mode === 'text' ? 'replace text' : next.request.intent}`, cid: next.request.cid, by: next.by }
1287
+ const action = next.request.mode === 'text' ? 'replace text'
1288
+ : next.request.mode === 'move' ? 'move element'
1289
+ : next.request.mode === 'image' ? 'replace image'
1290
+ : next.request.intent
1291
+ const visible = { kind: 'you', text: `Design edit · ${next.request.target.name || next.request.target.tag}: ${action}`, cid: next.request.cid, by: next.by }
1286
1292
  stampEvent(visible); pushLog(lane, visible); bcast('code-event', { term, evt: visible })
1287
1293
  try { lane.session.sendTurn(designPrompt({ record: next.record, request: next.request, by: next.by, restore: !!next.restoreRecord, priorRecord: next.restoreRecord })) }
1288
1294
  catch { finishDesign(term, 'failed', { message: 'The producing lane could not start the edit.' }) }
@@ -1646,7 +1652,7 @@ function worktreeSnapshot(cwd) {
1646
1652
  // relay STRUCTURED events. onEvent → broadcast `code-event` + print locally +
1647
1653
  // persist to the host file; tool calls round-trip through the perm card; the
1648
1654
  // rolling log replays to joiners and survives bridge restarts (session-store).
1649
- function openStructured({ id, runtime = 'claude', model, effort, resume, log, commands, mode, spawnedBy, sideParent, sideTask, pendingSideContexts, rolePrompt, flowSessionId, flowTaskKey, cwd, managedWorktree, reviewSliceRoots, openedAt, defer, provider, carryRecap, lastUsage }) {
1655
+ function openStructured({ id, runtime = 'claude', model, effort, resume, log, commands, mode, spawnedBy, sideParent, sideTask, pendingSideContexts, rolePrompt, flowSessionId, flowTaskKey, flowRole, flowReviewTarget, flowReviewTargets, flowReviewRound, dispatchBaseSha, revertTarget, cwd, managedWorktree, reviewSliceRoots, openedAt, defer, provider, carryRecap, lastUsage }) {
1650
1656
  if (sessions.has(id)) return
1651
1657
  runtime = runtime === 'codex' ? 'codex' : 'claude'
1652
1658
  // No explicit mode → a sensible default per runtime (see defaultModeForRuntime):
@@ -1656,6 +1662,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
1656
1662
  // entry.mode), a flow lane, and spawn_terminal (inherits the parent's mode +
1657
1663
  // its own bypass-escalation gate) all pass an explicit mode and skip this.
1658
1664
  mode = STRUCTURED_MODES.has(mode) ? mode : defaultModeForRuntime(runtime)
1665
+ assertRuntimeModelCompatible({ runtime, provider, model })
1659
1666
  // The lane's EFFECTIVE model — computed ONCE and used for BOTH the truthful display
1660
1667
  // label (entry.model) and the SDK's `model` option below. These used to be computed
1661
1668
  // separately: the label resolved the provider's configured model while the SDK got the
@@ -1663,14 +1670,15 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
1663
1670
  // lane opened while focused on Fable therefore showed `glm-4.6` and sent
1664
1671
  // `claude-fable-5` → 400 [1211][Unknown Model]. One value, one truth.
1665
1672
  // undefined → omit the SDK option entirely and let the provider env/endpoint default.
1673
+ const codexModels = runtime === 'codex' ? readCodexModels() : []
1666
1674
  const laneModel = runtime === 'codex'
1667
- ? (model || readCodexDefaultModel() || undefined)
1675
+ ? resolveCodexModel(model || readCodexDefaultModel() || undefined, codexModels)
1668
1676
  : effectiveLaneModel({ provider, model, configuredModel: resolveProviderEnv(provider)?.ANTHROPIC_MODEL })
1669
1677
  // spawnedBy: set when this lane was Dispatched (spawn_terminal). Restored from the
1670
1678
  // session store so the Ensemble flag survives a bridge restart (else a respin
1671
1679
  // stripped it and the lane reverted to a plain tab — the t6 "no chip" bug).
1672
1680
  effort = new Set(['low', 'medium', 'high', 'xhigh', 'max']).has(effort) ? effort : 'high'
1673
- const entry = { cmd: runtime, runtime, kind: 'structured', log: Array.isArray(log) ? log.slice(-STRUCTURED_LOG_MAX) : [], pending: new Map(), session: null, recovered: false, commands: Array.isArray(commands) ? commands : undefined, mode, effort, models: runtime === 'codex' ? readCodexModels() : undefined,
1681
+ const entry = { cmd: runtime, runtime, kind: 'structured', log: Array.isArray(log) ? log.slice(-STRUCTURED_LOG_MAX) : [], pending: new Map(), session: null, recovered: false, commands: Array.isArray(commands) ? commands : undefined, mode, effort, models: runtime === 'codex' ? codexModels : undefined,
1674
1682
  // model: truthful active-model label — now the SAME `laneModel` the SDK is given, so the
1675
1683
  // chip cannot disagree with the wire. When this lane runs on a custom (non-anthropic)
1676
1684
  // registered provider the SDK id is impersonated (see the onEvent guard below), so
@@ -1684,10 +1692,18 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
1684
1692
  ? (laneModel || null)
1685
1693
  : (laneModel || providerNameMap()[provider] || provider),
1686
1694
  provider: provider || null, spawnedBy: spawnedBy || undefined, sideParent: sideParent || undefined, sideTask: sideTask || undefined, pendingSideContexts: Array.isArray(pendingSideContexts) ? pendingSideContexts.filter(Boolean).slice(-4) : [], flowSessionId: flowSessionId || null, flowTaskKey: flowTaskKey || null, cwd: cwd || null, managedWorktree: managedWorktree || null,
1695
+ flowRole: flowRole || (flowSessionId ? (flowTaskKey ? ((flowReviewTargets?.length || reviewSliceRoots?.length) ? 'reviewer' : 'builder') : 'conductor') : null),
1696
+ flowReviewTarget: flowReviewTarget || null,
1697
+ flowReviewTargets: Array.isArray(flowReviewTargets) && flowReviewTargets.length ? flowReviewTargets.filter(Boolean) : (flowReviewTarget ? [flowReviewTarget] : []),
1698
+ flowReviewRound: Number.isInteger(flowReviewRound) && flowReviewRound >= 0 ? flowReviewRound : 0,
1699
+ dispatchBaseSha: dispatchBaseSha || null,
1700
+ revertTarget: revertTarget || null,
1701
+ cwd: cwd || null, managedWorktree: managedWorktree || null,
1687
1702
  // Stable creation order — persisted so a bridge restart restores tabs in the SAME
1688
1703
  // order (not readdir/filesystem order). Legacy recs (no openedAt) derive it from the
1689
1704
  // first transcript event ts, so even the first post-fix restart is ordered right.
1690
1705
  openedAt: openedAt || (Array.isArray(log) ? (log.find((e) => e?.ts)?.ts || 0) : 0) || Date.now() }
1706
+ entry.interruptedRecap = restoredTurnOpen(entry.log) ? buildRecapFromLog(entry.log, RECAP_CAP) : null
1691
1707
  // Slice 3 — a permission card left unanswered past the grace window pushes
1692
1708
  // "<lane> — needs you: <what>"; answering it anywhere retracts the banner
1693
1709
  // everywhere. Worker/flow lanes are excluded at arm() time (isUserFacingLane).
@@ -1715,11 +1731,15 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
1715
1731
  // MCP tool AND the FLOW_DONE sentinel-Write intercept: like the conductor's submit, the
1716
1732
  // MCP tool is DEFERRED (needs ToolSearch, which hangs), so lanes signal done by Writing a
1717
1733
  // FLOW_DONE file — a DIRECT tool — which the PreToolUse hook routes here. Returns a message.
1718
- const markFlowDone = async () => {
1734
+ const markFlowDone = async ({ reviewPass = false } = {}) => {
1719
1735
  if (!entry.flowSessionId || !entry.flowTaskKey) return 'Not a Flow lane — nothing to mark done.'
1736
+ if (entry.flowRole !== 'builder' && !(reviewPass && entry.flowRole === 'reviewer')) return 'This Flow role cannot mark a builder slice done.'
1720
1737
  if (entry.flowDone) return 'This slice is already recorded as done.'
1721
1738
  let commitSha = null
1722
1739
  try { commitSha = execFileSync('git', ['-C', entry.cwd || process.cwd(), 'rev-parse', 'HEAD'], { encoding: 'utf8' }).trim() } catch { /* no commits yet */ }
1740
+ if (!reviewPass && entry.dispatchBaseSha && commitSha === entry.dispatchBaseSha) {
1741
+ return `Slice "${entry.flowTaskKey}" is not done: HEAD is still the dispatch base (${commitSha.slice(0, 8)}). Commit the verified implementation first.`
1742
+ }
1723
1743
  let previewUrl = null
1724
1744
  try { const pv = await startPreview({ dir: entry.cwd || process.cwd(), id: `lane:${entry.flowSessionId}:${id}` }); previewUrl = pv.url } catch { /* preview best-effort */ }
1725
1745
  // S1 (context-offload) — digest THIS closed slice into the durable store so the next
@@ -1749,7 +1769,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
1749
1769
  } catch (e) {
1750
1770
  process.stderr.write(`\n ${A.dim}◆ flow digest skip ${entry.flowTaskKey} — ${e?.message || e}${A.rst}\n`)
1751
1771
  }
1752
- bcast('flow-task-done', { term: id, flowId: entry.flowSessionId, taskKey: entry.flowTaskKey, laneId: id, commitSha, previewUrl }, flowChannel)
1772
+ bcast('flow-task-done', { term: id, flowId: entry.flowSessionId, taskKey: entry.flowTaskKey, laneId: id, commitSha, previewUrl, ...(reviewPass ? { reviewAction: 'pass' } : {}) }, flowChannel)
1753
1773
  process.stderr.write(`\n ${A.cyan}◆ flow slice done — ${entry.flowTaskKey} (${commitSha ? commitSha.slice(0, 8) : 'no commit'})${previewUrl ? ` · preview ${previewUrl}` : ''}${A.rst}\n`)
1754
1774
  entry.flowDone = true // FL-B3 — retire immediately so it drops from the ≤8 lane cap
1755
1775
  setTimeout(() => { try { endStructured(id) } catch { /* already gone */ } }, 2500)
@@ -1761,14 +1781,22 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
1761
1781
  // never reverted). On FAIL → broadcast flow-revert for the reviewed slice (the client flips
1762
1782
  // it to pending → the next wave rebuilds it). Either way the review lane itself is done.
1763
1783
  const onReviewVerdict = async (raw) => {
1764
- if (!entry.flowSessionId || !entry.flowTaskKey) return { ok: false, message: 'Not a Flow review lane.' }
1784
+ if (!entry.flowSessionId || !entry.flowTaskKey || entry.flowRole !== 'reviewer') return { ok: false, message: 'Not a Flow review lane.' }
1765
1785
  let v, target = null
1766
1786
  try {
1767
1787
  v = parseReviewVerdict(raw)
1768
1788
  const o = typeof raw === 'string' ? JSON.parse(String(raw).replace(/```(?:json)?|```/g, '').trim()) : raw
1769
- target = (o && (o.taskKey || o.target)) || (entry.flowReviewTarget || null)
1789
+ const requested = o && (o.taskKey || o.target)
1790
+ const allowed = entry.flowReviewTargets
1791
+ if (!allowed.length) throw new Error('review lane has no authorized target')
1792
+ if (entry.runtime === 'codex' && allowed.length !== 1) throw new Error('Codex review lane must have exactly one authorized target')
1793
+ if (requested && !allowed.includes(requested)) throw new Error(`taskKey ${JSON.stringify(requested)} is outside this review lane`)
1794
+ target = requested || (allowed.length === 1 ? allowed[0] : null)
1795
+ if (!target) throw new Error('taskKey is required when reviewing more than one slice')
1770
1796
  } catch (e) {
1771
- return { ok: false, message: `Review verdict REJECTED: ${e?.message || e}. Re-Write FLOW_REVIEW.json with {"pass":<boolean>,"reasons":["<specific finding>"],"taskKey":"<the slice you reviewed>"}.` }
1797
+ return { ok: false, message: entry.runtime === 'codex'
1798
+ ? `Review verdict REJECTED: ${e?.message || e}. Re-call submit_flow_review with a valid authorized taskKey.`
1799
+ : `Review verdict REJECTED: ${e?.message || e}. Re-Write FLOW_REVIEW.json with {"pass":<boolean>,"reasons":["<specific finding>"],"taskKey":"<the slice you reviewed>"}.` }
1772
1800
  }
1773
1801
  // E1 A1/A2 — the BOUNDED reviewer loop, live side. Each FLOW_REVIEW.json write is ONE
1774
1802
  // hunt round; the governor decides continue-vs-stop from the round count + the lane's
@@ -1779,6 +1807,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
1779
1807
  // for live-room verification (bridge is inert until published); the decision logic is
1780
1808
  // unit-proven in flow-review.loop.test.mjs + flow-review-reflect.test.mjs.
1781
1809
  const round = (entry.flowReviewRound = (entry.flowReviewRound || 0) + 1)
1810
+ entry.flush?.()
1782
1811
  const b = flowBudgets.get(entry.flowSessionId) || null
1783
1812
  const decision = reviewReflectionDecision({
1784
1813
  round,
@@ -1794,7 +1823,9 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
1794
1823
  const next = round + 1
1795
1824
  try {
1796
1825
  entry.session?.sendTurn(
1797
- `[Flow review — round ${next}/${REVIEW_DEFAULTS.maxRounds}] The happy path held, but you have NOT reported your checks exhausted and you are under both the round and budget ceilings. Dig one more round: hunt the edge cases, the reload, the second click, concurrent use, the error path — the places the builder didn't. Then re-Write FLOW_REVIEW.json — pass:false with a specific reason if you break it, or pass:true AND exhausted:true if you genuinely have nothing left to check.`,
1826
+ entry.runtime === 'codex'
1827
+ ? `[Flow review — round ${next}/${REVIEW_DEFAULTS.maxRounds}] The happy path held, but you have NOT reported your checks exhausted. Dig one more round, then call submit_flow_review again — pass:false with specific evidence if you break it, or pass:true AND exhausted:true only when nothing remains to check.`
1828
+ : `[Flow review — round ${next}/${REVIEW_DEFAULTS.maxRounds}] The happy path held, but you have NOT reported your checks exhausted and you are under both the round and budget ceilings. Dig one more round: hunt the edge cases, the reload, the second click, concurrent use, the error path — the places the builder didn't. Then re-Write FLOW_REVIEW.json — pass:false with a specific reason if you break it, or pass:true AND exhausted:true if you genuinely have nothing left to check.`,
1798
1829
  )
1799
1830
  } catch { /* lane may have closed mid-verdict */ }
1800
1831
  process.stderr.write(`\n ${A.dim}◆ review round ${round} inconclusive — digging again (${next}/${REVIEW_DEFAULTS.maxRounds}) on ${target || entry.flowTaskKey}${A.rst}\n`)
@@ -1804,11 +1835,24 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
1804
1835
  // Terminal outcome. reject → revert the reviewed slice; surface → hand the inconclusive
1805
1836
  // result to the pair WITHOUT reverting (no failure was reproduced); pass → accept.
1806
1837
  if (decision.action === 'reject' && target) {
1807
- bcast('flow-revert', { term: id, flowId: entry.flowSessionId, taskKey: target }, flowChannel)
1838
+ const reverted = await handleFlowRevert({ term: id, flowId: entry.flowSessionId, taskKey: target, reviewTaskKey: entry.flowTaskKey })
1839
+ if (!reverted.ok) persistAgentEvent({ kind: 'needs-input', term: id, termName: termNames[id] || null, summary: `Flow review could not revert ${target}: ${reverted.error || reverted.ignored || 'unknown error'}` })
1808
1840
  process.stderr.write(`\n ${A.yel}◆ review REJECT (round ${round}) — reverting ${target}: ${v.reasons.join('; ').slice(0, 120)}${A.rst}\n`)
1809
1841
  } else {
1810
1842
  process.stderr.write(`\n ${A.cyan}◆ review ${decision.action.toUpperCase()} (round ${round}) — ${target || entry.flowTaskKey}${A.rst}\n`)
1811
1843
  }
1844
+ // Durable room-visible verdict. `persistAgentEvent` below is intentionally
1845
+ // filtered from the transcript (push/unread transport only), while this control
1846
+ // row is archived with the reviewer lane and broadcast to both room members.
1847
+ // In particular, a held review must show its exact findings instead of leaving
1848
+ // the red task state unexplained after the ephemeral Flow broadcast is gone.
1849
+ const visibleReview = {
1850
+ kind: 'control',
1851
+ text: `Flow review ${decision.action.toUpperCase()} — ${target || entry.flowTaskKey}: ${v.reasons.join('; ') || decision.reason}`,
1852
+ }
1853
+ pushLog(entry, visibleReview)
1854
+ bcast('code-event', { term: id, evt: visibleReview })
1855
+ entry.flush?.()
1812
1856
  // P3 (pair co-adjudication) — surface the TERMINAL verdict into the ROOM as a
1813
1857
  // challengeable prompt, ALONGSIDE the gate action above. A solo Bugbot's verdict is
1814
1858
  // final; ours is a prompt for the two humans + Pool to argue. Additive: an extra
@@ -1828,13 +1872,31 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
1828
1872
  findings: v.reasons,
1829
1873
  }),
1830
1874
  }, flowChannel)
1831
- const doneMsg = await markFlowDone()
1875
+ let doneMsg = ''
1876
+ if (decision.action === 'pass') {
1877
+ doneMsg = await markFlowDone({ reviewPass: true })
1878
+ } else {
1879
+ if (decision.action === 'surface') {
1880
+ bcast('flow-review-held', { term: id, flowId: entry.flowSessionId, reviewTaskKey: entry.flowTaskKey, taskKey: target, reasons: v.reasons }, flowChannel)
1881
+ persistAgentEvent({ kind: 'needs-input', term: id, termName: termNames[id] || null, summary: `Flow review held ${target || entry.flowTaskKey}: ${v.reasons.join('; ') || decision.reason}` })
1882
+ }
1883
+ // A reject must rebuild the target and then run a fresh reviewer. Retire that
1884
+ // reviewer now. A surfaced inconclusive review is different: keep its lane
1885
+ // alive with the durable control row above so either room member can open the
1886
+ // held review, read the exact findings, and continue/adjudicate after reload.
1887
+ // Neither outcome emits task-done or permits assembly.
1888
+ if (decision.action === 'reject') {
1889
+ entry.flowDone = true
1890
+ entry.flush?.()
1891
+ setTimeout(() => { try { endStructured(id) } catch { /* already gone */ } }, 2500)
1892
+ }
1893
+ }
1832
1894
  const label = decision.action === 'reject'
1833
1895
  ? `REJECT (round ${round}) — reverting ${target || '(no target)'} (${v.reasons.join('; ').slice(0, 140)})`
1834
1896
  : decision.action === 'surface'
1835
1897
  ? `SURFACED to the pair after ${round} round${round === 1 ? '' : 's'} — ${decision.reason}`
1836
1898
  : `PASS (round ${round})`
1837
- return { ok: true, message: `Review verdict recorded: ${label}. ${doneMsg}` }
1899
+ return { ok: true, message: `Review verdict recorded: ${label}.${doneMsg ? ` ${doneMsg}` : ''}` }
1838
1900
  }
1839
1901
  // Identity for the durable archive — pushLog appends every new transcript event to
1840
1902
  // <room>/<id>.events.jsonl keyed off these. Seed the archive once from the restored
@@ -1880,7 +1942,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
1880
1942
  // restart. Without this, sessionData omitted it → on restart the resumed session
1881
1943
  // re-launched on the host default (Opus) regardless of the last switch, and the
1882
1944
  // switch looked like it "never changed the model" (Max 2026-07-02). Restored below.
1883
- const sessionData = () => ({ sessionId: entry.session?.sessionId || null, runtime: entry.runtime, log: entry.log, commands: entry.commands, mode: entry.mode, effort: entry.effort, model: entry.model || null, provider: entry.provider || null, spawnedBy: entry.spawnedBy, sideParent: entry.sideParent, sideTask: entry.sideTask, pendingSideContexts: entry.pendingSideContexts, flowSessionId: entry.flowSessionId, flowTaskKey: entry.flowTaskKey, cwd: entry.cwd, managedWorktree: entry.managedWorktree, rolePrompt: entry.rolePrompt, flowReviewTarget: entry.flowReviewTarget || null, revertTarget: entry.revertTarget || null, reviewSliceRoots: entry.reviewSliceRoots || [], openedAt: entry.openedAt || null, lastUsage: entry.lastUsage || null, carryRecap: entry.pendingRecap || null })
1945
+ const sessionData = () => ({ sessionId: entry.session?.sessionId || null, runtime: entry.runtime, log: entry.log, commands: entry.commands, mode: entry.mode, effort: entry.effort, model: entry.model || null, provider: entry.provider || null, spawnedBy: entry.spawnedBy, sideParent: entry.sideParent, sideTask: entry.sideTask, pendingSideContexts: entry.pendingSideContexts, flowSessionId: entry.flowSessionId, flowTaskKey: entry.flowTaskKey, flowRole: entry.flowRole || null, flowReviewTargets: entry.flowReviewTargets || [], flowReviewRound: entry.flowReviewRound || 0, dispatchBaseSha: entry.dispatchBaseSha || null, cwd: entry.cwd, managedWorktree: entry.managedWorktree, rolePrompt: entry.rolePrompt, flowReviewTarget: entry.flowReviewTarget || null, revertTarget: entry.revertTarget || null, reviewSliceRoots: entry.reviewSliceRoots || [], openedAt: entry.openedAt || null, lastUsage: entry.lastUsage || null, carryRecap: entry.pendingRecap || null })
1884
1946
  const persist = () => saveSession(room, id, sessionData())
1885
1947
  // Synchronous flush of this session's record. Used on open (so a brand-new session
1886
1948
  // has a file under its id BEFORE its first event — surviving a restart inside the
@@ -1962,7 +2024,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
1962
2024
  // OWN receipt card — no text reaches the other agent without a person at BOTH
1963
2025
  // ends. crossRoomPostDecision caps it at one per turn and blocks a room reached
1964
2026
  // via cross-room post from posting onward. Kill-switch: TP_PAIRBUS_OFF.
1965
- tool(
2027
+ ...(!entry.flowRole ? [tool(
1966
2028
  'post_to_session',
1967
2029
  'Hand a task or message to an agent in ANOTHER of your ThinkPool Code sessions — your own room on this machine, or your partner\'s room reachable through the pair (a room code from list_sessions). Pass `session` (the room code), `text` (what to send), and optionally `terminal` (which agent lane in that room). A person in YOUR room approves sending AND a person in the TARGET room approves receiving before it is delivered. An agent reached this way cannot post onward to a third room. Read the target with read_session first, and use this sparingly — only when the people clearly want the rooms to coordinate.',
1968
2030
  {
@@ -1987,13 +2049,13 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
1987
2049
  if (res?.error) return okText(res.error)
1988
2050
  return okText(res?.ok ? `Delivered to room ${target}${res.ref ? ` (lane ${res.ref})` : ''} — it will respond in its own room; check back with read_session.` : `Room ${target} did not accept the message.`)
1989
2051
  },
1990
- ),
2052
+ )] : []),
1991
2053
  // Tier C — directed cross-lane action. WRITES into a sibling AGENT lane.
1992
2054
  // The PreToolUse gate (claude-session.mjs) already enforced: precheck passed
1993
2055
  // + a human approved the card. This handler routes only; it re-asserts the
1994
2056
  // bounds defensively. Targets AGENT terminals only (not human shells —
1995
2057
  // injecting keystrokes into someone's shell is out of scope for v1).
1996
- tool(
2058
+ ...(!entry.flowRole ? [tool(
1997
2059
  'post_to_terminal',
1998
2060
  'Send a message or task to ANOTHER AGENT terminal in this ThinkPool Code room. Use `terminal` (a ref/id/command from read_terminal) and `text` (what to send). A person in the room must approve before it is delivered, and an agent reached via a cross-post cannot post onward. Targets agent terminals only, not plain shells. Use read_terminal first to understand the sibling.',
1999
2061
  {
@@ -2030,7 +2092,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
2030
2092
  try { te.session.sendTurn(msg) } catch { return okText(`Could not deliver to terminal ${target.ref} — it may have just closed.`) }
2031
2093
  return okText(`Delivered to terminal ${target.ref} (${target.cmd}). It will respond in its own lane; check back with read_terminal.`)
2032
2094
  },
2033
- ),
2095
+ )] : []),
2034
2096
  // Tier C+ — SPAWN a fresh agent lane the caller OWNS. The motivating bug:
2035
2097
  // an agent fanned a research task into SIBLINGS that were already busy (a
2036
2098
  // release lane, a Q&A lane) because it had no way to make its own lanes.
@@ -2041,13 +2103,14 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
2041
2103
  // cap, TP_SPAWN_OFF kill-switch). A person's lane (hop 0) may spawn conductors
2042
2104
  // (hop 1) which may spawn workers (hop 2); hop 2 can never spawn onward — the
2043
2105
  // fork-bomb breaker, now one level deeper (2026-07-10 cascade-spawn-depth spec).
2044
- tool(
2106
+ ...(!entry.flowRole ? [tool(
2045
2107
  'spawn_terminal',
2046
- 'Open a NEW agent terminal (Claude or Codex) in this ThinkPool Code room — YOUR lane to use, not a sibling that is already working. The runtime defaults to inheriting this lane, or pass `runtime` to dispatch across runtimes. Choose `model` explicitly from advertised metadata when practical: for Codex use Luna/Mini for mechanical work, Terra for ordinary feature/fix work, and Sol only for repeated-failure, architecture/conductor, or adversarial final-review work; never inherit Sol by default, and omit `model` for runtime fallback when the named tier is unavailable. Give the lane an initial `task`, collect its result later with read_terminal, then ALWAYS close_terminal it — a finished lane left open holds a slot against the room cap and blocks the next fan-out. By default the new lane INHERITS your current permission mode. ALWAYS prefer spawning a fresh lane over post_to_terminal into one that is already busy. Bounded by cascade depth, the room terminal cap, and per-turn limits.',
2108
+ 'Open a NEW agent terminal (Claude or Codex) in this ThinkPool Code room — YOUR lane to use, not a sibling that is already working. The runtime defaults to inheriting this lane, or pass `runtime` to dispatch across runtimes. Choose `model` explicitly from advertised metadata when practical: for Codex use Luna/Mini for mechanical work, Terra for ordinary feature/fix work, and Sol only for repeated-failure, architecture/conductor, or adversarial final-review work; never inherit Sol by default, and omit `model` for runtime fallback when the named tier is unavailable. For cascades, pass sliceType=scaffold for mechanical work, feature/fix for builders, and review for adversarial verification (review never routes to the cheap tier); an explicit model overrides slice tiering. Omit sliceType to preserve ordinary model inheritance/default behavior. Give the lane an initial `task`, collect its result with read_terminal, then ALWAYS close_terminal it — a finished lane left open holds a slot against the room cap and blocks the next fan-out. By default the new lane INHERITS your current permission mode. ALWAYS prefer spawning a fresh lane over post_to_terminal into one that is already busy. Bounded by cascade depth, the room terminal cap, and per-turn limits.',
2047
2109
  {
2048
2110
  name: z.string().max(80).optional().describe('a short label for the new lane so the room (and you) can find it, e.g. "Research: topologies"'),
2049
2111
  task: z.string().optional().describe('an initial task to hand the new lane immediately; omit to open it idle'),
2050
2112
  model: z.string().optional().describe('optional model for the chosen runtime, e.g. opus / sonnet / gpt-5.6-sol'),
2113
+ sliceType: z.enum(['scaffold', 'feature', 'fix', 'review']).optional().describe('optional cascade slice tier; omitted preserves normal inheritance/default, explicit model wins'),
2051
2114
  runtime: z.enum(['claude', 'codex']).optional().describe('agent runtime for the new lane; defaults to inheriting this lane'),
2052
2115
  provider: z.string().optional().describe('optional registered LLM provider id to run this lane on (from the account\'s provider registry); omit for the default Claude/Anthropic path'),
2053
2116
  mode: z.enum(['default', 'acceptEdits', 'bypassPermissions', 'plan']).optional().describe('permission mode for the new lane; defaults to inheriting YOUR current mode. Raising a lane to bypassPermissions from a non-bypass lane asks the room to confirm once.'),
@@ -2056,6 +2119,16 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
2056
2119
  const okText = (t) => ({ content: [{ type: 'text', text: t }] })
2057
2120
  const childRuntime = args?.runtime || entry.runtime || 'claude'
2058
2121
  if (childRuntime === 'codex' && args?.provider) return okText('A Codex lane uses its Codex/OpenAI login in v1; custom bridge providers are not wired to Codex yet. Omit provider or spawn a Claude lane for that provider.')
2122
+ const childCatalog = childRuntime === 'codex' ? readCodexModels() : []
2123
+ if (childRuntime === 'codex' && args?.model && !modelCatalogValues(childCatalog).has(args.model)) {
2124
+ return okText(`Could not open a Codex lane on ${JSON.stringify(args.model)} — that model is not in this host's visible Codex catalog.`)
2125
+ }
2126
+ if (childRuntime === 'claude' && !args?.provider && args?.model && /^gpt-/i.test(args.model)) {
2127
+ return okText(`Could not open a Claude lane on Codex model ${JSON.stringify(args.model)}. Choose runtime="codex" or a Claude model.`)
2128
+ }
2129
+ const childModel = args?.model || (args?.sliceType
2130
+ ? spawnedLaneModelFor({ sliceType: args.sliceType, runtime: childRuntime, catalog: childCatalog })
2131
+ : undefined)
2059
2132
  const now = Date.now()
2060
2133
  const gate = spawnDecision({
2061
2134
  hop: entry.hop || 0,
@@ -2090,7 +2163,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
2090
2163
  // plan-mode lane would stall on ExitPlanMode cards. An explicit args.mode still wins;
2091
2164
  // otherwise inherit, but fall plan → default (an autonomous worker doesn't plan-gate).
2092
2165
  const childMode = args?.mode || (entry.mode === 'plan' ? 'default' : entry.mode)
2093
- openStructured({ id: newId, runtime: childRuntime, model: args?.model, provider: args?.provider, mode: childMode })
2166
+ openStructured({ id: newId, runtime: childRuntime, model: childModel, provider: args?.provider, mode: childMode })
2094
2167
  const ne = sessions.get(newId)
2095
2168
  if (!ne) { if (args?.name) { delete termNames[newId]; saveNames(room, termNames) } return okText('Could not open a new lane — the terminal cap may have just been reached. Close one and retry.') }
2096
2169
  ne.spawnedBy = id // ownership: only the spawner may close_terminal it
@@ -2109,11 +2182,11 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
2109
2182
  }
2110
2183
  return okText(`Opened idle agent lane ${newRef}${args?.name ? ` ("${args.name}")` : ''}. Hand it work with post_to_terminal, or a person can type into it.`)
2111
2184
  },
2112
- ),
2185
+ )] : []),
2113
2186
  // Research lane — run a REAL multi-source search + adversarial verification and
2114
2187
  // return sourced, verdict-tagged findings. Calls the verified web backend
2115
2188
  // (/api/research-run) as the room owner; plan-gated + budget-capped server-side.
2116
- tool(
2189
+ ...(!entry.flowRole ? [tool(
2117
2190
  'research',
2118
2191
  'Run a REAL multi-source web research + verification on a factual question and surface sourced, verdict-tagged findings to the room. Use when the people would genuinely benefit from looking something up or settling an external-fact question — pricing, "is X still maintained / deprecated", "is that benchmark real", a debate over current facts. OFFER it first in plain language ("want me to spawn a research lane on that?") and only call it once they agree — it spends (plan-gated: Free 5 / Plus 100 runs per month) and takes ~1 minute. It searches the web, reads sources, and returns each claim marked HELD or REJECTED with citations. Present the findings clearly and let both people weigh the sources.',
2119
2192
  {
@@ -2144,13 +2217,13 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
2144
2217
  `\n\nPresent these to the room and let both people weigh the sources — flag which held claims rest on a source they might not trust.`
2145
2218
  )
2146
2219
  },
2147
- ),
2220
+ )] : []),
2148
2221
  // Tier C+ — CLOSE a lane the caller spawned. Restricted to self-spawned agent
2149
2222
  // lanes (spawnedBy === this id): an agent must never be able to kill a sibling
2150
2223
  // someone else is working in, nor the host/attached terminal. The fan-out
2151
2224
  // cleanup half of spawn_terminal. Routes through endStructured (same path as
2152
2225
  // the web's code-close), which no-ops if the id isn't a live session.
2153
- tool(
2226
+ ...(!entry.flowRole ? [tool(
2154
2227
  'close_terminal',
2155
2228
  'Close an agent lane that YOU opened with spawn_terminal (identified by ref/id/name). You can only close lanes you spawned yourself — never a sibling someone else is working in, and never the main terminal. Use it to clean up after a fan-out once you have collected the results with read_terminal.',
2156
2229
  {
@@ -2167,38 +2240,47 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
2167
2240
  endStructured(target.id)
2168
2241
  return okText(`Closed agent lane ${target.ref}${target.name ? ` ("${target.name}")` : ''}.`)
2169
2242
  },
2170
- ),
2243
+ )] : []),
2171
2244
  // Flow lane → mark this slice done (lane-done return path). Reads the lane's
2172
2245
  // worktree HEAD as commit_sha (the atomic-revert target) + broadcasts
2173
2246
  // flow-task-done → the room flips the task done + dispatches the next wave
2174
2247
  // (slices whose deps just got satisfied).
2175
- tool(
2248
+ ...(entry.flowRole === 'builder' ? [tool(
2176
2249
  'mark_flow_done',
2177
2250
  "ThinkPool Flow ONLY — call this ONCE when your slice is built, RUNS, and meets its acceptance criteria. Records your latest commit as the slice's atomic-revert target and signals the room that this slice is done (which unblocks slices that depended on you). No arguments: your commit is read from your worktree. Do not call before your slice actually runs + meets acceptance.",
2178
2251
  {},
2179
2252
  async () => ({ content: [{ type: 'text', text: await markFlowDone() }] }),
2180
- ),
2253
+ )] : []),
2181
2254
  // FL-B1 — the conductor SUBMITS its decomposition through THIS tool, not the built-in
2182
2255
  // ExitPlanMode. In the current SDK, ExitPlanMode is a deferred tool the conductor must
2183
2256
  // ToolSearch for — and it hangs there, wedging every flow in `planning`. A dedicated MCP
2184
2257
  // tool removes all dependence on plan-mode/ExitPlanMode/deferred-tool discovery: validate
2185
2258
  // the task-graph here, broadcast flow-plan (the same path the old interception used), and
2186
2259
  // reject malformed plans back to the conductor so it re-emits.
2187
- tool(
2260
+ ...(entry.flowRole === 'conductor' ? [tool(
2188
2261
  'submit_flow_plan',
2189
2262
  'ThinkPool Flow CONDUCTOR ONLY — submit your decomposition for human approval. Call this ONCE when your task-graph is ready. Pass it as `plan`: a JSON string {"summary":"<one line>","tasks":[{"key":"<kebab>","title":"…","scope":"…","acceptance":"…","deps":["<key>"],"sliceType":"feature|scaffold|review|fix"}]}. The room validates it is an acyclic DAG, persists the tasks, and shows the approval card. This is how a conductor FINISHES — do NOT use ExitPlanMode, do NOT write a plan file.',
2190
2263
  { plan: z.string().describe('the task-graph as a JSON string (the {summary, tasks:[…]} object)') },
2191
2264
  async (args) => {
2192
2265
  const okText = (t) => ({ content: [{ type: 'text', text: t }] })
2193
- if (!entry.flowSessionId) return okText('Not a Flow conductor — there is no flow to submit a plan for.')
2266
+ if (!entry.flowSessionId || entry.flowRole !== 'conductor') return okText('Not a Flow conductor — there is no flow to submit a plan for.')
2194
2267
  let norm
2195
- try { norm = normalizePlanOutput(args?.plan || '') }
2268
+ try { norm = validatePlanForRuntime(normalizePlanOutput(args?.plan || ''), entry.runtime) }
2196
2269
  catch (e) { return okText(`Plan REJECTED: ${e?.message || e}. Re-call submit_flow_plan with valid JSON — a non-empty "tasks" array, every "deps" entry referencing an existing task "key", and no cycles.`) }
2197
2270
  bcast('flow-plan', { term: id, flowId: entry.flowSessionId, plan: JSON.stringify(norm) }, flowChannel)
2198
2271
  process.stderr.write(`\n ${A.mag}◆ flow plan submitted — flow ${String(entry.flowSessionId).slice(0, 8)} (${norm.tasks.length} slice${norm.tasks.length === 1 ? '' : 's'}).${A.rst}\n`)
2199
2272
  return okText(`Plan submitted (${norm.tasks.length} slice${norm.tasks.length === 1 ? '' : 's'}) — the room is showing the human an approval card. You are DONE; stop here and wait for approval (lane dispatch is the room's job).`)
2200
2273
  },
2201
- ),
2274
+ )] : []),
2275
+ ...(entry.flowRole === 'reviewer' ? [tool(
2276
+ 'submit_flow_review',
2277
+ 'ThinkPool Flow REVIEWER ONLY — submit one bounded adversarial review round. Pass `verdict` as JSON: {"pass":boolean,"reasons":["specific evidence"],"taskKey":"authorized reviewed task","exhausted":boolean}. This is the only review completion path; do not call mark_flow_done and do not write FLOW_REVIEW.json.',
2278
+ { verdict: z.string().describe('the structured review verdict as a JSON string') },
2279
+ async (args) => {
2280
+ const result = await onReviewVerdict(args?.verdict || '')
2281
+ return { content: [{ type: 'text', text: result.message }] }
2282
+ },
2283
+ )] : []),
2202
2284
  ],
2203
2285
  })
2204
2286
  // S5 (slice 1b) — the review-lane write-block, wired into the live PreToolUse path.
@@ -2265,15 +2347,15 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
2265
2347
  // A Flow CONDUCTOR (flowSessionId set, no flowTaskKey) must DECOMPOSE, not explore via
2266
2348
  // a fan-out of Task/Explore subagents, and must not build. Hard-block Task/Bash/Write for
2267
2349
  // conductors so it reads with Read/Grep/Glob and submits via the FLOW_PLAN.json sentinel.
2268
- blockSubagents: !!flowSessionId && !flowTaskKey,
2350
+ blockSubagents: entry.flowRole === 'conductor',
2269
2351
  // FL-B1 — the conductor submits its task-graph by Writing FLOW_PLAN.json; the PreToolUse
2270
2352
  // hook intercepts that Write and calls this. Validate the DAG and broadcast flow-plan (the
2271
2353
  // client persists it + shows the approval card); a malformed plan comes straight back to
2272
2354
  // the conductor as the tool result so it re-emits. Returns { ok, message }.
2273
2355
  onSubmitPlan: (planText) => {
2274
- if (!entry.flowSessionId) return { ok: false, message: 'Not a Flow conductor — nothing to submit.' }
2356
+ if (!entry.flowSessionId || entry.flowRole !== 'conductor') return { ok: false, message: 'Not a Flow conductor — nothing to submit.' }
2275
2357
  let norm
2276
- try { norm = normalizePlanOutput(planText || '') }
2358
+ try { norm = validatePlanForRuntime(normalizePlanOutput(planText || ''), entry.runtime) }
2277
2359
  catch (e) { return { ok: false, message: `Plan REJECTED: ${e?.message || e}. Re-Write FLOW_PLAN.json with valid JSON — a non-empty "tasks" array, every "deps" entry referencing an existing task "key", and no cycles.` } }
2278
2360
  bcast('flow-plan', { term: id, flowId: entry.flowSessionId, plan: JSON.stringify(norm) }, flowChannel)
2279
2361
  process.stderr.write(`\n ${A.mag}◆ flow plan submitted — flow ${String(entry.flowSessionId).slice(0, 8)} (${norm.tasks.length} slice${norm.tasks.length === 1 ? '' : 's'}).${A.rst}\n`)
@@ -2281,9 +2363,9 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
2281
2363
  },
2282
2364
  // FL-B1 (lane side) — a flow lane signals slice-done by Writing FLOW_DONE; the hook
2283
2365
  // routes that Write here (mark_flow_done MCP tool is deferred → ToolSearch → hangs).
2284
- onLaneDone: flowTaskKey ? (async () => ({ ok: true, message: await markFlowDone() })) : null,
2366
+ onLaneDone: entry.flowRole === 'builder' ? (async () => ({ ok: true, message: await markFlowDone() })) : null,
2285
2367
  // FL-B4 — a review lane records its verdict via FLOW_REVIEW.json (revert-on-fail).
2286
- onReviewVerdict: flowTaskKey ? onReviewVerdict : null,
2368
+ onReviewVerdict: entry.flowRole === 'reviewer' ? onReviewVerdict : null,
2287
2369
  mcpServers: { thinkpool: peekServer },
2288
2370
  // Tier C precheck — the PreToolUse gate calls this BEFORE raising a card, so a
2289
2371
  // disabled/looping/over-cap post never bothers a person. Closes over `entry`.
@@ -2316,15 +2398,32 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
2316
2398
  onEvent: (evt) => {
2317
2399
  // Self-heal a stale resume — the saved SDK session expired. Reopen fresh,
2318
2400
  // keeping the transcript (scrollback survives; live context is gone).
2319
- if (resume && !entry.recovered && evt.kind === 'error' && /No conversation found/i.test(evt.message || '')) {
2320
- entry.recovered = true
2321
- process.stderr.write(`\n ◆ saved session expired — starting fresh (transcript kept).\n`)
2322
- try { entry.session?.end() } catch { /* noop */ }
2323
- try { entry.mockupWatcher?.close() } catch { /* noop */ }
2324
- try { void entry.viewport?.stop()?.catch(() => {}) } catch { /* noop */ }
2325
- sessions.delete(id)
2326
- openStructured({ id, runtime: entry.runtime, model, log: entry.log, commands: entry.commands, mode: entry.mode, provider: entry.provider, spawnedBy: entry.spawnedBy, sideParent: entry.sideParent, sideTask: entry.sideTask, pendingSideContexts: entry.pendingSideContexts, rolePrompt: entry.rolePrompt, cwd: entry.cwd, managedWorktree: entry.managedWorktree })
2327
- return
2401
+ if (resume && evt.kind === 'error') {
2402
+ const recoveryRecap = entry.interruptedRecap || buildRecapFromLog(entry.log, RECAP_CAP)
2403
+ const recovery = recoverMissingResumeOnce(entry, {
2404
+ message: evt.message,
2405
+ recap: recoveryRecap,
2406
+ reopen: (carryRecap) => {
2407
+ try { entry.mockupWatcher?.close() } catch { /* noop */ }
2408
+ sessions.delete(id)
2409
+ openStructured({
2410
+ id, runtime: entry.runtime, model: entry.model || model, effort: entry.effort,
2411
+ provider: entry.provider, log: entry.log, commands: entry.commands, mode: entry.mode,
2412
+ spawnedBy: entry.spawnedBy, sideParent: entry.sideParent, sideTask: entry.sideTask, pendingSideContexts: entry.pendingSideContexts, flowSessionId: entry.flowSessionId,
2413
+ flowTaskKey: entry.flowTaskKey, flowRole: entry.flowRole, flowReviewTarget: entry.flowReviewTarget,
2414
+ flowReviewTargets: entry.flowReviewTargets, flowReviewRound: entry.flowReviewRound,
2415
+ dispatchBaseSha: entry.dispatchBaseSha, revertTarget: entry.revertTarget, cwd: entry.cwd,
2416
+ managedWorktree: entry.managedWorktree, rolePrompt: entry.rolePrompt,
2417
+ reviewSliceRoots: entry.reviewSliceRoots, openedAt: entry.openedAt,
2418
+ lastUsage: entry.lastUsage, carryRecap,
2419
+ })
2420
+ return sessions.get(id) || null
2421
+ },
2422
+ })
2423
+ if (recovery.recovered) {
2424
+ process.stderr.write(`\n ◆ saved session expired — started fresh with transcript recap (${recovery.delivery}).\n`)
2425
+ return
2426
+ }
2328
2427
  }
2329
2428
  // Stamp a wall-clock ts AND a stable cid on every transcript event before
2330
2429
  // it's logged + broadcast. ts: lets the web client sort agent turns
@@ -2398,7 +2497,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
2398
2497
  // explicit signal (FLOW_DONE / mark_flow_done), which left the slice stuck 'dispatched'
2399
2498
  // forever → the flow never assembled. Auto-record on turn-end (idempotent — markFlowDone
2400
2499
  // no-ops if already done, so the explicit sentinel still works as the fast path).
2401
- if (entry.flowSessionId && entry.flowTaskKey && evt.kind === 'result' && !entry.flowDone) {
2500
+ if (entry.flowSessionId && entry.flowTaskKey && !entry.flowDone && legacyBuilderCompletionAllowed({ runtime: entry.runtime, flowRole: entry.flowRole, eventKind: evt.kind, eventSubtype: evt.subtype })) {
2402
2501
  Promise.resolve().then(() => markFlowDone()).catch(() => { /* best-effort */ })
2403
2502
  }
2404
2503
  // The init system event carries the session's slash command list. Stash it
@@ -2419,11 +2518,8 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
2419
2518
  // human turn arrived first, the code-turn handler already consumed + cleared it
2420
2519
  // (prepended to that turn), so this no-ops. Spec: docs/specs/2026-07-08-provider-switch-context-carry.md.
2421
2520
  if (evt.kind === 'system' && entry.pendingRecap) {
2422
- const recap = entry.pendingRecap
2423
- entry.pendingRecap = null
2424
- entry.flush?.()
2425
- try { entry.session?.sendTurn(recap) } catch { /* not live — stays idle, user can continue manually */ }
2426
- process.stderr.write(`\n ◆ carried ${recap.length}c context recap into fresh session (${id.slice(0, 8)}) after provider/restore switch.\n`)
2521
+ const recapLength = entry.pendingRecap.length
2522
+ if (dispatchPendingRecap(entry)) process.stderr.write(`\n ◆ carried ${recapLength}c context recap into fresh session (${id.slice(0, 8)}) after provider/restore switch.\n`)
2427
2523
  }
2428
2524
  // A background-warmed terminal is now live — free its warmer slot so the next queued
2429
2525
  // idle terminal starts booting.
@@ -2730,7 +2826,7 @@ function respawnStructured(id, provider) {
2730
2826
  // openStructured seed from the TARGET provider's configured model, which is the
2731
2827
  // only model this lane was ever asked for. A same-env model change never reaches
2732
2828
  // here — that path is an in-place setModel (see provider-switch).
2733
- const { runtime, log, commands, mode, spawnedBy, sideParent, sideTask, pendingSideContexts, flowSessionId, flowTaskKey, cwd, rolePrompt, reviewSliceRoots, openedAt } = s
2829
+ const { runtime, log, commands, mode, spawnedBy, sideParent, sideTask, pendingSideContexts, flowSessionId, flowTaskKey, flowRole, flowReviewTarget, flowReviewTargets, flowReviewRound, dispatchBaseSha, revertTarget, cwd, managedWorktree, rolePrompt, reviewSliceRoots, openedAt, lastUsage } = s
2734
2830
  // Context-carry (2026-07-08): a provider switch is not an SDK resume — the new backend
2735
2831
  // starts blank mid-conversation. Synthesize a plain-text recap from the VISIBLE log NOW
2736
2832
  // (before teardown) and hand it to the fresh session as its first turn so the agent
@@ -2749,7 +2845,7 @@ function respawnStructured(id, provider) {
2749
2845
  // sessionData() (provider included) synchronously on open, so a bridge restart
2750
2846
  // restores the lane on its CURRENT provider, not the original — and its next
2751
2847
  // announce carries the new provider badge (additive {id,name} projection).
2752
- openStructured({ id, runtime, provider, log, commands, mode, spawnedBy, sideParent, sideTask, pendingSideContexts, flowSessionId, flowTaskKey, cwd, rolePrompt, reviewSliceRoots, openedAt, carryRecap })
2848
+ openStructured({ id, runtime, provider, log, commands, mode, spawnedBy, sideParent, sideTask, pendingSideContexts, flowSessionId, flowTaskKey, flowRole, flowReviewTarget, flowReviewTargets, flowReviewRound, dispatchBaseSha, revertTarget, cwd, managedWorktree, rolePrompt, reviewSliceRoots, openedAt, lastUsage, carryRecap })
2753
2849
  return true
2754
2850
  }
2755
2851
 
@@ -3530,49 +3626,32 @@ channel
3530
3626
  const okey = (r) => r._bt || r.openedAt || r.savedAt || 0
3531
3627
  all.sort((a, b) => okey(a) - okey(b))
3532
3628
  if (all.length) for (const rec of all) {
3629
+ const wasInterrupted = restoredTurnOpen(rec.log || [])
3630
+ const resumable = canResume(rec)
3631
+ const recoveryRecap = wasInterrupted ? buildRecapFromLog(rec.log || [], RECAP_CAP) : ''
3533
3632
  // FL-M6 — restore the flow context (id/role/cwd) so an in-flight flow survives a
3534
3633
  // bridge restart: the conductor keeps its subagent-block + plan interception, and
3535
3634
  // lanes keep their worktree cwd + the ability to mark done.
3536
- openStructured({ id: rec.id, runtime: rec.runtime || 'claude', model: rec.model || undefined, effort: rec.effort, provider: rec.runtime === 'codex' ? undefined : rec.provider || undefined, resume: canResume(rec) ? rec.sessionId : undefined, log: rec.log, commands: rec.commands, mode: rec.mode, spawnedBy: rec.spawnedBy, sideParent: rec.sideParent, sideTask: rec.sideTask, pendingSideContexts: rec.pendingSideContexts, flowSessionId: rec.flowSessionId, flowTaskKey: rec.flowTaskKey, cwd: rec.cwd, managedWorktree: rec.managedWorktree, rolePrompt: rec.rolePrompt, reviewSliceRoots: rec.reviewSliceRoots, openedAt: rec.openedAt, lastUsage: rec.lastUsage, carryRecap: rec.carryRecap,
3635
+ openStructured({ id: rec.id, runtime: rec.runtime || 'claude', model: rec.model || undefined, effort: rec.effort, provider: rec.runtime === 'codex' ? undefined : rec.provider || undefined, resume: resumable ? rec.sessionId : undefined, log: rec.log, commands: rec.commands, mode: rec.mode, spawnedBy: rec.spawnedBy, sideParent: rec.sideParent, sideTask: rec.sideTask, pendingSideContexts: rec.pendingSideContexts, flowSessionId: rec.flowSessionId, flowTaskKey: rec.flowTaskKey, flowRole: rec.flowRole, flowReviewTarget: rec.flowReviewTarget, flowReviewTargets: rec.flowReviewTargets, flowReviewRound: rec.flowReviewRound, dispatchBaseSha: rec.dispatchBaseSha, revertTarget: rec.revertTarget, cwd: rec.cwd, managedWorktree: rec.managedWorktree, rolePrompt: rec.rolePrompt, reviewSliceRoots: rec.reviewSliceRoots, openedAt: rec.openedAt, lastUsage: rec.lastUsage, carryRecap: wasInterrupted ? recoveryRecap : rec.carryRecap,
3537
3636
  // Lazy-boot restored terminals that were IDLE + not part of a flow: their transcript
3538
3637
  // shows immediately; the query boots on first turn. Mid-turn + flow terminals boot now
3539
3638
  // (mid-turn needs auto-resume; flow needs its lane live).
3540
- defer: !rec.flowSessionId && !restoredTurnOpen(rec.log || []) })
3639
+ defer: !rec.flowSessionId && !wasInterrupted })
3541
3640
  const re = sessions.get(rec.id)
3542
- if (re && rec.flowReviewTarget) re.flowReviewTarget = rec.flowReviewTarget
3543
- // S4 — a resumed re-dispatch lane keeps its surviving revert target across a bridge restart.
3544
- if (re && rec.revertTarget) re.revertTarget = rec.revertTarget
3545
3641
  // FL-M6 — a flow LANE whose turn had already ENDED before the restart finished its
3546
3642
  // slice, but its done-signal was lost while the bridge was down (resume won't re-run
3547
3643
  // an ended turn, so it would sit idle → the flow stalls forever). Reconcile: re-record
3548
3644
  // it done. A lane still MID-turn (restoredTurnOpen) resumes + finishes normally.
3549
- if (re && rec.flowTaskKey && !restoredTurnOpen(rec.log || [])) {
3645
+ const lastTerminal = [...(rec.log || [])].reverse().find((event) => event?.kind === 'result' || event?.kind === 'error')
3646
+ if (re && rec.flowTaskKey && legacyBuilderCompletionAllowed({ runtime: re.runtime, flowRole: re.flowRole, eventKind: lastTerminal?.kind, eventSubtype: lastTerminal?.subtype, interrupted: wasInterrupted })) {
3550
3647
  setTimeout(() => { try { re.markFlowDone?.() } catch { /* noop */ } }, 2000)
3551
3648
  }
3552
- // Auto-resume (Max, 2026-07-02): a PLAIN terminal (NOT a flow lane — those have their
3553
- // own re-dispatch) that was MID-TURN when the bridge stopped picks the work back up
3554
- // instead of sitting aborted. If its SDK context resumes (canResume), send ONE
3555
- // "continue" — like the user typing "continue" after a restart. We DON'T fire on a
3556
- // fixed timer: a resume takes ~40s to become live (MCP boot + sanitizeSession), and a
3557
- // 2.5s sendTurn pushed into an input stream nothing is consuming yet is silently lost
3558
- // (the 2026-07-02 "didn't autocontinue" bug). Instead FLAG it here and fire the single
3559
- // continue when the session's init `system` event actually arrives (onEvent), i.e. the
3560
- // instant it's live. openStructured already closed the interrupted turn idle, so a
3561
- // continue that can't be delivered just leaves it idle (recoverable) — never a hang.
3562
- if (re && !rec.flowSessionId && canResume(rec) && restoredTurnOpen(rec.log || [])) {
3563
- const recovery = armInterruptedResume(re)
3649
+ // Every interrupted role resumes exactly once. Flow lanes do not have a
3650
+ // startup redispatch path; excluding them here stranded conductors/builders.
3651
+ if (re && wasInterrupted) {
3652
+ const recovery = recoverInterruptedTurn(re, { resumable, recap: recoveryRecap })
3564
3653
  if (recovery === 'sent') process.stderr.write(`\n ◆ auto-resumed interrupted Codex turn (${rec.id.slice(0, 8)}) — sent continue.\n`)
3565
- }
3566
- // Context-carry (2026-07-08): the SAME mid-turn case but the SDK context CANNOT
3567
- // resume (canResume false — a stale/expired session). A bare "continue" would land
3568
- // in a blank backend, so instead carry a plain-text recap of the visible log as the
3569
- // first turn — the interrupted work resumes WITH its history. Mid-turn only
3570
- // (restoredTurnOpen): we never arm an unprompted turn on an idle restored lane (its
3571
- // transcript stays visible; the person starts their next request fresh). If the
3572
- // person types before this recap fires, the code-turn handler prepends it instead.
3573
- // Flow lanes re-dispatch separately (excluded above).
3574
- else if (re && !rec.flowSessionId && !canResume(rec) && restoredTurnOpen(rec.log || [])) {
3575
- re.pendingRecap = buildRecapFromLog(rec.log || [], RECAP_CAP)
3654
+ else if (recovery === 'recap-sent') process.stderr.write(`\n ◆ resumed interrupted Codex turn (${rec.id.slice(0, 8)}) with a fresh context recap.\n`)
3576
3655
  }
3577
3656
  }
3578
3657
  // Fresh open: ONLY for an explicit `-- <claude>` share. In account mode
@@ -3612,6 +3691,36 @@ channel
3612
3691
  }
3613
3692
  })
3614
3693
 
3694
+ async function handleFlowRevert(payload) {
3695
+ try {
3696
+ return await executeFlowRevert({
3697
+ payload,
3698
+ bridgeName: name,
3699
+ lanes: [...sessions.entries()],
3700
+ prepareLane: (entry, p) => prepareRedispatch({
3701
+ lane: { cwd: entry.cwd, session: entry.session, commitSha: entry.commitSha, revertTarget: p.commitSha ?? entry.commitSha ?? entry.revertTarget ?? null },
3702
+ sanitize: sanitizeSession,
3703
+ }),
3704
+ rememberRedispatch: (prepared, _entry, p) => {
3705
+ flowRedispatch.set(redispatchKey(p.flowId, p.taskKey), { resumeSessionId: prepared.resumeSessionId, revertTarget: prepared.revertTarget })
3706
+ if (prepared.healed) process.stderr.write(`\n ${A.dim}◆ flow re-dispatch heal — ${p.taskKey}: ${prepared.healed} dangling tool block(s) healed before resume.${A.rst}\n`)
3707
+ },
3708
+ endLane: (laneId) => { try { endStructured(laneId) } catch { /* already gone */ } },
3709
+ stopPreview: stopFlowPreviews,
3710
+ wait: () => new Promise((resolve) => setTimeout(resolve, 400)),
3711
+ revert: (p) => revertLane({ flowId: p.flowId, taskKey: p.taskKey }),
3712
+ broadcastReverted: (out) => {
3713
+ bcast('flow-reverted', { term: 'flow', flowId: out.flowId, taskKey: out.taskKey, reviewTaskKey: out.reviewTaskKey, branch: out.branch }, flowChannel)
3714
+ process.stderr.write(`\n ${A.yel}◆ flow slice reverted — ${out.taskKey} (${out.branch}).${A.rst}\n`)
3715
+ },
3716
+ onPrepError: (error, _entry, p) => process.stderr.write(`\n ${A.yel}◆ flow re-dispatch prep failed (${p.taskKey}): ${error?.message || error} — will re-dispatch fresh.${A.rst}\n`),
3717
+ })
3718
+ } catch (error) {
3719
+ process.stderr.write(`\n ${A.yel}◆ flow revert failed: ${error?.message || error}${A.rst}\n`)
3720
+ return { ok: false, error: error?.message || String(error) }
3721
+ }
3722
+ }
3723
+
3615
3724
  // ── Thinkpool Flow control channel (tpflow:<room>) ──────────────────────────
3616
3725
  // Separate from the room channel above so the web's useFlow hook can own its own
3617
3726
  // topic without colliding with the room's tpcode channel (see the flowChannel def +
@@ -3630,6 +3739,16 @@ flowChannel
3630
3739
  if (!payload?.flowId) return
3631
3740
  if (payload.host && payload.host !== name) return
3632
3741
  for (const [, e] of sessions) if (e.flowSessionId === payload.flowId) return // already conducting this flow
3742
+ const origin = payload.originTerm ? sessions.get(payload.originTerm) : null
3743
+ // A current client names the invoking terminal. Only the bridge that actually
3744
+ // owns that terminal may launch its conductor; never trust a spoofed runtime/model
3745
+ // from a broadcast received by another machine. Legacy clients omitted originTerm
3746
+ // and were Claude-only, so that exact path remains supported.
3747
+ if (payload.originTerm && !origin) return
3748
+ const flowRuntime = origin ? normalizeFlowRuntime(origin.runtime, null) : 'claude'
3749
+ if (!flowRuntime) return
3750
+ const flowCatalog = flowRuntime === 'codex' ? (origin?.models || readCodexModels()) : []
3751
+ const conductorModel = flowConductorModelFor({ runtime: flowRuntime, originModel: origin?.model, catalog: flowCatalog })
3633
3752
  const cid = randomUUID()
3634
3753
  termNames[cid] = `Flow · ${String(payload.flowId).slice(0, 6)}`
3635
3754
  saveNames(room, termNames)
@@ -3640,7 +3759,7 @@ flowChannel
3640
3759
  // Model tiers (2026-07-03-flow-lane-model-tiers): the conductor keeps whatever brain it was
3641
3760
  // given by default (TP_FLOW_CONDUCTOR_MODEL unset → undefined → today's behavior); set the env
3642
3761
  // to pin a cheaper/smarter conductor. Lanes get tiered below via laneModelFor.
3643
- openStructured({ id: cid, model: process.env.TP_FLOW_CONDUCTOR_MODEL || undefined, mode: 'default', rolePrompt: FLOW_CONDUCTOR_PROMPT, flowSessionId: payload.flowId, spawnedBy: `flow:${payload.flowId}` })
3762
+ openStructured({ id: cid, runtime: flowRuntime, model: conductorModel, mode: flowRuntime === 'codex' ? 'plan' : 'default', rolePrompt: flowRuntime === 'codex' ? FLOW_CODEX_CONDUCTOR_PROMPT : FLOW_CONDUCTOR_PROMPT, flowSessionId: payload.flowId, flowRole: 'conductor', spawnedBy: `flow:${payload.flowId}` })
3644
3763
  const ce = sessions.get(cid)
3645
3764
  if (ce?.session) { try { ce.session.sendTurn(payload.prompt || '') } catch { /* session still starting */ } }
3646
3765
  process.stderr.write(`\n ${A.mag}◆ flow conductor launched — flow ${String(payload.flowId).slice(0, 8)} (plan mode).${A.rst}\n`)
@@ -3654,6 +3773,14 @@ flowChannel
3654
3773
  // building (DB writes stay room-side, member-authed).
3655
3774
  if (!payload?.flowId || !Array.isArray(payload.tasks)) return
3656
3775
  if (payload.host && payload.host !== name) return
3776
+ const conductor = [...sessions.values()].find((entry) => entry.flowSessionId === payload.flowId && entry.flowRole === 'conductor')
3777
+ if (!conductor) {
3778
+ process.stderr.write(`\n ${A.yel}◆ flow dispatch held — this bridge has no live/restored conductor for ${String(payload.flowId).slice(0, 8)}.${A.rst}\n`)
3779
+ return
3780
+ }
3781
+ const flowRuntime = normalizeFlowRuntime(conductor.runtime, null)
3782
+ if (!flowRuntime) return
3783
+ const flowCatalog = flowRuntime === 'codex' ? (conductor.models || readCodexModels()) : []
3657
3784
  const assignments = []
3658
3785
  // Step 6 — cap concurrent Flow lanes (invariant: ≤ FLOW_LIMITS.maxConcurrentLanes).
3659
3786
  // Overflow tasks stay pending; the room re-dispatches them in the next wave.
@@ -3682,7 +3809,13 @@ flowChannel
3682
3809
  // Step 4 — a `review` slice gets the adversarial reviewer prompt (try-to-break),
3683
3810
  // every other slice gets the builder prompt.
3684
3811
  const isReview = t.slice_type === 'review'
3812
+ if (!validReviewTargetShape(t, flowRuntime)) {
3813
+ process.stderr.write(`\n ${A.yel}◆ flow dispatch held — Codex review ${t.task_key} must target exactly one dependency.${A.rst}\n`)
3814
+ continue
3815
+ }
3685
3816
  const { dir } = createFlowWorktree({ flowId: payload.flowId, taskKey: t.task_key })
3817
+ let dispatchBaseSha = null
3818
+ try { dispatchBaseSha = execFileSync('git', ['-C', dir, 'rev-parse', 'HEAD'], { encoding: 'utf8' }).trim() } catch { /* non-git fixture */ }
3686
3819
  const laneId = randomUUID()
3687
3820
  termNames[laneId] = `Flow · ${t.task_key}`
3688
3821
  saveNames(room, termNames)
@@ -3715,12 +3848,14 @@ flowChannel
3715
3848
  // a lane later, on demand, via activateLaneSkill — never the base prompt here.
3716
3849
  // S4 — resume: on a re-dispatch, replay the killed lane's HEALED transcript (Heal-3'd,
3717
3850
  // no dangling tool_use → no 400) instead of a cold start; undefined for a fresh lane.
3718
- const laneBase = isReview ? FLOW_REVIEWER_PROMPT : FLOW_LANE_PROMPT
3851
+ const laneBase = flowRuntime === 'codex'
3852
+ ? (isReview ? FLOW_CODEX_REVIEWER_PROMPT : FLOW_CODEX_LANE_PROMPT)
3853
+ : (isReview ? FLOW_REVIEWER_PROMPT : FLOW_LANE_PROMPT)
3719
3854
  const laneRolePrompt = buildLanePrompt({ base: laneBase })
3720
3855
  // Model tiers (2026-07-03-flow-lane-model-tiers): pick the lane's brain by slice_type
3721
3856
  // (scaffold→sonnet, feature/fix/review→opus; env can blanket-override or `inherit` to
3722
3857
  // restore today's exact behavior). undefined → no model key passed (openStructured default).
3723
- openStructured({ id: laneId, cwd: dir, model: laneModelFor(t.slice_type), mode: 'bypassPermissions', rolePrompt: laneRolePrompt, flowSessionId: payload.flowId, flowTaskKey: t.task_key, spawnedBy: `flow:${payload.flowId}`, resume: redispatch?.resumeSessionId || undefined, reviewSliceRoots })
3858
+ openStructured({ id: laneId, runtime: flowRuntime, cwd: dir, model: flowLaneModelFor({ sliceType: t.slice_type, runtime: flowRuntime, catalog: flowCatalog }), mode: flowRuntime === 'codex' && isReview ? 'review' : 'bypassPermissions', rolePrompt: laneRolePrompt, flowSessionId: payload.flowId, flowTaskKey: t.task_key, flowRole: isReview ? 'reviewer' : 'builder', flowReviewTargets: isReview ? (t.deps || []) : [], dispatchBaseSha, revertTarget: redispatch?.revertTarget || null, spawnedBy: `flow:${payload.flowId}`, resume: redispatch?.resumeSessionId || undefined, reviewSliceRoots })
3724
3859
  const le = sessions.get(laneId)
3725
3860
  if (le) {
3726
3861
  // S4 — stamp the surviving revert target on the resumed lane so a later reviewer still
@@ -3745,7 +3880,9 @@ flowChannel
3745
3880
  : '') +
3746
3881
  `ACCEPTANCE to INDEPENDENTLY verify: ${t.acceptance || t.title}\n` +
3747
3882
  `\nProject (context): ${payload.flowPrompt || ''}\n\n` +
3748
- `Review it — run it, hunt for failure. When done, use the Write tool to write FLOW_REVIEW.json with {"pass":<boolean>,"reasons":["<specific finding>"],"taskKey":"<the slice you reviewed>"}. That Write IS your verdict.`
3883
+ (flowRuntime === 'codex'
3884
+ ? `Review it without mutating the builder worktree. For write-producing install/build/test commands, copy its source into scratch under your own current worktree first. Submit the structured verdict with the ThinkPool submit_flow_review MCP tool; never write FLOW_REVIEW.json or call mark_flow_done.`
3885
+ : `Review it — run it, hunt for failure. When done, use the Write tool to write FLOW_REVIEW.json with {"pass":<boolean>,"reasons":["<specific finding>"],"taskKey":"<the slice you reviewed>"}. That Write IS your verdict.`)
3749
3886
  : `[Flow lane — slice "${t.task_key}" of flow ${String(payload.flowId).slice(0, 8)}]\n` +
3750
3887
  `TITLE: ${t.title}\n` +
3751
3888
  `SCOPE (files you OWN — edit ONLY these): ${t.scope || '(none stated)'}\n` +
@@ -3757,7 +3894,8 @@ flowChannel
3757
3894
  ((crossWave) => crossWave ? `${crossWave}\n` : '')(
3758
3895
  assembleCrossWaveContext(payload.flowId, { baseDir: process.cwd(), deps: t.deps }).text) +
3759
3896
  `\nProject (context): ${payload.flowPrompt || ''}\n\n` +
3760
- `Build your slice. Own only your files. Done = it runs + meets acceptance. Commit when done.`
3897
+ `Build your slice. Own only your files. Done = it runs + meets acceptance. Commit when done.` +
3898
+ (flowRuntime === 'codex' ? ` Then call the ThinkPool mark_flow_done MCP tool; do not write FLOW_DONE.` : '')
3761
3899
  try { le.session.sendTurn(spec) } catch { /* session still starting */ }
3762
3900
  }
3763
3901
  }
@@ -3809,45 +3947,7 @@ flowChannel
3809
3947
  // Step 4 — atomic per-lane revert: roll back ONE slice (remove its worktree +
3810
3948
  // branch) without touching the others, when adversarial review rejects it. The
3811
3949
  // room re-dispatches the reverted task on the next wave.
3812
- if (!payload?.flowId || !payload?.taskKey) return
3813
- if (payload.host && payload.host !== name) return
3814
- try {
3815
- // FL-M8 — tear down the lane's live session FIRST. revertLane does `git worktree
3816
- // remove --force`; pulling the worktree out from under a still-running lane corrupts
3817
- // its process state. Retire it, let the SDK release the cwd, THEN remove the tree.
3818
- let killed = false
3819
- for (const [sid, e] of sessions) {
3820
- if (e.flowSessionId === payload.flowId && e.flowTaskKey === payload.taskKey && !e.flowDone) {
3821
- // S4 slice 2 — prepare a CLEAN re-dispatch BEFORE we kill the lane + remove its
3822
- // worktree. prepareRedispatch heals the transcript (Heal-3 consumer — a dangling
3823
- // tool_use from a mid-tool-call kill gets a synthetic tool_result so a resume can't
3824
- // 400) and captures the resume sessionId + the revert target (which revertLane's
3825
- // branch delete would otherwise destroy). We record it under flowId::taskKey so the
3826
- // next flow-dispatch wave resumes the healed session. Idempotent: healing an
3827
- // already-clean transcript is a no-op. No per-lane topic is captured — the resumed
3828
- // lane reuses the room-wide tpflow broadcast, so there is no H41 collision to guard.
3829
- try {
3830
- const prep = prepareRedispatch({
3831
- lane: { cwd: e.cwd, session: e.session, commitSha: e.commitSha, revertTarget: payload.commitSha ?? e.commitSha ?? null },
3832
- sanitize: sanitizeSession,
3833
- })
3834
- flowRedispatch.set(redispatchKey(payload.flowId, payload.taskKey), {
3835
- resumeSessionId: prep.resumeSessionId,
3836
- revertTarget: prep.revertTarget,
3837
- })
3838
- if (prep.healed) process.stderr.write(`\n ${A.dim}◆ flow re-dispatch heal — ${payload.taskKey}: ${prep.healed} dangling tool block(s) healed before resume.${A.rst}\n`)
3839
- } catch (err) { process.stderr.write(`\n ${A.yel}◆ flow re-dispatch prep failed (${payload.taskKey}): ${err?.message || err} — will re-dispatch fresh.${A.rst}\n`) }
3840
- e.flowDone = true
3841
- try { endStructured(sid) } catch { /* already gone */ }
3842
- stopFlowPreviews(payload.flowId, sid) // FL-M9 — drop the reverted lane's preview
3843
- killed = true
3844
- }
3845
- }
3846
- if (killed) await new Promise((r) => setTimeout(r, 400))
3847
- const { branch } = revertLane({ flowId: payload.flowId, taskKey: payload.taskKey })
3848
- bcast('flow-reverted', { term: 'flow', flowId: payload.flowId, taskKey: payload.taskKey, branch }, flowChannel)
3849
- process.stderr.write(`\n ${A.yel}◆ flow slice reverted — ${payload.taskKey} (${branch}).${A.rst}\n`)
3850
- } catch (e) { process.stderr.write(`\n ${A.yel}◆ flow revert failed: ${e?.message || e}${A.rst}\n`) }
3950
+ await handleFlowRevert(payload)
3851
3951
  })
3852
3952
  .on('broadcast', { event: 'flow-cancel' }, ({ payload }) => {
3853
3953
  // A user cancelled the flow (or it was interrupted). The client also flips the DB
@@ -3882,8 +3982,8 @@ designChannel
3882
3982
  const ok = !!record && !!live && live.revision === record.revision && record.revision === payload?.revision && sessions.has(record.term)
3883
3983
  designChannel.send({ type: 'broadcast', event: 'design-capability-res', payload: { previewId: String(payload?.previewId || ''), revision: String(payload?.revision || ''), ok } })
3884
3984
  })
3885
- .on('broadcast', { event: 'design-apply' }, ({ payload }) => {
3886
- const verdict = validateDesignRequest(payload, designArtifacts)
3985
+ .on('broadcast', { event: 'design-apply' }, async ({ payload }) => {
3986
+ const verdict = validateDesignRequest(payload, designArtifacts, room)
3887
3987
  const previewId = String(payload?.previewId || '')
3888
3988
  const requestId = String(payload?.cid || '').slice(0, 80)
3889
3989
  if (!verdict.ok) {
@@ -3893,6 +3993,20 @@ designChannel
3893
3993
  if (!requestId || seenDesignRequests.has(requestId)) return
3894
3994
  seenDesignRequests.add(requestId)
3895
3995
  if (seenDesignRequests.size > 500) seenDesignRequests.delete(seenDesignRequests.values().next().value)
3996
+ if (verdict.request.mode === 'image') {
3997
+ const localPath = await materializeDesignAsset({
3998
+ webBase: WEB_BASE,
3999
+ token: codeAuthToken,
4000
+ room,
4001
+ asset: verdict.request.asset,
4002
+ requestId,
4003
+ })
4004
+ if (!localPath) {
4005
+ designStatus({ previewId, requestId, state: 'failed', message: 'The private image draft could not be opened.' })
4006
+ return
4007
+ }
4008
+ verdict.request.asset.localPath = localPath
4009
+ }
3896
4010
  const term = verdict.record.term
3897
4011
  const queue = designQueues.get(term) || []
3898
4012
  designQueues.set(term, queue)