thinkpool-pair 0.7.221 → 0.7.222
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bridge.mjs +227 -131
- package/codex-session.mjs +17 -3
- package/flow-conductor.mjs +13 -1
- package/flow-host-revert.mjs +42 -0
- package/flow-models.mjs +77 -0
- package/flow-review.mjs +7 -0
- package/flow-task-graph.mjs +21 -0
- package/interrupted-resume.mjs +52 -1
- package/lane-worktree.mjs +16 -1
- package/package.json +3 -1
package/bridge.mjs
CHANGED
|
@@ -55,9 +55,10 @@ import { canonicalRoomFilePath, waitForNativeImages } from './codex-images.mjs'
|
|
|
55
55
|
import { createManagedLaneWorktree, removeManagedLaneWorktree } from './lane-worktree.mjs'
|
|
56
56
|
import { commandOnPath } from './agent-detect.mjs'
|
|
57
57
|
|
|
58
|
-
const STRUCTURED_MODES = new Set(['default', 'acceptEdits', 'plan', 'bypassPermissions'])
|
|
59
|
-
import { FLOW_CONDUCTOR_PROMPT, FLOW_LANE_PROMPT, buildConductorEnv, assembleCrossWaveContext, buildLanePrompt } from './flow-conductor.mjs'
|
|
60
|
-
import { normalizePlanOutput,
|
|
58
|
+
const STRUCTURED_MODES = new Set(['default', 'acceptEdits', 'plan', 'review', 'bypassPermissions'])
|
|
59
|
+
import { FLOW_CONDUCTOR_PROMPT, FLOW_LANE_PROMPT, FLOW_CODEX_CONDUCTOR_PROMPT, FLOW_CODEX_LANE_PROMPT, buildConductorEnv, assembleCrossWaveContext, buildLanePrompt } from './flow-conductor.mjs'
|
|
60
|
+
import { legacyBuilderCompletionAllowed, normalizePlanOutput, validatePlanForRuntime, validReviewTargetShape } from './flow-task-graph.mjs'
|
|
61
|
+
import { normalizeFlowRuntime, flowLaneModelFor, flowConductorModelFor, spawnedLaneModelFor, modelCatalogValues, resolveCodexModel, assertRuntimeModelCompatible } from './flow-models.mjs'
|
|
61
62
|
// S1 (context-offload) — durable digest store. mark_flow_done digests a closed slice in;
|
|
62
63
|
// the dispatch loop reads the bounded cross-wave context back out (via assembleCrossWaveContext,
|
|
63
64
|
// which enforces the CEILING). Consume the store — the internals live in flow-context-store.mjs.
|
|
@@ -75,7 +76,7 @@ function stopFlowPreviews (flowId, laneId = null) {
|
|
|
75
76
|
if (laneId ? key === `lane:${flowId}:${laneId}` : key.startsWith(`lane:${flowId}:`)) { try { h.stop() } catch { /* noop */ } }
|
|
76
77
|
}
|
|
77
78
|
}
|
|
78
|
-
import { FLOW_REVIEWER_PROMPT, revertLane, parseReviewVerdict, reviewVerdictToReflection } from './flow-review.mjs'
|
|
79
|
+
import { FLOW_REVIEWER_PROMPT, FLOW_CODEX_REVIEWER_PROMPT, revertLane, parseReviewVerdict, reviewVerdictToReflection } from './flow-review.mjs'
|
|
79
80
|
import { reviewGateDecision } from './flow-review-gate.mjs'
|
|
80
81
|
import { pairAdjudicationPrompt, reviewReflectionDecision, REVIEW_DEFAULTS } from './flow-review-reflect.mjs'
|
|
81
82
|
import { mergeWorktrees, inlineSingleHtml, initRepo } from './flow-assembly.mjs'
|
|
@@ -89,6 +90,7 @@ import { canDispatch, FLOW_LIMITS, makeBudget, recordSpend, killSwitchEnv } from
|
|
|
89
90
|
// Spec: docs/specs/2026-06-30-flow-build-s4-clean-redispatch.md
|
|
90
91
|
import { prepareRedispatch, redispatchKey } from './flow-redispatch.mjs'
|
|
91
92
|
import { sanitizeSession } from './transcript-sanitize.mjs'
|
|
93
|
+
import { executeFlowRevert } from './flow-host-revert.mjs'
|
|
92
94
|
// ACCEPTED LIMITATION — this registry is in-memory only. A bridge restart in the window between
|
|
93
95
|
// a flow-revert kill and the next dispatch wave loses the pending resume record, so the task
|
|
94
96
|
// re-dispatches COLD (fresh lane, no resume). That's safe: the healed transcript survives on
|
|
@@ -101,7 +103,7 @@ const flowRedispatch = new Map()
|
|
|
101
103
|
// broadcasts; without persistent state the cap can never bite.
|
|
102
104
|
const flowBudgets = new Map()
|
|
103
105
|
import { formatPeek, PEEK, siblingsOf, resolveSibling, crossPostDecision, CROSSPOST, spawnDecision, SPAWN, CROSSROOM, formatPairRoster, crossRoomPostDecision, formatRoomNow, formatClosableHint, laneStatusOf } from './cross-terminal.mjs'
|
|
104
|
-
import { armInterruptedResume, sendInterruptedContinue, supersedeInterruptedResume } from './interrupted-resume.mjs'
|
|
106
|
+
import { dispatchPendingRecap, recoverInterruptedTurn, recoverMissingResumeOnce, armInterruptedResume, sendInterruptedContinue, supersedeInterruptedResume } from './interrupted-resume.mjs'
|
|
105
107
|
import { turnInFlight } from './update-gate.mjs'
|
|
106
108
|
import { saveSession, flushSession, deleteSession, loadAll, canResume, loadPtyId, savePtyId, loadNames, saveNames, appendDurableEvents, seedDurableEvents, readDurablePage, readDurableOldestSeq, hasDurableArchive, servesHistoryPage } from './session-store.mjs'
|
|
107
109
|
import { stampEvent, makeSeqCounter, maxSeq, seqable, capReplayEvents, chunkReplayEvents, boundEventForBroadcast, inlineImageBlocks, usageReportLine, codexUsageReportLine, buildRecapFromLog, RECAP_CAP, trimmedBeforeSeq, firstSeq } from './event-id.mjs'
|
|
@@ -1646,7 +1648,7 @@ function worktreeSnapshot(cwd) {
|
|
|
1646
1648
|
// relay STRUCTURED events. onEvent → broadcast `code-event` + print locally +
|
|
1647
1649
|
// persist to the host file; tool calls round-trip through the perm card; the
|
|
1648
1650
|
// rolling log replays to joiners and survives bridge restarts (session-store).
|
|
1649
|
-
function openStructured({ id, runtime = 'claude', model, effort, resume, log, commands, mode, spawnedBy, sideParent, sideTask, pendingSideContexts, rolePrompt, flowSessionId, flowTaskKey, cwd, managedWorktree, reviewSliceRoots, openedAt, defer, provider, carryRecap, lastUsage }) {
|
|
1651
|
+
function openStructured({ id, runtime = 'claude', model, effort, resume, log, commands, mode, spawnedBy, sideParent, sideTask, pendingSideContexts, rolePrompt, flowSessionId, flowTaskKey, flowRole, flowReviewTarget, flowReviewTargets, flowReviewRound, dispatchBaseSha, revertTarget, cwd, managedWorktree, reviewSliceRoots, openedAt, defer, provider, carryRecap, lastUsage }) {
|
|
1650
1652
|
if (sessions.has(id)) return
|
|
1651
1653
|
runtime = runtime === 'codex' ? 'codex' : 'claude'
|
|
1652
1654
|
// No explicit mode → a sensible default per runtime (see defaultModeForRuntime):
|
|
@@ -1656,6 +1658,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
1656
1658
|
// entry.mode), a flow lane, and spawn_terminal (inherits the parent's mode +
|
|
1657
1659
|
// its own bypass-escalation gate) all pass an explicit mode and skip this.
|
|
1658
1660
|
mode = STRUCTURED_MODES.has(mode) ? mode : defaultModeForRuntime(runtime)
|
|
1661
|
+
assertRuntimeModelCompatible({ runtime, provider, model })
|
|
1659
1662
|
// The lane's EFFECTIVE model — computed ONCE and used for BOTH the truthful display
|
|
1660
1663
|
// label (entry.model) and the SDK's `model` option below. These used to be computed
|
|
1661
1664
|
// separately: the label resolved the provider's configured model while the SDK got the
|
|
@@ -1663,14 +1666,15 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
1663
1666
|
// lane opened while focused on Fable therefore showed `glm-4.6` and sent
|
|
1664
1667
|
// `claude-fable-5` → 400 [1211][Unknown Model]. One value, one truth.
|
|
1665
1668
|
// undefined → omit the SDK option entirely and let the provider env/endpoint default.
|
|
1669
|
+
const codexModels = runtime === 'codex' ? readCodexModels() : []
|
|
1666
1670
|
const laneModel = runtime === 'codex'
|
|
1667
|
-
? (model || readCodexDefaultModel() || undefined)
|
|
1671
|
+
? resolveCodexModel(model || readCodexDefaultModel() || undefined, codexModels)
|
|
1668
1672
|
: effectiveLaneModel({ provider, model, configuredModel: resolveProviderEnv(provider)?.ANTHROPIC_MODEL })
|
|
1669
1673
|
// spawnedBy: set when this lane was Dispatched (spawn_terminal). Restored from the
|
|
1670
1674
|
// session store so the Ensemble flag survives a bridge restart (else a respin
|
|
1671
1675
|
// stripped it and the lane reverted to a plain tab — the t6 "no chip" bug).
|
|
1672
1676
|
effort = new Set(['low', 'medium', 'high', 'xhigh', 'max']).has(effort) ? effort : 'high'
|
|
1673
|
-
const entry = { cmd: runtime, runtime, kind: 'structured', log: Array.isArray(log) ? log.slice(-STRUCTURED_LOG_MAX) : [], pending: new Map(), session: null, recovered: false, commands: Array.isArray(commands) ? commands : undefined, mode, effort, models: runtime === 'codex' ?
|
|
1677
|
+
const entry = { cmd: runtime, runtime, kind: 'structured', log: Array.isArray(log) ? log.slice(-STRUCTURED_LOG_MAX) : [], pending: new Map(), session: null, recovered: false, commands: Array.isArray(commands) ? commands : undefined, mode, effort, models: runtime === 'codex' ? codexModels : undefined,
|
|
1674
1678
|
// model: truthful active-model label — now the SAME `laneModel` the SDK is given, so the
|
|
1675
1679
|
// chip cannot disagree with the wire. When this lane runs on a custom (non-anthropic)
|
|
1676
1680
|
// registered provider the SDK id is impersonated (see the onEvent guard below), so
|
|
@@ -1684,10 +1688,18 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
1684
1688
|
? (laneModel || null)
|
|
1685
1689
|
: (laneModel || providerNameMap()[provider] || provider),
|
|
1686
1690
|
provider: provider || null, spawnedBy: spawnedBy || undefined, sideParent: sideParent || undefined, sideTask: sideTask || undefined, pendingSideContexts: Array.isArray(pendingSideContexts) ? pendingSideContexts.filter(Boolean).slice(-4) : [], flowSessionId: flowSessionId || null, flowTaskKey: flowTaskKey || null, cwd: cwd || null, managedWorktree: managedWorktree || null,
|
|
1691
|
+
flowRole: flowRole || (flowSessionId ? (flowTaskKey ? ((flowReviewTargets?.length || reviewSliceRoots?.length) ? 'reviewer' : 'builder') : 'conductor') : null),
|
|
1692
|
+
flowReviewTarget: flowReviewTarget || null,
|
|
1693
|
+
flowReviewTargets: Array.isArray(flowReviewTargets) && flowReviewTargets.length ? flowReviewTargets.filter(Boolean) : (flowReviewTarget ? [flowReviewTarget] : []),
|
|
1694
|
+
flowReviewRound: Number.isInteger(flowReviewRound) && flowReviewRound >= 0 ? flowReviewRound : 0,
|
|
1695
|
+
dispatchBaseSha: dispatchBaseSha || null,
|
|
1696
|
+
revertTarget: revertTarget || null,
|
|
1697
|
+
cwd: cwd || null, managedWorktree: managedWorktree || null,
|
|
1687
1698
|
// Stable creation order — persisted so a bridge restart restores tabs in the SAME
|
|
1688
1699
|
// order (not readdir/filesystem order). Legacy recs (no openedAt) derive it from the
|
|
1689
1700
|
// first transcript event ts, so even the first post-fix restart is ordered right.
|
|
1690
1701
|
openedAt: openedAt || (Array.isArray(log) ? (log.find((e) => e?.ts)?.ts || 0) : 0) || Date.now() }
|
|
1702
|
+
entry.interruptedRecap = restoredTurnOpen(entry.log) ? buildRecapFromLog(entry.log, RECAP_CAP) : null
|
|
1691
1703
|
// Slice 3 — a permission card left unanswered past the grace window pushes
|
|
1692
1704
|
// "<lane> — needs you: <what>"; answering it anywhere retracts the banner
|
|
1693
1705
|
// everywhere. Worker/flow lanes are excluded at arm() time (isUserFacingLane).
|
|
@@ -1715,11 +1727,15 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
1715
1727
|
// MCP tool AND the FLOW_DONE sentinel-Write intercept: like the conductor's submit, the
|
|
1716
1728
|
// MCP tool is DEFERRED (needs ToolSearch, which hangs), so lanes signal done by Writing a
|
|
1717
1729
|
// FLOW_DONE file — a DIRECT tool — which the PreToolUse hook routes here. Returns a message.
|
|
1718
|
-
const markFlowDone = async () => {
|
|
1730
|
+
const markFlowDone = async ({ reviewPass = false } = {}) => {
|
|
1719
1731
|
if (!entry.flowSessionId || !entry.flowTaskKey) return 'Not a Flow lane — nothing to mark done.'
|
|
1732
|
+
if (entry.flowRole !== 'builder' && !(reviewPass && entry.flowRole === 'reviewer')) return 'This Flow role cannot mark a builder slice done.'
|
|
1720
1733
|
if (entry.flowDone) return 'This slice is already recorded as done.'
|
|
1721
1734
|
let commitSha = null
|
|
1722
1735
|
try { commitSha = execFileSync('git', ['-C', entry.cwd || process.cwd(), 'rev-parse', 'HEAD'], { encoding: 'utf8' }).trim() } catch { /* no commits yet */ }
|
|
1736
|
+
if (!reviewPass && entry.dispatchBaseSha && commitSha === entry.dispatchBaseSha) {
|
|
1737
|
+
return `Slice "${entry.flowTaskKey}" is not done: HEAD is still the dispatch base (${commitSha.slice(0, 8)}). Commit the verified implementation first.`
|
|
1738
|
+
}
|
|
1723
1739
|
let previewUrl = null
|
|
1724
1740
|
try { const pv = await startPreview({ dir: entry.cwd || process.cwd(), id: `lane:${entry.flowSessionId}:${id}` }); previewUrl = pv.url } catch { /* preview best-effort */ }
|
|
1725
1741
|
// S1 (context-offload) — digest THIS closed slice into the durable store so the next
|
|
@@ -1749,7 +1765,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
1749
1765
|
} catch (e) {
|
|
1750
1766
|
process.stderr.write(`\n ${A.dim}◆ flow digest skip ${entry.flowTaskKey} — ${e?.message || e}${A.rst}\n`)
|
|
1751
1767
|
}
|
|
1752
|
-
bcast('flow-task-done', { term: id, flowId: entry.flowSessionId, taskKey: entry.flowTaskKey, laneId: id, commitSha, previewUrl }, flowChannel)
|
|
1768
|
+
bcast('flow-task-done', { term: id, flowId: entry.flowSessionId, taskKey: entry.flowTaskKey, laneId: id, commitSha, previewUrl, ...(reviewPass ? { reviewAction: 'pass' } : {}) }, flowChannel)
|
|
1753
1769
|
process.stderr.write(`\n ${A.cyan}◆ flow slice done — ${entry.flowTaskKey} (${commitSha ? commitSha.slice(0, 8) : 'no commit'})${previewUrl ? ` · preview ${previewUrl}` : ''}${A.rst}\n`)
|
|
1754
1770
|
entry.flowDone = true // FL-B3 — retire immediately so it drops from the ≤8 lane cap
|
|
1755
1771
|
setTimeout(() => { try { endStructured(id) } catch { /* already gone */ } }, 2500)
|
|
@@ -1761,14 +1777,22 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
1761
1777
|
// never reverted). On FAIL → broadcast flow-revert for the reviewed slice (the client flips
|
|
1762
1778
|
// it to pending → the next wave rebuilds it). Either way the review lane itself is done.
|
|
1763
1779
|
const onReviewVerdict = async (raw) => {
|
|
1764
|
-
if (!entry.flowSessionId || !entry.flowTaskKey) return { ok: false, message: 'Not a Flow review lane.' }
|
|
1780
|
+
if (!entry.flowSessionId || !entry.flowTaskKey || entry.flowRole !== 'reviewer') return { ok: false, message: 'Not a Flow review lane.' }
|
|
1765
1781
|
let v, target = null
|
|
1766
1782
|
try {
|
|
1767
1783
|
v = parseReviewVerdict(raw)
|
|
1768
1784
|
const o = typeof raw === 'string' ? JSON.parse(String(raw).replace(/```(?:json)?|```/g, '').trim()) : raw
|
|
1769
|
-
|
|
1785
|
+
const requested = o && (o.taskKey || o.target)
|
|
1786
|
+
const allowed = entry.flowReviewTargets
|
|
1787
|
+
if (!allowed.length) throw new Error('review lane has no authorized target')
|
|
1788
|
+
if (entry.runtime === 'codex' && allowed.length !== 1) throw new Error('Codex review lane must have exactly one authorized target')
|
|
1789
|
+
if (requested && !allowed.includes(requested)) throw new Error(`taskKey ${JSON.stringify(requested)} is outside this review lane`)
|
|
1790
|
+
target = requested || (allowed.length === 1 ? allowed[0] : null)
|
|
1791
|
+
if (!target) throw new Error('taskKey is required when reviewing more than one slice')
|
|
1770
1792
|
} catch (e) {
|
|
1771
|
-
return { ok: false, message:
|
|
1793
|
+
return { ok: false, message: entry.runtime === 'codex'
|
|
1794
|
+
? `Review verdict REJECTED: ${e?.message || e}. Re-call submit_flow_review with a valid authorized taskKey.`
|
|
1795
|
+
: `Review verdict REJECTED: ${e?.message || e}. Re-Write FLOW_REVIEW.json with {"pass":<boolean>,"reasons":["<specific finding>"],"taskKey":"<the slice you reviewed>"}.` }
|
|
1772
1796
|
}
|
|
1773
1797
|
// E1 A1/A2 — the BOUNDED reviewer loop, live side. Each FLOW_REVIEW.json write is ONE
|
|
1774
1798
|
// hunt round; the governor decides continue-vs-stop from the round count + the lane's
|
|
@@ -1779,6 +1803,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
1779
1803
|
// for live-room verification (bridge is inert until published); the decision logic is
|
|
1780
1804
|
// unit-proven in flow-review.loop.test.mjs + flow-review-reflect.test.mjs.
|
|
1781
1805
|
const round = (entry.flowReviewRound = (entry.flowReviewRound || 0) + 1)
|
|
1806
|
+
entry.flush?.()
|
|
1782
1807
|
const b = flowBudgets.get(entry.flowSessionId) || null
|
|
1783
1808
|
const decision = reviewReflectionDecision({
|
|
1784
1809
|
round,
|
|
@@ -1794,7 +1819,9 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
1794
1819
|
const next = round + 1
|
|
1795
1820
|
try {
|
|
1796
1821
|
entry.session?.sendTurn(
|
|
1797
|
-
|
|
1822
|
+
entry.runtime === 'codex'
|
|
1823
|
+
? `[Flow review — round ${next}/${REVIEW_DEFAULTS.maxRounds}] The happy path held, but you have NOT reported your checks exhausted. Dig one more round, then call submit_flow_review again — pass:false with specific evidence if you break it, or pass:true AND exhausted:true only when nothing remains to check.`
|
|
1824
|
+
: `[Flow review — round ${next}/${REVIEW_DEFAULTS.maxRounds}] The happy path held, but you have NOT reported your checks exhausted and you are under both the round and budget ceilings. Dig one more round: hunt the edge cases, the reload, the second click, concurrent use, the error path — the places the builder didn't. Then re-Write FLOW_REVIEW.json — pass:false with a specific reason if you break it, or pass:true AND exhausted:true if you genuinely have nothing left to check.`,
|
|
1798
1825
|
)
|
|
1799
1826
|
} catch { /* lane may have closed mid-verdict */ }
|
|
1800
1827
|
process.stderr.write(`\n ${A.dim}◆ review round ${round} inconclusive — digging again (${next}/${REVIEW_DEFAULTS.maxRounds}) on ${target || entry.flowTaskKey}${A.rst}\n`)
|
|
@@ -1804,11 +1831,24 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
1804
1831
|
// Terminal outcome. reject → revert the reviewed slice; surface → hand the inconclusive
|
|
1805
1832
|
// result to the pair WITHOUT reverting (no failure was reproduced); pass → accept.
|
|
1806
1833
|
if (decision.action === 'reject' && target) {
|
|
1807
|
-
|
|
1834
|
+
const reverted = await handleFlowRevert({ term: id, flowId: entry.flowSessionId, taskKey: target, reviewTaskKey: entry.flowTaskKey })
|
|
1835
|
+
if (!reverted.ok) persistAgentEvent({ kind: 'needs-input', term: id, termName: termNames[id] || null, summary: `Flow review could not revert ${target}: ${reverted.error || reverted.ignored || 'unknown error'}` })
|
|
1808
1836
|
process.stderr.write(`\n ${A.yel}◆ review REJECT (round ${round}) — reverting ${target}: ${v.reasons.join('; ').slice(0, 120)}${A.rst}\n`)
|
|
1809
1837
|
} else {
|
|
1810
1838
|
process.stderr.write(`\n ${A.cyan}◆ review ${decision.action.toUpperCase()} (round ${round}) — ${target || entry.flowTaskKey}${A.rst}\n`)
|
|
1811
1839
|
}
|
|
1840
|
+
// Durable room-visible verdict. `persistAgentEvent` below is intentionally
|
|
1841
|
+
// filtered from the transcript (push/unread transport only), while this control
|
|
1842
|
+
// row is archived with the reviewer lane and broadcast to both room members.
|
|
1843
|
+
// In particular, a held review must show its exact findings instead of leaving
|
|
1844
|
+
// the red task state unexplained after the ephemeral Flow broadcast is gone.
|
|
1845
|
+
const visibleReview = {
|
|
1846
|
+
kind: 'control',
|
|
1847
|
+
text: `Flow review ${decision.action.toUpperCase()} — ${target || entry.flowTaskKey}: ${v.reasons.join('; ') || decision.reason}`,
|
|
1848
|
+
}
|
|
1849
|
+
pushLog(entry, visibleReview)
|
|
1850
|
+
bcast('code-event', { term: id, evt: visibleReview })
|
|
1851
|
+
entry.flush?.()
|
|
1812
1852
|
// P3 (pair co-adjudication) — surface the TERMINAL verdict into the ROOM as a
|
|
1813
1853
|
// challengeable prompt, ALONGSIDE the gate action above. A solo Bugbot's verdict is
|
|
1814
1854
|
// final; ours is a prompt for the two humans + Pool to argue. Additive: an extra
|
|
@@ -1828,13 +1868,31 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
1828
1868
|
findings: v.reasons,
|
|
1829
1869
|
}),
|
|
1830
1870
|
}, flowChannel)
|
|
1831
|
-
|
|
1871
|
+
let doneMsg = ''
|
|
1872
|
+
if (decision.action === 'pass') {
|
|
1873
|
+
doneMsg = await markFlowDone({ reviewPass: true })
|
|
1874
|
+
} else {
|
|
1875
|
+
if (decision.action === 'surface') {
|
|
1876
|
+
bcast('flow-review-held', { term: id, flowId: entry.flowSessionId, reviewTaskKey: entry.flowTaskKey, taskKey: target, reasons: v.reasons }, flowChannel)
|
|
1877
|
+
persistAgentEvent({ kind: 'needs-input', term: id, termName: termNames[id] || null, summary: `Flow review held ${target || entry.flowTaskKey}: ${v.reasons.join('; ') || decision.reason}` })
|
|
1878
|
+
}
|
|
1879
|
+
// A reject must rebuild the target and then run a fresh reviewer. Retire that
|
|
1880
|
+
// reviewer now. A surfaced inconclusive review is different: keep its lane
|
|
1881
|
+
// alive with the durable control row above so either room member can open the
|
|
1882
|
+
// held review, read the exact findings, and continue/adjudicate after reload.
|
|
1883
|
+
// Neither outcome emits task-done or permits assembly.
|
|
1884
|
+
if (decision.action === 'reject') {
|
|
1885
|
+
entry.flowDone = true
|
|
1886
|
+
entry.flush?.()
|
|
1887
|
+
setTimeout(() => { try { endStructured(id) } catch { /* already gone */ } }, 2500)
|
|
1888
|
+
}
|
|
1889
|
+
}
|
|
1832
1890
|
const label = decision.action === 'reject'
|
|
1833
1891
|
? `REJECT (round ${round}) — reverting ${target || '(no target)'} (${v.reasons.join('; ').slice(0, 140)})`
|
|
1834
1892
|
: decision.action === 'surface'
|
|
1835
1893
|
? `SURFACED to the pair after ${round} round${round === 1 ? '' : 's'} — ${decision.reason}`
|
|
1836
1894
|
: `PASS (round ${round})`
|
|
1837
|
-
return { ok: true, message: `Review verdict recorded: ${label}
|
|
1895
|
+
return { ok: true, message: `Review verdict recorded: ${label}.${doneMsg ? ` ${doneMsg}` : ''}` }
|
|
1838
1896
|
}
|
|
1839
1897
|
// Identity for the durable archive — pushLog appends every new transcript event to
|
|
1840
1898
|
// <room>/<id>.events.jsonl keyed off these. Seed the archive once from the restored
|
|
@@ -1880,7 +1938,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
1880
1938
|
// restart. Without this, sessionData omitted it → on restart the resumed session
|
|
1881
1939
|
// re-launched on the host default (Opus) regardless of the last switch, and the
|
|
1882
1940
|
// switch looked like it "never changed the model" (Max 2026-07-02). Restored below.
|
|
1883
|
-
const sessionData = () => ({ sessionId: entry.session?.sessionId || null, runtime: entry.runtime, log: entry.log, commands: entry.commands, mode: entry.mode, effort: entry.effort, model: entry.model || null, provider: entry.provider || null, spawnedBy: entry.spawnedBy, sideParent: entry.sideParent, sideTask: entry.sideTask, pendingSideContexts: entry.pendingSideContexts, flowSessionId: entry.flowSessionId, flowTaskKey: entry.flowTaskKey, cwd: entry.cwd, managedWorktree: entry.managedWorktree, rolePrompt: entry.rolePrompt, flowReviewTarget: entry.flowReviewTarget || null, revertTarget: entry.revertTarget || null, reviewSliceRoots: entry.reviewSliceRoots || [], openedAt: entry.openedAt || null, lastUsage: entry.lastUsage || null, carryRecap: entry.pendingRecap || null })
|
|
1941
|
+
const sessionData = () => ({ sessionId: entry.session?.sessionId || null, runtime: entry.runtime, log: entry.log, commands: entry.commands, mode: entry.mode, effort: entry.effort, model: entry.model || null, provider: entry.provider || null, spawnedBy: entry.spawnedBy, sideParent: entry.sideParent, sideTask: entry.sideTask, pendingSideContexts: entry.pendingSideContexts, flowSessionId: entry.flowSessionId, flowTaskKey: entry.flowTaskKey, flowRole: entry.flowRole || null, flowReviewTargets: entry.flowReviewTargets || [], flowReviewRound: entry.flowReviewRound || 0, dispatchBaseSha: entry.dispatchBaseSha || null, cwd: entry.cwd, managedWorktree: entry.managedWorktree, rolePrompt: entry.rolePrompt, flowReviewTarget: entry.flowReviewTarget || null, revertTarget: entry.revertTarget || null, reviewSliceRoots: entry.reviewSliceRoots || [], openedAt: entry.openedAt || null, lastUsage: entry.lastUsage || null, carryRecap: entry.pendingRecap || null })
|
|
1884
1942
|
const persist = () => saveSession(room, id, sessionData())
|
|
1885
1943
|
// Synchronous flush of this session's record. Used on open (so a brand-new session
|
|
1886
1944
|
// has a file under its id BEFORE its first event — surviving a restart inside the
|
|
@@ -1962,7 +2020,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
1962
2020
|
// OWN receipt card — no text reaches the other agent without a person at BOTH
|
|
1963
2021
|
// ends. crossRoomPostDecision caps it at one per turn and blocks a room reached
|
|
1964
2022
|
// via cross-room post from posting onward. Kill-switch: TP_PAIRBUS_OFF.
|
|
1965
|
-
tool(
|
|
2023
|
+
...(!entry.flowRole ? [tool(
|
|
1966
2024
|
'post_to_session',
|
|
1967
2025
|
'Hand a task or message to an agent in ANOTHER of your ThinkPool Code sessions — your own room on this machine, or your partner\'s room reachable through the pair (a room code from list_sessions). Pass `session` (the room code), `text` (what to send), and optionally `terminal` (which agent lane in that room). A person in YOUR room approves sending AND a person in the TARGET room approves receiving before it is delivered. An agent reached this way cannot post onward to a third room. Read the target with read_session first, and use this sparingly — only when the people clearly want the rooms to coordinate.',
|
|
1968
2026
|
{
|
|
@@ -1987,13 +2045,13 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
1987
2045
|
if (res?.error) return okText(res.error)
|
|
1988
2046
|
return okText(res?.ok ? `Delivered to room ${target}${res.ref ? ` (lane ${res.ref})` : ''} — it will respond in its own room; check back with read_session.` : `Room ${target} did not accept the message.`)
|
|
1989
2047
|
},
|
|
1990
|
-
),
|
|
2048
|
+
)] : []),
|
|
1991
2049
|
// Tier C — directed cross-lane action. WRITES into a sibling AGENT lane.
|
|
1992
2050
|
// The PreToolUse gate (claude-session.mjs) already enforced: precheck passed
|
|
1993
2051
|
// + a human approved the card. This handler routes only; it re-asserts the
|
|
1994
2052
|
// bounds defensively. Targets AGENT terminals only (not human shells —
|
|
1995
2053
|
// injecting keystrokes into someone's shell is out of scope for v1).
|
|
1996
|
-
tool(
|
|
2054
|
+
...(!entry.flowRole ? [tool(
|
|
1997
2055
|
'post_to_terminal',
|
|
1998
2056
|
'Send a message or task to ANOTHER AGENT terminal in this ThinkPool Code room. Use `terminal` (a ref/id/command from read_terminal) and `text` (what to send). A person in the room must approve before it is delivered, and an agent reached via a cross-post cannot post onward. Targets agent terminals only, not plain shells. Use read_terminal first to understand the sibling.',
|
|
1999
2057
|
{
|
|
@@ -2030,7 +2088,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
2030
2088
|
try { te.session.sendTurn(msg) } catch { return okText(`Could not deliver to terminal ${target.ref} — it may have just closed.`) }
|
|
2031
2089
|
return okText(`Delivered to terminal ${target.ref} (${target.cmd}). It will respond in its own lane; check back with read_terminal.`)
|
|
2032
2090
|
},
|
|
2033
|
-
),
|
|
2091
|
+
)] : []),
|
|
2034
2092
|
// Tier C+ — SPAWN a fresh agent lane the caller OWNS. The motivating bug:
|
|
2035
2093
|
// an agent fanned a research task into SIBLINGS that were already busy (a
|
|
2036
2094
|
// release lane, a Q&A lane) because it had no way to make its own lanes.
|
|
@@ -2041,13 +2099,14 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
2041
2099
|
// cap, TP_SPAWN_OFF kill-switch). A person's lane (hop 0) may spawn conductors
|
|
2042
2100
|
// (hop 1) which may spawn workers (hop 2); hop 2 can never spawn onward — the
|
|
2043
2101
|
// fork-bomb breaker, now one level deeper (2026-07-10 cascade-spawn-depth spec).
|
|
2044
|
-
tool(
|
|
2102
|
+
...(!entry.flowRole ? [tool(
|
|
2045
2103
|
'spawn_terminal',
|
|
2046
|
-
'Open a NEW agent terminal (Claude or Codex) in this ThinkPool Code room — YOUR lane to use, not a sibling that is already working. The runtime defaults to inheriting this lane, or pass `runtime` to dispatch across runtimes. Choose `model` explicitly from advertised metadata when practical: for Codex use Luna/Mini for mechanical work, Terra for ordinary feature/fix work, and Sol only for repeated-failure, architecture/conductor, or adversarial final-review work; never inherit Sol by default, and omit `model` for runtime fallback when the named tier is unavailable. Give the lane an initial `task`, collect its result
|
|
2104
|
+
'Open a NEW agent terminal (Claude or Codex) in this ThinkPool Code room — YOUR lane to use, not a sibling that is already working. The runtime defaults to inheriting this lane, or pass `runtime` to dispatch across runtimes. Choose `model` explicitly from advertised metadata when practical: for Codex use Luna/Mini for mechanical work, Terra for ordinary feature/fix work, and Sol only for repeated-failure, architecture/conductor, or adversarial final-review work; never inherit Sol by default, and omit `model` for runtime fallback when the named tier is unavailable. For cascades, pass sliceType=scaffold for mechanical work, feature/fix for builders, and review for adversarial verification (review never routes to the cheap tier); an explicit model overrides slice tiering. Omit sliceType to preserve ordinary model inheritance/default behavior. Give the lane an initial `task`, collect its result with read_terminal, then ALWAYS close_terminal it — a finished lane left open holds a slot against the room cap and blocks the next fan-out. By default the new lane INHERITS your current permission mode. ALWAYS prefer spawning a fresh lane over post_to_terminal into one that is already busy. Bounded by cascade depth, the room terminal cap, and per-turn limits.',
|
|
2047
2105
|
{
|
|
2048
2106
|
name: z.string().max(80).optional().describe('a short label for the new lane so the room (and you) can find it, e.g. "Research: topologies"'),
|
|
2049
2107
|
task: z.string().optional().describe('an initial task to hand the new lane immediately; omit to open it idle'),
|
|
2050
2108
|
model: z.string().optional().describe('optional model for the chosen runtime, e.g. opus / sonnet / gpt-5.6-sol'),
|
|
2109
|
+
sliceType: z.enum(['scaffold', 'feature', 'fix', 'review']).optional().describe('optional cascade slice tier; omitted preserves normal inheritance/default, explicit model wins'),
|
|
2051
2110
|
runtime: z.enum(['claude', 'codex']).optional().describe('agent runtime for the new lane; defaults to inheriting this lane'),
|
|
2052
2111
|
provider: z.string().optional().describe('optional registered LLM provider id to run this lane on (from the account\'s provider registry); omit for the default Claude/Anthropic path'),
|
|
2053
2112
|
mode: z.enum(['default', 'acceptEdits', 'bypassPermissions', 'plan']).optional().describe('permission mode for the new lane; defaults to inheriting YOUR current mode. Raising a lane to bypassPermissions from a non-bypass lane asks the room to confirm once.'),
|
|
@@ -2056,6 +2115,16 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
2056
2115
|
const okText = (t) => ({ content: [{ type: 'text', text: t }] })
|
|
2057
2116
|
const childRuntime = args?.runtime || entry.runtime || 'claude'
|
|
2058
2117
|
if (childRuntime === 'codex' && args?.provider) return okText('A Codex lane uses its Codex/OpenAI login in v1; custom bridge providers are not wired to Codex yet. Omit provider or spawn a Claude lane for that provider.')
|
|
2118
|
+
const childCatalog = childRuntime === 'codex' ? readCodexModels() : []
|
|
2119
|
+
if (childRuntime === 'codex' && args?.model && !modelCatalogValues(childCatalog).has(args.model)) {
|
|
2120
|
+
return okText(`Could not open a Codex lane on ${JSON.stringify(args.model)} — that model is not in this host's visible Codex catalog.`)
|
|
2121
|
+
}
|
|
2122
|
+
if (childRuntime === 'claude' && !args?.provider && args?.model && /^gpt-/i.test(args.model)) {
|
|
2123
|
+
return okText(`Could not open a Claude lane on Codex model ${JSON.stringify(args.model)}. Choose runtime="codex" or a Claude model.`)
|
|
2124
|
+
}
|
|
2125
|
+
const childModel = args?.model || (args?.sliceType
|
|
2126
|
+
? spawnedLaneModelFor({ sliceType: args.sliceType, runtime: childRuntime, catalog: childCatalog })
|
|
2127
|
+
: undefined)
|
|
2059
2128
|
const now = Date.now()
|
|
2060
2129
|
const gate = spawnDecision({
|
|
2061
2130
|
hop: entry.hop || 0,
|
|
@@ -2090,7 +2159,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
2090
2159
|
// plan-mode lane would stall on ExitPlanMode cards. An explicit args.mode still wins;
|
|
2091
2160
|
// otherwise inherit, but fall plan → default (an autonomous worker doesn't plan-gate).
|
|
2092
2161
|
const childMode = args?.mode || (entry.mode === 'plan' ? 'default' : entry.mode)
|
|
2093
|
-
openStructured({ id: newId, runtime: childRuntime, model:
|
|
2162
|
+
openStructured({ id: newId, runtime: childRuntime, model: childModel, provider: args?.provider, mode: childMode })
|
|
2094
2163
|
const ne = sessions.get(newId)
|
|
2095
2164
|
if (!ne) { if (args?.name) { delete termNames[newId]; saveNames(room, termNames) } return okText('Could not open a new lane — the terminal cap may have just been reached. Close one and retry.') }
|
|
2096
2165
|
ne.spawnedBy = id // ownership: only the spawner may close_terminal it
|
|
@@ -2109,11 +2178,11 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
2109
2178
|
}
|
|
2110
2179
|
return okText(`Opened idle agent lane ${newRef}${args?.name ? ` ("${args.name}")` : ''}. Hand it work with post_to_terminal, or a person can type into it.`)
|
|
2111
2180
|
},
|
|
2112
|
-
),
|
|
2181
|
+
)] : []),
|
|
2113
2182
|
// Research lane — run a REAL multi-source search + adversarial verification and
|
|
2114
2183
|
// return sourced, verdict-tagged findings. Calls the verified web backend
|
|
2115
2184
|
// (/api/research-run) as the room owner; plan-gated + budget-capped server-side.
|
|
2116
|
-
tool(
|
|
2185
|
+
...(!entry.flowRole ? [tool(
|
|
2117
2186
|
'research',
|
|
2118
2187
|
'Run a REAL multi-source web research + verification on a factual question and surface sourced, verdict-tagged findings to the room. Use when the people would genuinely benefit from looking something up or settling an external-fact question — pricing, "is X still maintained / deprecated", "is that benchmark real", a debate over current facts. OFFER it first in plain language ("want me to spawn a research lane on that?") and only call it once they agree — it spends (plan-gated: Free 5 / Plus 100 runs per month) and takes ~1 minute. It searches the web, reads sources, and returns each claim marked HELD or REJECTED with citations. Present the findings clearly and let both people weigh the sources.',
|
|
2119
2188
|
{
|
|
@@ -2144,13 +2213,13 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
2144
2213
|
`\n\nPresent these to the room and let both people weigh the sources — flag which held claims rest on a source they might not trust.`
|
|
2145
2214
|
)
|
|
2146
2215
|
},
|
|
2147
|
-
),
|
|
2216
|
+
)] : []),
|
|
2148
2217
|
// Tier C+ — CLOSE a lane the caller spawned. Restricted to self-spawned agent
|
|
2149
2218
|
// lanes (spawnedBy === this id): an agent must never be able to kill a sibling
|
|
2150
2219
|
// someone else is working in, nor the host/attached terminal. The fan-out
|
|
2151
2220
|
// cleanup half of spawn_terminal. Routes through endStructured (same path as
|
|
2152
2221
|
// the web's code-close), which no-ops if the id isn't a live session.
|
|
2153
|
-
tool(
|
|
2222
|
+
...(!entry.flowRole ? [tool(
|
|
2154
2223
|
'close_terminal',
|
|
2155
2224
|
'Close an agent lane that YOU opened with spawn_terminal (identified by ref/id/name). You can only close lanes you spawned yourself — never a sibling someone else is working in, and never the main terminal. Use it to clean up after a fan-out once you have collected the results with read_terminal.',
|
|
2156
2225
|
{
|
|
@@ -2167,38 +2236,47 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
2167
2236
|
endStructured(target.id)
|
|
2168
2237
|
return okText(`Closed agent lane ${target.ref}${target.name ? ` ("${target.name}")` : ''}.`)
|
|
2169
2238
|
},
|
|
2170
|
-
),
|
|
2239
|
+
)] : []),
|
|
2171
2240
|
// Flow lane → mark this slice done (lane-done return path). Reads the lane's
|
|
2172
2241
|
// worktree HEAD as commit_sha (the atomic-revert target) + broadcasts
|
|
2173
2242
|
// flow-task-done → the room flips the task done + dispatches the next wave
|
|
2174
2243
|
// (slices whose deps just got satisfied).
|
|
2175
|
-
tool(
|
|
2244
|
+
...(entry.flowRole === 'builder' ? [tool(
|
|
2176
2245
|
'mark_flow_done',
|
|
2177
2246
|
"ThinkPool Flow ONLY — call this ONCE when your slice is built, RUNS, and meets its acceptance criteria. Records your latest commit as the slice's atomic-revert target and signals the room that this slice is done (which unblocks slices that depended on you). No arguments: your commit is read from your worktree. Do not call before your slice actually runs + meets acceptance.",
|
|
2178
2247
|
{},
|
|
2179
2248
|
async () => ({ content: [{ type: 'text', text: await markFlowDone() }] }),
|
|
2180
|
-
),
|
|
2249
|
+
)] : []),
|
|
2181
2250
|
// FL-B1 — the conductor SUBMITS its decomposition through THIS tool, not the built-in
|
|
2182
2251
|
// ExitPlanMode. In the current SDK, ExitPlanMode is a deferred tool the conductor must
|
|
2183
2252
|
// ToolSearch for — and it hangs there, wedging every flow in `planning`. A dedicated MCP
|
|
2184
2253
|
// tool removes all dependence on plan-mode/ExitPlanMode/deferred-tool discovery: validate
|
|
2185
2254
|
// the task-graph here, broadcast flow-plan (the same path the old interception used), and
|
|
2186
2255
|
// reject malformed plans back to the conductor so it re-emits.
|
|
2187
|
-
tool(
|
|
2256
|
+
...(entry.flowRole === 'conductor' ? [tool(
|
|
2188
2257
|
'submit_flow_plan',
|
|
2189
2258
|
'ThinkPool Flow CONDUCTOR ONLY — submit your decomposition for human approval. Call this ONCE when your task-graph is ready. Pass it as `plan`: a JSON string {"summary":"<one line>","tasks":[{"key":"<kebab>","title":"…","scope":"…","acceptance":"…","deps":["<key>"],"sliceType":"feature|scaffold|review|fix"}]}. The room validates it is an acyclic DAG, persists the tasks, and shows the approval card. This is how a conductor FINISHES — do NOT use ExitPlanMode, do NOT write a plan file.',
|
|
2190
2259
|
{ plan: z.string().describe('the task-graph as a JSON string (the {summary, tasks:[…]} object)') },
|
|
2191
2260
|
async (args) => {
|
|
2192
2261
|
const okText = (t) => ({ content: [{ type: 'text', text: t }] })
|
|
2193
|
-
if (!entry.flowSessionId) return okText('Not a Flow conductor — there is no flow to submit a plan for.')
|
|
2262
|
+
if (!entry.flowSessionId || entry.flowRole !== 'conductor') return okText('Not a Flow conductor — there is no flow to submit a plan for.')
|
|
2194
2263
|
let norm
|
|
2195
|
-
try { norm = normalizePlanOutput(args?.plan || '') }
|
|
2264
|
+
try { norm = validatePlanForRuntime(normalizePlanOutput(args?.plan || ''), entry.runtime) }
|
|
2196
2265
|
catch (e) { return okText(`Plan REJECTED: ${e?.message || e}. Re-call submit_flow_plan with valid JSON — a non-empty "tasks" array, every "deps" entry referencing an existing task "key", and no cycles.`) }
|
|
2197
2266
|
bcast('flow-plan', { term: id, flowId: entry.flowSessionId, plan: JSON.stringify(norm) }, flowChannel)
|
|
2198
2267
|
process.stderr.write(`\n ${A.mag}◆ flow plan submitted — flow ${String(entry.flowSessionId).slice(0, 8)} (${norm.tasks.length} slice${norm.tasks.length === 1 ? '' : 's'}).${A.rst}\n`)
|
|
2199
2268
|
return okText(`Plan submitted (${norm.tasks.length} slice${norm.tasks.length === 1 ? '' : 's'}) — the room is showing the human an approval card. You are DONE; stop here and wait for approval (lane dispatch is the room's job).`)
|
|
2200
2269
|
},
|
|
2201
|
-
),
|
|
2270
|
+
)] : []),
|
|
2271
|
+
...(entry.flowRole === 'reviewer' ? [tool(
|
|
2272
|
+
'submit_flow_review',
|
|
2273
|
+
'ThinkPool Flow REVIEWER ONLY — submit one bounded adversarial review round. Pass `verdict` as JSON: {"pass":boolean,"reasons":["specific evidence"],"taskKey":"authorized reviewed task","exhausted":boolean}. This is the only review completion path; do not call mark_flow_done and do not write FLOW_REVIEW.json.',
|
|
2274
|
+
{ verdict: z.string().describe('the structured review verdict as a JSON string') },
|
|
2275
|
+
async (args) => {
|
|
2276
|
+
const result = await onReviewVerdict(args?.verdict || '')
|
|
2277
|
+
return { content: [{ type: 'text', text: result.message }] }
|
|
2278
|
+
},
|
|
2279
|
+
)] : []),
|
|
2202
2280
|
],
|
|
2203
2281
|
})
|
|
2204
2282
|
// S5 (slice 1b) — the review-lane write-block, wired into the live PreToolUse path.
|
|
@@ -2265,15 +2343,15 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
2265
2343
|
// A Flow CONDUCTOR (flowSessionId set, no flowTaskKey) must DECOMPOSE, not explore via
|
|
2266
2344
|
// a fan-out of Task/Explore subagents, and must not build. Hard-block Task/Bash/Write for
|
|
2267
2345
|
// conductors so it reads with Read/Grep/Glob and submits via the FLOW_PLAN.json sentinel.
|
|
2268
|
-
blockSubagents:
|
|
2346
|
+
blockSubagents: entry.flowRole === 'conductor',
|
|
2269
2347
|
// FL-B1 — the conductor submits its task-graph by Writing FLOW_PLAN.json; the PreToolUse
|
|
2270
2348
|
// hook intercepts that Write and calls this. Validate the DAG and broadcast flow-plan (the
|
|
2271
2349
|
// client persists it + shows the approval card); a malformed plan comes straight back to
|
|
2272
2350
|
// the conductor as the tool result so it re-emits. Returns { ok, message }.
|
|
2273
2351
|
onSubmitPlan: (planText) => {
|
|
2274
|
-
if (!entry.flowSessionId) return { ok: false, message: 'Not a Flow conductor — nothing to submit.' }
|
|
2352
|
+
if (!entry.flowSessionId || entry.flowRole !== 'conductor') return { ok: false, message: 'Not a Flow conductor — nothing to submit.' }
|
|
2275
2353
|
let norm
|
|
2276
|
-
try { norm = normalizePlanOutput(planText || '') }
|
|
2354
|
+
try { norm = validatePlanForRuntime(normalizePlanOutput(planText || ''), entry.runtime) }
|
|
2277
2355
|
catch (e) { return { ok: false, message: `Plan REJECTED: ${e?.message || e}. Re-Write FLOW_PLAN.json with valid JSON — a non-empty "tasks" array, every "deps" entry referencing an existing task "key", and no cycles.` } }
|
|
2278
2356
|
bcast('flow-plan', { term: id, flowId: entry.flowSessionId, plan: JSON.stringify(norm) }, flowChannel)
|
|
2279
2357
|
process.stderr.write(`\n ${A.mag}◆ flow plan submitted — flow ${String(entry.flowSessionId).slice(0, 8)} (${norm.tasks.length} slice${norm.tasks.length === 1 ? '' : 's'}).${A.rst}\n`)
|
|
@@ -2281,9 +2359,9 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
2281
2359
|
},
|
|
2282
2360
|
// FL-B1 (lane side) — a flow lane signals slice-done by Writing FLOW_DONE; the hook
|
|
2283
2361
|
// routes that Write here (mark_flow_done MCP tool is deferred → ToolSearch → hangs).
|
|
2284
|
-
onLaneDone:
|
|
2362
|
+
onLaneDone: entry.flowRole === 'builder' ? (async () => ({ ok: true, message: await markFlowDone() })) : null,
|
|
2285
2363
|
// FL-B4 — a review lane records its verdict via FLOW_REVIEW.json (revert-on-fail).
|
|
2286
|
-
onReviewVerdict:
|
|
2364
|
+
onReviewVerdict: entry.flowRole === 'reviewer' ? onReviewVerdict : null,
|
|
2287
2365
|
mcpServers: { thinkpool: peekServer },
|
|
2288
2366
|
// Tier C precheck — the PreToolUse gate calls this BEFORE raising a card, so a
|
|
2289
2367
|
// disabled/looping/over-cap post never bothers a person. Closes over `entry`.
|
|
@@ -2316,15 +2394,32 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
2316
2394
|
onEvent: (evt) => {
|
|
2317
2395
|
// Self-heal a stale resume — the saved SDK session expired. Reopen fresh,
|
|
2318
2396
|
// keeping the transcript (scrollback survives; live context is gone).
|
|
2319
|
-
if (resume &&
|
|
2320
|
-
entry.
|
|
2321
|
-
|
|
2322
|
-
|
|
2323
|
-
|
|
2324
|
-
|
|
2325
|
-
|
|
2326
|
-
|
|
2327
|
-
|
|
2397
|
+
if (resume && evt.kind === 'error') {
|
|
2398
|
+
const recoveryRecap = entry.interruptedRecap || buildRecapFromLog(entry.log, RECAP_CAP)
|
|
2399
|
+
const recovery = recoverMissingResumeOnce(entry, {
|
|
2400
|
+
message: evt.message,
|
|
2401
|
+
recap: recoveryRecap,
|
|
2402
|
+
reopen: (carryRecap) => {
|
|
2403
|
+
try { entry.mockupWatcher?.close() } catch { /* noop */ }
|
|
2404
|
+
sessions.delete(id)
|
|
2405
|
+
openStructured({
|
|
2406
|
+
id, runtime: entry.runtime, model: entry.model || model, effort: entry.effort,
|
|
2407
|
+
provider: entry.provider, log: entry.log, commands: entry.commands, mode: entry.mode,
|
|
2408
|
+
spawnedBy: entry.spawnedBy, sideParent: entry.sideParent, sideTask: entry.sideTask, pendingSideContexts: entry.pendingSideContexts, flowSessionId: entry.flowSessionId,
|
|
2409
|
+
flowTaskKey: entry.flowTaskKey, flowRole: entry.flowRole, flowReviewTarget: entry.flowReviewTarget,
|
|
2410
|
+
flowReviewTargets: entry.flowReviewTargets, flowReviewRound: entry.flowReviewRound,
|
|
2411
|
+
dispatchBaseSha: entry.dispatchBaseSha, revertTarget: entry.revertTarget, cwd: entry.cwd,
|
|
2412
|
+
managedWorktree: entry.managedWorktree, rolePrompt: entry.rolePrompt,
|
|
2413
|
+
reviewSliceRoots: entry.reviewSliceRoots, openedAt: entry.openedAt,
|
|
2414
|
+
lastUsage: entry.lastUsage, carryRecap,
|
|
2415
|
+
})
|
|
2416
|
+
return sessions.get(id) || null
|
|
2417
|
+
},
|
|
2418
|
+
})
|
|
2419
|
+
if (recovery.recovered) {
|
|
2420
|
+
process.stderr.write(`\n ◆ saved session expired — started fresh with transcript recap (${recovery.delivery}).\n`)
|
|
2421
|
+
return
|
|
2422
|
+
}
|
|
2328
2423
|
}
|
|
2329
2424
|
// Stamp a wall-clock ts AND a stable cid on every transcript event before
|
|
2330
2425
|
// it's logged + broadcast. ts: lets the web client sort agent turns
|
|
@@ -2398,7 +2493,7 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
2398
2493
|
// explicit signal (FLOW_DONE / mark_flow_done), which left the slice stuck 'dispatched'
|
|
2399
2494
|
// forever → the flow never assembled. Auto-record on turn-end (idempotent — markFlowDone
|
|
2400
2495
|
// no-ops if already done, so the explicit sentinel still works as the fast path).
|
|
2401
|
-
if (entry.flowSessionId && entry.flowTaskKey &&
|
|
2496
|
+
if (entry.flowSessionId && entry.flowTaskKey && !entry.flowDone && legacyBuilderCompletionAllowed({ runtime: entry.runtime, flowRole: entry.flowRole, eventKind: evt.kind, eventSubtype: evt.subtype })) {
|
|
2402
2497
|
Promise.resolve().then(() => markFlowDone()).catch(() => { /* best-effort */ })
|
|
2403
2498
|
}
|
|
2404
2499
|
// The init system event carries the session's slash command list. Stash it
|
|
@@ -2419,11 +2514,8 @@ function openStructured({ id, runtime = 'claude', model, effort, resume, log, co
|
|
|
2419
2514
|
// human turn arrived first, the code-turn handler already consumed + cleared it
|
|
2420
2515
|
// (prepended to that turn), so this no-ops. Spec: docs/specs/2026-07-08-provider-switch-context-carry.md.
|
|
2421
2516
|
if (evt.kind === 'system' && entry.pendingRecap) {
|
|
2422
|
-
const
|
|
2423
|
-
entry.
|
|
2424
|
-
entry.flush?.()
|
|
2425
|
-
try { entry.session?.sendTurn(recap) } catch { /* not live — stays idle, user can continue manually */ }
|
|
2426
|
-
process.stderr.write(`\n ◆ carried ${recap.length}c context recap into fresh session (${id.slice(0, 8)}) after provider/restore switch.\n`)
|
|
2517
|
+
const recapLength = entry.pendingRecap.length
|
|
2518
|
+
if (dispatchPendingRecap(entry)) process.stderr.write(`\n ◆ carried ${recapLength}c context recap into fresh session (${id.slice(0, 8)}) after provider/restore switch.\n`)
|
|
2427
2519
|
}
|
|
2428
2520
|
// A background-warmed terminal is now live — free its warmer slot so the next queued
|
|
2429
2521
|
// idle terminal starts booting.
|
|
@@ -2730,7 +2822,7 @@ function respawnStructured(id, provider) {
|
|
|
2730
2822
|
// openStructured seed from the TARGET provider's configured model, which is the
|
|
2731
2823
|
// only model this lane was ever asked for. A same-env model change never reaches
|
|
2732
2824
|
// here — that path is an in-place setModel (see provider-switch).
|
|
2733
|
-
const { runtime, log, commands, mode, spawnedBy, sideParent, sideTask, pendingSideContexts, flowSessionId, flowTaskKey, cwd, rolePrompt, reviewSliceRoots, openedAt } = s
|
|
2825
|
+
const { runtime, log, commands, mode, spawnedBy, sideParent, sideTask, pendingSideContexts, flowSessionId, flowTaskKey, flowRole, flowReviewTarget, flowReviewTargets, flowReviewRound, dispatchBaseSha, revertTarget, cwd, managedWorktree, rolePrompt, reviewSliceRoots, openedAt, lastUsage } = s
|
|
2734
2826
|
// Context-carry (2026-07-08): a provider switch is not an SDK resume — the new backend
|
|
2735
2827
|
// starts blank mid-conversation. Synthesize a plain-text recap from the VISIBLE log NOW
|
|
2736
2828
|
// (before teardown) and hand it to the fresh session as its first turn so the agent
|
|
@@ -2749,7 +2841,7 @@ function respawnStructured(id, provider) {
|
|
|
2749
2841
|
// sessionData() (provider included) synchronously on open, so a bridge restart
|
|
2750
2842
|
// restores the lane on its CURRENT provider, not the original — and its next
|
|
2751
2843
|
// announce carries the new provider badge (additive {id,name} projection).
|
|
2752
|
-
openStructured({ id, runtime, provider, log, commands, mode, spawnedBy, sideParent, sideTask, pendingSideContexts, flowSessionId, flowTaskKey, cwd, rolePrompt, reviewSliceRoots, openedAt, carryRecap })
|
|
2844
|
+
openStructured({ id, runtime, provider, log, commands, mode, spawnedBy, sideParent, sideTask, pendingSideContexts, flowSessionId, flowTaskKey, flowRole, flowReviewTarget, flowReviewTargets, flowReviewRound, dispatchBaseSha, revertTarget, cwd, managedWorktree, rolePrompt, reviewSliceRoots, openedAt, lastUsage, carryRecap })
|
|
2753
2845
|
return true
|
|
2754
2846
|
}
|
|
2755
2847
|
|
|
@@ -3530,49 +3622,32 @@ channel
|
|
|
3530
3622
|
const okey = (r) => r._bt || r.openedAt || r.savedAt || 0
|
|
3531
3623
|
all.sort((a, b) => okey(a) - okey(b))
|
|
3532
3624
|
if (all.length) for (const rec of all) {
|
|
3625
|
+
const wasInterrupted = restoredTurnOpen(rec.log || [])
|
|
3626
|
+
const resumable = canResume(rec)
|
|
3627
|
+
const recoveryRecap = wasInterrupted ? buildRecapFromLog(rec.log || [], RECAP_CAP) : ''
|
|
3533
3628
|
// FL-M6 — restore the flow context (id/role/cwd) so an in-flight flow survives a
|
|
3534
3629
|
// bridge restart: the conductor keeps its subagent-block + plan interception, and
|
|
3535
3630
|
// lanes keep their worktree cwd + the ability to mark done.
|
|
3536
|
-
openStructured({ id: rec.id, runtime: rec.runtime || 'claude', model: rec.model || undefined, effort: rec.effort, provider: rec.runtime === 'codex' ? undefined : rec.provider || undefined, resume:
|
|
3631
|
+
openStructured({ id: rec.id, runtime: rec.runtime || 'claude', model: rec.model || undefined, effort: rec.effort, provider: rec.runtime === 'codex' ? undefined : rec.provider || undefined, resume: resumable ? rec.sessionId : undefined, log: rec.log, commands: rec.commands, mode: rec.mode, spawnedBy: rec.spawnedBy, sideParent: rec.sideParent, sideTask: rec.sideTask, pendingSideContexts: rec.pendingSideContexts, flowSessionId: rec.flowSessionId, flowTaskKey: rec.flowTaskKey, flowRole: rec.flowRole, flowReviewTarget: rec.flowReviewTarget, flowReviewTargets: rec.flowReviewTargets, flowReviewRound: rec.flowReviewRound, dispatchBaseSha: rec.dispatchBaseSha, revertTarget: rec.revertTarget, cwd: rec.cwd, managedWorktree: rec.managedWorktree, rolePrompt: rec.rolePrompt, reviewSliceRoots: rec.reviewSliceRoots, openedAt: rec.openedAt, lastUsage: rec.lastUsage, carryRecap: wasInterrupted ? recoveryRecap : rec.carryRecap,
|
|
3537
3632
|
// Lazy-boot restored terminals that were IDLE + not part of a flow: their transcript
|
|
3538
3633
|
// shows immediately; the query boots on first turn. Mid-turn + flow terminals boot now
|
|
3539
3634
|
// (mid-turn needs auto-resume; flow needs its lane live).
|
|
3540
|
-
defer: !rec.flowSessionId && !
|
|
3635
|
+
defer: !rec.flowSessionId && !wasInterrupted })
|
|
3541
3636
|
const re = sessions.get(rec.id)
|
|
3542
|
-
if (re && rec.flowReviewTarget) re.flowReviewTarget = rec.flowReviewTarget
|
|
3543
|
-
// S4 — a resumed re-dispatch lane keeps its surviving revert target across a bridge restart.
|
|
3544
|
-
if (re && rec.revertTarget) re.revertTarget = rec.revertTarget
|
|
3545
3637
|
// FL-M6 — a flow LANE whose turn had already ENDED before the restart finished its
|
|
3546
3638
|
// slice, but its done-signal was lost while the bridge was down (resume won't re-run
|
|
3547
3639
|
// an ended turn, so it would sit idle → the flow stalls forever). Reconcile: re-record
|
|
3548
3640
|
// it done. A lane still MID-turn (restoredTurnOpen) resumes + finishes normally.
|
|
3549
|
-
|
|
3641
|
+
const lastTerminal = [...(rec.log || [])].reverse().find((event) => event?.kind === 'result' || event?.kind === 'error')
|
|
3642
|
+
if (re && rec.flowTaskKey && legacyBuilderCompletionAllowed({ runtime: re.runtime, flowRole: re.flowRole, eventKind: lastTerminal?.kind, eventSubtype: lastTerminal?.subtype, interrupted: wasInterrupted })) {
|
|
3550
3643
|
setTimeout(() => { try { re.markFlowDone?.() } catch { /* noop */ } }, 2000)
|
|
3551
3644
|
}
|
|
3552
|
-
//
|
|
3553
|
-
//
|
|
3554
|
-
|
|
3555
|
-
|
|
3556
|
-
// fixed timer: a resume takes ~40s to become live (MCP boot + sanitizeSession), and a
|
|
3557
|
-
// 2.5s sendTurn pushed into an input stream nothing is consuming yet is silently lost
|
|
3558
|
-
// (the 2026-07-02 "didn't autocontinue" bug). Instead FLAG it here and fire the single
|
|
3559
|
-
// continue when the session's init `system` event actually arrives (onEvent), i.e. the
|
|
3560
|
-
// instant it's live. openStructured already closed the interrupted turn idle, so a
|
|
3561
|
-
// continue that can't be delivered just leaves it idle (recoverable) — never a hang.
|
|
3562
|
-
if (re && !rec.flowSessionId && canResume(rec) && restoredTurnOpen(rec.log || [])) {
|
|
3563
|
-
const recovery = armInterruptedResume(re)
|
|
3645
|
+
// Every interrupted role resumes exactly once. Flow lanes do not have a
|
|
3646
|
+
// startup redispatch path; excluding them here stranded conductors/builders.
|
|
3647
|
+
if (re && wasInterrupted) {
|
|
3648
|
+
const recovery = recoverInterruptedTurn(re, { resumable, recap: recoveryRecap })
|
|
3564
3649
|
if (recovery === 'sent') process.stderr.write(`\n ◆ auto-resumed interrupted Codex turn (${rec.id.slice(0, 8)}) — sent continue.\n`)
|
|
3565
|
-
|
|
3566
|
-
// Context-carry (2026-07-08): the SAME mid-turn case but the SDK context CANNOT
|
|
3567
|
-
// resume (canResume false — a stale/expired session). A bare "continue" would land
|
|
3568
|
-
// in a blank backend, so instead carry a plain-text recap of the visible log as the
|
|
3569
|
-
// first turn — the interrupted work resumes WITH its history. Mid-turn only
|
|
3570
|
-
// (restoredTurnOpen): we never arm an unprompted turn on an idle restored lane (its
|
|
3571
|
-
// transcript stays visible; the person starts their next request fresh). If the
|
|
3572
|
-
// person types before this recap fires, the code-turn handler prepends it instead.
|
|
3573
|
-
// Flow lanes re-dispatch separately (excluded above).
|
|
3574
|
-
else if (re && !rec.flowSessionId && !canResume(rec) && restoredTurnOpen(rec.log || [])) {
|
|
3575
|
-
re.pendingRecap = buildRecapFromLog(rec.log || [], RECAP_CAP)
|
|
3650
|
+
else if (recovery === 'recap-sent') process.stderr.write(`\n ◆ resumed interrupted Codex turn (${rec.id.slice(0, 8)}) with a fresh context recap.\n`)
|
|
3576
3651
|
}
|
|
3577
3652
|
}
|
|
3578
3653
|
// Fresh open: ONLY for an explicit `-- <claude>` share. In account mode
|
|
@@ -3612,6 +3687,36 @@ channel
|
|
|
3612
3687
|
}
|
|
3613
3688
|
})
|
|
3614
3689
|
|
|
3690
|
+
async function handleFlowRevert(payload) {
|
|
3691
|
+
try {
|
|
3692
|
+
return await executeFlowRevert({
|
|
3693
|
+
payload,
|
|
3694
|
+
bridgeName: name,
|
|
3695
|
+
lanes: [...sessions.entries()],
|
|
3696
|
+
prepareLane: (entry, p) => prepareRedispatch({
|
|
3697
|
+
lane: { cwd: entry.cwd, session: entry.session, commitSha: entry.commitSha, revertTarget: p.commitSha ?? entry.commitSha ?? entry.revertTarget ?? null },
|
|
3698
|
+
sanitize: sanitizeSession,
|
|
3699
|
+
}),
|
|
3700
|
+
rememberRedispatch: (prepared, _entry, p) => {
|
|
3701
|
+
flowRedispatch.set(redispatchKey(p.flowId, p.taskKey), { resumeSessionId: prepared.resumeSessionId, revertTarget: prepared.revertTarget })
|
|
3702
|
+
if (prepared.healed) process.stderr.write(`\n ${A.dim}◆ flow re-dispatch heal — ${p.taskKey}: ${prepared.healed} dangling tool block(s) healed before resume.${A.rst}\n`)
|
|
3703
|
+
},
|
|
3704
|
+
endLane: (laneId) => { try { endStructured(laneId) } catch { /* already gone */ } },
|
|
3705
|
+
stopPreview: stopFlowPreviews,
|
|
3706
|
+
wait: () => new Promise((resolve) => setTimeout(resolve, 400)),
|
|
3707
|
+
revert: (p) => revertLane({ flowId: p.flowId, taskKey: p.taskKey }),
|
|
3708
|
+
broadcastReverted: (out) => {
|
|
3709
|
+
bcast('flow-reverted', { term: 'flow', flowId: out.flowId, taskKey: out.taskKey, reviewTaskKey: out.reviewTaskKey, branch: out.branch }, flowChannel)
|
|
3710
|
+
process.stderr.write(`\n ${A.yel}◆ flow slice reverted — ${out.taskKey} (${out.branch}).${A.rst}\n`)
|
|
3711
|
+
},
|
|
3712
|
+
onPrepError: (error, _entry, p) => process.stderr.write(`\n ${A.yel}◆ flow re-dispatch prep failed (${p.taskKey}): ${error?.message || error} — will re-dispatch fresh.${A.rst}\n`),
|
|
3713
|
+
})
|
|
3714
|
+
} catch (error) {
|
|
3715
|
+
process.stderr.write(`\n ${A.yel}◆ flow revert failed: ${error?.message || error}${A.rst}\n`)
|
|
3716
|
+
return { ok: false, error: error?.message || String(error) }
|
|
3717
|
+
}
|
|
3718
|
+
}
|
|
3719
|
+
|
|
3615
3720
|
// ── Thinkpool Flow control channel (tpflow:<room>) ──────────────────────────
|
|
3616
3721
|
// Separate from the room channel above so the web's useFlow hook can own its own
|
|
3617
3722
|
// topic without colliding with the room's tpcode channel (see the flowChannel def +
|
|
@@ -3630,6 +3735,16 @@ flowChannel
|
|
|
3630
3735
|
if (!payload?.flowId) return
|
|
3631
3736
|
if (payload.host && payload.host !== name) return
|
|
3632
3737
|
for (const [, e] of sessions) if (e.flowSessionId === payload.flowId) return // already conducting this flow
|
|
3738
|
+
const origin = payload.originTerm ? sessions.get(payload.originTerm) : null
|
|
3739
|
+
// A current client names the invoking terminal. Only the bridge that actually
|
|
3740
|
+
// owns that terminal may launch its conductor; never trust a spoofed runtime/model
|
|
3741
|
+
// from a broadcast received by another machine. Legacy clients omitted originTerm
|
|
3742
|
+
// and were Claude-only, so that exact path remains supported.
|
|
3743
|
+
if (payload.originTerm && !origin) return
|
|
3744
|
+
const flowRuntime = origin ? normalizeFlowRuntime(origin.runtime, null) : 'claude'
|
|
3745
|
+
if (!flowRuntime) return
|
|
3746
|
+
const flowCatalog = flowRuntime === 'codex' ? (origin?.models || readCodexModels()) : []
|
|
3747
|
+
const conductorModel = flowConductorModelFor({ runtime: flowRuntime, originModel: origin?.model, catalog: flowCatalog })
|
|
3633
3748
|
const cid = randomUUID()
|
|
3634
3749
|
termNames[cid] = `Flow · ${String(payload.flowId).slice(0, 6)}`
|
|
3635
3750
|
saveNames(room, termNames)
|
|
@@ -3640,7 +3755,7 @@ flowChannel
|
|
|
3640
3755
|
// Model tiers (2026-07-03-flow-lane-model-tiers): the conductor keeps whatever brain it was
|
|
3641
3756
|
// given by default (TP_FLOW_CONDUCTOR_MODEL unset → undefined → today's behavior); set the env
|
|
3642
3757
|
// to pin a cheaper/smarter conductor. Lanes get tiered below via laneModelFor.
|
|
3643
|
-
openStructured({ id: cid,
|
|
3758
|
+
openStructured({ id: cid, runtime: flowRuntime, model: conductorModel, mode: flowRuntime === 'codex' ? 'plan' : 'default', rolePrompt: flowRuntime === 'codex' ? FLOW_CODEX_CONDUCTOR_PROMPT : FLOW_CONDUCTOR_PROMPT, flowSessionId: payload.flowId, flowRole: 'conductor', spawnedBy: `flow:${payload.flowId}` })
|
|
3644
3759
|
const ce = sessions.get(cid)
|
|
3645
3760
|
if (ce?.session) { try { ce.session.sendTurn(payload.prompt || '') } catch { /* session still starting */ } }
|
|
3646
3761
|
process.stderr.write(`\n ${A.mag}◆ flow conductor launched — flow ${String(payload.flowId).slice(0, 8)} (plan mode).${A.rst}\n`)
|
|
@@ -3654,6 +3769,14 @@ flowChannel
|
|
|
3654
3769
|
// building (DB writes stay room-side, member-authed).
|
|
3655
3770
|
if (!payload?.flowId || !Array.isArray(payload.tasks)) return
|
|
3656
3771
|
if (payload.host && payload.host !== name) return
|
|
3772
|
+
const conductor = [...sessions.values()].find((entry) => entry.flowSessionId === payload.flowId && entry.flowRole === 'conductor')
|
|
3773
|
+
if (!conductor) {
|
|
3774
|
+
process.stderr.write(`\n ${A.yel}◆ flow dispatch held — this bridge has no live/restored conductor for ${String(payload.flowId).slice(0, 8)}.${A.rst}\n`)
|
|
3775
|
+
return
|
|
3776
|
+
}
|
|
3777
|
+
const flowRuntime = normalizeFlowRuntime(conductor.runtime, null)
|
|
3778
|
+
if (!flowRuntime) return
|
|
3779
|
+
const flowCatalog = flowRuntime === 'codex' ? (conductor.models || readCodexModels()) : []
|
|
3657
3780
|
const assignments = []
|
|
3658
3781
|
// Step 6 — cap concurrent Flow lanes (invariant: ≤ FLOW_LIMITS.maxConcurrentLanes).
|
|
3659
3782
|
// Overflow tasks stay pending; the room re-dispatches them in the next wave.
|
|
@@ -3682,7 +3805,13 @@ flowChannel
|
|
|
3682
3805
|
// Step 4 — a `review` slice gets the adversarial reviewer prompt (try-to-break),
|
|
3683
3806
|
// every other slice gets the builder prompt.
|
|
3684
3807
|
const isReview = t.slice_type === 'review'
|
|
3808
|
+
if (!validReviewTargetShape(t, flowRuntime)) {
|
|
3809
|
+
process.stderr.write(`\n ${A.yel}◆ flow dispatch held — Codex review ${t.task_key} must target exactly one dependency.${A.rst}\n`)
|
|
3810
|
+
continue
|
|
3811
|
+
}
|
|
3685
3812
|
const { dir } = createFlowWorktree({ flowId: payload.flowId, taskKey: t.task_key })
|
|
3813
|
+
let dispatchBaseSha = null
|
|
3814
|
+
try { dispatchBaseSha = execFileSync('git', ['-C', dir, 'rev-parse', 'HEAD'], { encoding: 'utf8' }).trim() } catch { /* non-git fixture */ }
|
|
3686
3815
|
const laneId = randomUUID()
|
|
3687
3816
|
termNames[laneId] = `Flow · ${t.task_key}`
|
|
3688
3817
|
saveNames(room, termNames)
|
|
@@ -3715,12 +3844,14 @@ flowChannel
|
|
|
3715
3844
|
// a lane later, on demand, via activateLaneSkill — never the base prompt here.
|
|
3716
3845
|
// S4 — resume: on a re-dispatch, replay the killed lane's HEALED transcript (Heal-3'd,
|
|
3717
3846
|
// no dangling tool_use → no 400) instead of a cold start; undefined for a fresh lane.
|
|
3718
|
-
const laneBase =
|
|
3847
|
+
const laneBase = flowRuntime === 'codex'
|
|
3848
|
+
? (isReview ? FLOW_CODEX_REVIEWER_PROMPT : FLOW_CODEX_LANE_PROMPT)
|
|
3849
|
+
: (isReview ? FLOW_REVIEWER_PROMPT : FLOW_LANE_PROMPT)
|
|
3719
3850
|
const laneRolePrompt = buildLanePrompt({ base: laneBase })
|
|
3720
3851
|
// Model tiers (2026-07-03-flow-lane-model-tiers): pick the lane's brain by slice_type
|
|
3721
3852
|
// (scaffold→sonnet, feature/fix/review→opus; env can blanket-override or `inherit` to
|
|
3722
3853
|
// restore today's exact behavior). undefined → no model key passed (openStructured default).
|
|
3723
|
-
openStructured({ id: laneId, cwd: dir, model:
|
|
3854
|
+
openStructured({ id: laneId, runtime: flowRuntime, cwd: dir, model: flowLaneModelFor({ sliceType: t.slice_type, runtime: flowRuntime, catalog: flowCatalog }), mode: flowRuntime === 'codex' && isReview ? 'review' : 'bypassPermissions', rolePrompt: laneRolePrompt, flowSessionId: payload.flowId, flowTaskKey: t.task_key, flowRole: isReview ? 'reviewer' : 'builder', flowReviewTargets: isReview ? (t.deps || []) : [], dispatchBaseSha, revertTarget: redispatch?.revertTarget || null, spawnedBy: `flow:${payload.flowId}`, resume: redispatch?.resumeSessionId || undefined, reviewSliceRoots })
|
|
3724
3855
|
const le = sessions.get(laneId)
|
|
3725
3856
|
if (le) {
|
|
3726
3857
|
// S4 — stamp the surviving revert target on the resumed lane so a later reviewer still
|
|
@@ -3745,7 +3876,9 @@ flowChannel
|
|
|
3745
3876
|
: '') +
|
|
3746
3877
|
`ACCEPTANCE to INDEPENDENTLY verify: ${t.acceptance || t.title}\n` +
|
|
3747
3878
|
`\nProject (context): ${payload.flowPrompt || ''}\n\n` +
|
|
3748
|
-
|
|
3879
|
+
(flowRuntime === 'codex'
|
|
3880
|
+
? `Review it without mutating the builder worktree. For write-producing install/build/test commands, copy its source into scratch under your own current worktree first. Submit the structured verdict with the ThinkPool submit_flow_review MCP tool; never write FLOW_REVIEW.json or call mark_flow_done.`
|
|
3881
|
+
: `Review it — run it, hunt for failure. When done, use the Write tool to write FLOW_REVIEW.json with {"pass":<boolean>,"reasons":["<specific finding>"],"taskKey":"<the slice you reviewed>"}. That Write IS your verdict.`)
|
|
3749
3882
|
: `[Flow lane — slice "${t.task_key}" of flow ${String(payload.flowId).slice(0, 8)}]\n` +
|
|
3750
3883
|
`TITLE: ${t.title}\n` +
|
|
3751
3884
|
`SCOPE (files you OWN — edit ONLY these): ${t.scope || '(none stated)'}\n` +
|
|
@@ -3757,7 +3890,8 @@ flowChannel
|
|
|
3757
3890
|
((crossWave) => crossWave ? `${crossWave}\n` : '')(
|
|
3758
3891
|
assembleCrossWaveContext(payload.flowId, { baseDir: process.cwd(), deps: t.deps }).text) +
|
|
3759
3892
|
`\nProject (context): ${payload.flowPrompt || ''}\n\n` +
|
|
3760
|
-
`Build your slice. Own only your files. Done = it runs + meets acceptance. Commit when done.`
|
|
3893
|
+
`Build your slice. Own only your files. Done = it runs + meets acceptance. Commit when done.` +
|
|
3894
|
+
(flowRuntime === 'codex' ? ` Then call the ThinkPool mark_flow_done MCP tool; do not write FLOW_DONE.` : '')
|
|
3761
3895
|
try { le.session.sendTurn(spec) } catch { /* session still starting */ }
|
|
3762
3896
|
}
|
|
3763
3897
|
}
|
|
@@ -3809,45 +3943,7 @@ flowChannel
|
|
|
3809
3943
|
// Step 4 — atomic per-lane revert: roll back ONE slice (remove its worktree +
|
|
3810
3944
|
// branch) without touching the others, when adversarial review rejects it. The
|
|
3811
3945
|
// room re-dispatches the reverted task on the next wave.
|
|
3812
|
-
|
|
3813
|
-
if (payload.host && payload.host !== name) return
|
|
3814
|
-
try {
|
|
3815
|
-
// FL-M8 — tear down the lane's live session FIRST. revertLane does `git worktree
|
|
3816
|
-
// remove --force`; pulling the worktree out from under a still-running lane corrupts
|
|
3817
|
-
// its process state. Retire it, let the SDK release the cwd, THEN remove the tree.
|
|
3818
|
-
let killed = false
|
|
3819
|
-
for (const [sid, e] of sessions) {
|
|
3820
|
-
if (e.flowSessionId === payload.flowId && e.flowTaskKey === payload.taskKey && !e.flowDone) {
|
|
3821
|
-
// S4 slice 2 — prepare a CLEAN re-dispatch BEFORE we kill the lane + remove its
|
|
3822
|
-
// worktree. prepareRedispatch heals the transcript (Heal-3 consumer — a dangling
|
|
3823
|
-
// tool_use from a mid-tool-call kill gets a synthetic tool_result so a resume can't
|
|
3824
|
-
// 400) and captures the resume sessionId + the revert target (which revertLane's
|
|
3825
|
-
// branch delete would otherwise destroy). We record it under flowId::taskKey so the
|
|
3826
|
-
// next flow-dispatch wave resumes the healed session. Idempotent: healing an
|
|
3827
|
-
// already-clean transcript is a no-op. No per-lane topic is captured — the resumed
|
|
3828
|
-
// lane reuses the room-wide tpflow broadcast, so there is no H41 collision to guard.
|
|
3829
|
-
try {
|
|
3830
|
-
const prep = prepareRedispatch({
|
|
3831
|
-
lane: { cwd: e.cwd, session: e.session, commitSha: e.commitSha, revertTarget: payload.commitSha ?? e.commitSha ?? null },
|
|
3832
|
-
sanitize: sanitizeSession,
|
|
3833
|
-
})
|
|
3834
|
-
flowRedispatch.set(redispatchKey(payload.flowId, payload.taskKey), {
|
|
3835
|
-
resumeSessionId: prep.resumeSessionId,
|
|
3836
|
-
revertTarget: prep.revertTarget,
|
|
3837
|
-
})
|
|
3838
|
-
if (prep.healed) process.stderr.write(`\n ${A.dim}◆ flow re-dispatch heal — ${payload.taskKey}: ${prep.healed} dangling tool block(s) healed before resume.${A.rst}\n`)
|
|
3839
|
-
} catch (err) { process.stderr.write(`\n ${A.yel}◆ flow re-dispatch prep failed (${payload.taskKey}): ${err?.message || err} — will re-dispatch fresh.${A.rst}\n`) }
|
|
3840
|
-
e.flowDone = true
|
|
3841
|
-
try { endStructured(sid) } catch { /* already gone */ }
|
|
3842
|
-
stopFlowPreviews(payload.flowId, sid) // FL-M9 — drop the reverted lane's preview
|
|
3843
|
-
killed = true
|
|
3844
|
-
}
|
|
3845
|
-
}
|
|
3846
|
-
if (killed) await new Promise((r) => setTimeout(r, 400))
|
|
3847
|
-
const { branch } = revertLane({ flowId: payload.flowId, taskKey: payload.taskKey })
|
|
3848
|
-
bcast('flow-reverted', { term: 'flow', flowId: payload.flowId, taskKey: payload.taskKey, branch }, flowChannel)
|
|
3849
|
-
process.stderr.write(`\n ${A.yel}◆ flow slice reverted — ${payload.taskKey} (${branch}).${A.rst}\n`)
|
|
3850
|
-
} catch (e) { process.stderr.write(`\n ${A.yel}◆ flow revert failed: ${e?.message || e}${A.rst}\n`) }
|
|
3946
|
+
await handleFlowRevert(payload)
|
|
3851
3947
|
})
|
|
3852
3948
|
.on('broadcast', { event: 'flow-cancel' }, ({ payload }) => {
|
|
3853
3949
|
// A user cancelled the flow (or it was interrupted). The client also flips the DB
|
package/codex-session.mjs
CHANGED
|
@@ -39,6 +39,7 @@ export const CODEX_MODE_CONFIG = {
|
|
|
39
39
|
default: { sandbox: 'workspace-write', approvalPolicy: 'untrusted' },
|
|
40
40
|
acceptEdits: { sandbox: 'workspace-write', approvalPolicy: 'on-request' },
|
|
41
41
|
plan: { sandbox: 'read-only', approvalPolicy: 'never' },
|
|
42
|
+
review: { sandbox: 'workspace-write', approvalPolicy: 'never' },
|
|
42
43
|
bypassPermissions: { sandbox: 'danger-full-access', approvalPolicy: 'never' },
|
|
43
44
|
}
|
|
44
45
|
|
|
@@ -73,6 +74,15 @@ export function defaultModeForRuntime(runtime) {
|
|
|
73
74
|
return runtime === 'codex' ? 'bypassPermissions' : 'default'
|
|
74
75
|
}
|
|
75
76
|
|
|
77
|
+
export function codexWritableDirsForSandbox(sandbox, writableDirs = []) {
|
|
78
|
+
return normalizeCodexSandbox(sandbox) === 'read-only' ? [] : writableDirs
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
export function codexExtraWritableDirsForMode(mode, sandbox, writableDirs = []) {
|
|
82
|
+
if (mode === 'review') return []
|
|
83
|
+
return codexWritableDirsForSandbox(sandbox, writableDirs)
|
|
84
|
+
}
|
|
85
|
+
|
|
76
86
|
export function readCodexDefaultModel({ home = process.env.CODEX_HOME || path.join(os.homedir(), '.codex'), fsImpl = fs } = {}) {
|
|
77
87
|
try {
|
|
78
88
|
const text = fsImpl.readFileSync(path.join(home, 'config.toml'), 'utf8')
|
|
@@ -183,11 +193,15 @@ export function codexPublishAccess({ cwd, env = process.env, tmpdir = os.tmpdir(
|
|
|
183
193
|
}
|
|
184
194
|
}
|
|
185
195
|
|
|
196
|
+
export const CODEX_ROOM_CASCADE_REMINDER = [
|
|
197
|
+
'THINKPOOL ROOM WORKFLOW: delegated work belongs in visible spawn_terminal lanes, never hidden in-process subagents. Each lane owns its own worktree. For a genuinely decomposable task, state a short plan in chat, use sliceType scaffold for mechanical work, feature/fix for builders, and review for adversarial verification; review must use a balanced model tier. Read results, verify the integrated outcome, and close every lane you spawned so it does not consume room capacity.',
|
|
198
|
+
].join(' ')
|
|
199
|
+
|
|
186
200
|
export function buildCodexPrompt({ text, rolePrompt, roomContext, firstTurn = false }) {
|
|
187
201
|
const context = typeof roomContext === 'function' ? roomContext() : roomContext
|
|
188
202
|
const additions = [
|
|
189
203
|
firstTurn ? CODEX_THINKPOOL_FIRST_TURN_PREAMBLE : '',
|
|
190
|
-
rolePrompt,
|
|
204
|
+
rolePrompt || CODEX_ROOM_CASCADE_REMINDER,
|
|
191
205
|
context,
|
|
192
206
|
].map((v) => String(v || '').trim()).filter(Boolean)
|
|
193
207
|
if (!additions.length) return String(text ?? '')
|
|
@@ -470,7 +484,7 @@ export function startCodexSession({ cwd, model, effort: initialEffort = 'high',
|
|
|
470
484
|
appServer = appServerFactory({
|
|
471
485
|
cwd,
|
|
472
486
|
env: childEnv,
|
|
473
|
-
args: buildCodexAppServerArgs({ providerConfig, mcpUrl: peer?.url, writableDirs: publishAccess.writableDirs }),
|
|
487
|
+
args: buildCodexAppServerArgs({ providerConfig, mcpUrl: peer?.url, writableDirs: codexExtraWritableDirsForMode(activeMode, sandbox, publishAccess.writableDirs) }),
|
|
474
488
|
onNotification: appServerNotification,
|
|
475
489
|
onServerRequest: appServerRequest,
|
|
476
490
|
onClose: (error) => {
|
|
@@ -572,7 +586,7 @@ export function startCodexSession({ cwd, model, effort: initialEffort = 'high',
|
|
|
572
586
|
approvalPolicy: modeConfig.approvalPolicy,
|
|
573
587
|
providerConfig,
|
|
574
588
|
mcpUrl: peer?.url,
|
|
575
|
-
writableDirs: publishAccess.writableDirs,
|
|
589
|
+
writableDirs: codexExtraWritableDirsForMode(activeMode, sandbox, publishAccess.writableDirs),
|
|
576
590
|
images: Array.isArray(options.images) ? options.images : [],
|
|
577
591
|
prompt,
|
|
578
592
|
})
|
package/flow-conductor.mjs
CHANGED
|
@@ -63,6 +63,12 @@ export const FLOW_CONDUCTOR_PROMPT = [
|
|
|
63
63
|
'MODE-AWARE. The user picked a mode — steer (watch every lane live), guide (DEFAULT: approve this plan, then review at phase boundaries), or autopilot (full-auto with a budget cap + adaptive escalation that pulls them in only if a lane stalls). Produce the full plan now regardless; the mode changes how much the human intervenes later, not how you decompose.',
|
|
64
64
|
].join(' ')
|
|
65
65
|
|
|
66
|
+
export const FLOW_CODEX_CONDUCTOR_PROMPT = [
|
|
67
|
+
'THINKPOOL FLOW CONDUCTOR — you are decomposition-only. Do not build, edit files, run commands, or spawn agents. Read only when needed to make the decomposition concrete.',
|
|
68
|
+
'Turn the request into a small acyclic graph of independently runnable slices with disjoint file ownership. Each task needs key, title, scope, acceptance, deps, and sliceType (scaffold, feature, fix, or review). Review is intelligence-sensitive and must never be assigned a cheap/mechanical tier. Every review task must depend on exactly ONE builder task; express builder -> review -> downstream explicitly in the DAG.',
|
|
69
|
+
'Submit the finished graph by calling the ThinkPool submit_flow_plan MCP tool with one JSON object encoded in its plan argument. Do not write FLOW_PLAN.json and do not use ExitPlanMode. The MCP submission is your only completion path; after it succeeds, stop and wait for the room.',
|
|
70
|
+
].join(' ')
|
|
71
|
+
|
|
66
72
|
// Env vars handed to a conductor session via startClaudeSession({ env }). The conductor
|
|
67
73
|
// reads these to know which Flow run it owns + what mode the human picked.
|
|
68
74
|
export function buildConductorEnv ({ flowSessionId, mode, budgetCapAutopilot = null }) {
|
|
@@ -98,6 +104,13 @@ export const FLOW_LANE_PROMPT = [
|
|
|
98
104
|
'BE RIGOROUS, NOT VIBES. Trace the data path, root-cause before fixing, verify before claiming done. A lane that ships a guess costs the whole ensemble.',
|
|
99
105
|
].join(' ')
|
|
100
106
|
|
|
107
|
+
export const FLOW_CODEX_LANE_PROMPT = [
|
|
108
|
+
'THINKPOOL FLOW — you are one builder lane in an ensemble. The conductor gives you one runnable slice with disjoint scope. Edit only that scope and never touch another lane or worktree.',
|
|
109
|
+
'Done means the slice runs and its acceptance proof has been observed. Diagnose before changing code, verify with the real command or behavior, and make one focused git commit in your assigned worktree. Do not push.',
|
|
110
|
+
'After the commit and verification succeed, call the ThinkPool mark_flow_done MCP tool exactly once. Do not write FLOW_DONE. The MCP call is the completion signal that records your commit and unblocks dependent slices.',
|
|
111
|
+
'Do not spawn more lanes. If the scope is unsafe or a dependency is missing, report the blocker instead of editing outside your ownership.',
|
|
112
|
+
].join(' ')
|
|
113
|
+
|
|
101
114
|
// Env vars handed to a lane session so it knows its Flow run + which task it owns.
|
|
102
115
|
export function buildLaneEnv ({ flowSessionId, taskKey, laneId }) {
|
|
103
116
|
return {
|
|
@@ -238,4 +251,3 @@ export function activateLaneSkill (name, { publicDir, customDir, fs } = {}) {
|
|
|
238
251
|
const dirs = publicDir === undefined && customDir === undefined ? flowSkillDirs() : { publicDir, customDir }
|
|
239
252
|
return skillLoadBody(name, { ...dirs, ...(fs ? { fs } : {}) })
|
|
240
253
|
}
|
|
241
|
-
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
// Host-owned Flow revert lifecycle. Supabase broadcast channels use self:false,
|
|
2
|
+
// so a reviewer on this bridge must call this handler directly; remote broadcasts
|
|
3
|
+
// delegate to the same function.
|
|
4
|
+
export async function executeFlowRevert({
|
|
5
|
+
payload,
|
|
6
|
+
bridgeName,
|
|
7
|
+
lanes = [],
|
|
8
|
+
prepareLane,
|
|
9
|
+
rememberRedispatch,
|
|
10
|
+
endLane,
|
|
11
|
+
stopPreview,
|
|
12
|
+
revert,
|
|
13
|
+
broadcastReverted,
|
|
14
|
+
wait = () => Promise.resolve(),
|
|
15
|
+
onPrepError = () => {},
|
|
16
|
+
} = {}) {
|
|
17
|
+
if (!payload?.flowId || !payload?.taskKey) return { ok: false, ignored: 'invalid' }
|
|
18
|
+
if (payload.host && payload.host !== bridgeName) return { ok: false, ignored: 'wrong-host' }
|
|
19
|
+
let killed = false
|
|
20
|
+
for (const [laneId, entry] of lanes) {
|
|
21
|
+
if (entry?.flowSessionId !== payload.flowId || entry?.flowTaskKey !== payload.taskKey || entry?.flowDone) continue
|
|
22
|
+
try {
|
|
23
|
+
const prepared = prepareLane?.(entry, payload)
|
|
24
|
+
if (prepared) rememberRedispatch?.(prepared, entry, payload)
|
|
25
|
+
} catch (error) { onPrepError(error, entry, payload) }
|
|
26
|
+
entry.flowDone = true
|
|
27
|
+
endLane?.(laneId, entry)
|
|
28
|
+
stopPreview?.(payload.flowId, laneId)
|
|
29
|
+
killed = true
|
|
30
|
+
}
|
|
31
|
+
if (killed) await wait()
|
|
32
|
+
const reverted = await revert(payload)
|
|
33
|
+
const result = {
|
|
34
|
+
ok: true,
|
|
35
|
+
flowId: payload.flowId,
|
|
36
|
+
taskKey: payload.taskKey,
|
|
37
|
+
reviewTaskKey: payload.reviewTaskKey || null,
|
|
38
|
+
branch: reverted?.branch || null,
|
|
39
|
+
}
|
|
40
|
+
await broadcastReverted(result)
|
|
41
|
+
return result
|
|
42
|
+
}
|
package/flow-models.mjs
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
// Runtime-aware model routing for Flow and ordinary room cascades.
|
|
2
|
+
// Pure over the supplied environment/catalog so the bridge never sends a model slug
|
|
3
|
+
// that the selected runtime did not advertise.
|
|
4
|
+
|
|
5
|
+
const CLAUDE_TIERS = {
|
|
6
|
+
scaffold: 'sonnet',
|
|
7
|
+
feature: 'opus',
|
|
8
|
+
fix: 'opus',
|
|
9
|
+
review: 'opus',
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
const CODEX_SCAFFOLD = ['gpt-5.6-luna', 'gpt-5.4-mini', 'gpt-5.3-codex-spark']
|
|
13
|
+
const CODEX_BALANCED = ['gpt-5.6-terra', 'gpt-5.4']
|
|
14
|
+
|
|
15
|
+
export function normalizeFlowRuntime(runtime, fallback = null) {
|
|
16
|
+
if (runtime === 'claude' || runtime === 'codex') return runtime
|
|
17
|
+
return fallback === 'claude' || fallback === 'codex' ? fallback : null
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export function modelCatalogValues(catalog = []) {
|
|
21
|
+
return new Set((Array.isArray(catalog) ? catalog : []).map((model) => (
|
|
22
|
+
typeof model === 'string' ? model : model?.value || model?.slug
|
|
23
|
+
)).filter(Boolean))
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
function firstVisible(candidates, catalog) {
|
|
27
|
+
const visible = modelCatalogValues(catalog)
|
|
28
|
+
for (const candidate of candidates) if (visible.has(candidate)) return candidate
|
|
29
|
+
if (candidates === CODEX_SCAFFOLD) {
|
|
30
|
+
for (const model of visible) if (/spark/i.test(model)) return model
|
|
31
|
+
}
|
|
32
|
+
return undefined
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export function flowLaneModelFor({ sliceType, runtime = 'claude', catalog = [], env = process.env } = {}) {
|
|
36
|
+
if (normalizeFlowRuntime(runtime, 'claude') === 'codex') {
|
|
37
|
+
const override = env.TP_FLOW_CODEX_LANE_MODEL
|
|
38
|
+
if (override === 'inherit') return undefined
|
|
39
|
+
if (override) return modelCatalogValues(catalog).has(override) ? override : undefined
|
|
40
|
+
return firstVisible(sliceType === 'scaffold' ? CODEX_SCAFFOLD : CODEX_BALANCED, catalog)
|
|
41
|
+
}
|
|
42
|
+
const override = env.TP_FLOW_CLAUDE_LANE_MODEL || env.TP_FLOW_LANE_MODEL
|
|
43
|
+
if (override === 'inherit') return undefined
|
|
44
|
+
if (override) return override
|
|
45
|
+
return CLAUDE_TIERS[sliceType] || 'opus'
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export function flowConductorModelFor({ runtime = 'claude', originModel, catalog = [], env = process.env } = {}) {
|
|
49
|
+
if (normalizeFlowRuntime(runtime, 'claude') === 'codex') {
|
|
50
|
+
const override = env.TP_FLOW_CODEX_CONDUCTOR_MODEL
|
|
51
|
+
if (override === 'inherit') return undefined
|
|
52
|
+
const visible = modelCatalogValues(catalog)
|
|
53
|
+
if (override) return visible.has(override) ? override : undefined
|
|
54
|
+
return originModel && visible.has(originModel) ? originModel : undefined
|
|
55
|
+
}
|
|
56
|
+
const override = env.TP_FLOW_CLAUDE_CONDUCTOR_MODEL || env.TP_FLOW_CONDUCTOR_MODEL
|
|
57
|
+
if (override === 'inherit') return undefined
|
|
58
|
+
// Claude non-regression: the conductor historically inherited the host default,
|
|
59
|
+
// not the invoking lane's selected model.
|
|
60
|
+
return override || undefined
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
export function spawnedLaneModelFor({ explicitModel, sliceType, runtime = 'claude', catalog = [], env = process.env } = {}) {
|
|
64
|
+
return explicitModel || flowLaneModelFor({ sliceType, runtime, catalog, env })
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
export function resolveCodexModel(requested, catalog = []) {
|
|
68
|
+
if (!requested) return undefined
|
|
69
|
+
return modelCatalogValues(catalog).has(requested) ? requested : undefined
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
export function assertRuntimeModelCompatible({ runtime = 'claude', provider = null, model } = {}) {
|
|
73
|
+
if (runtime === 'claude' && (!provider || provider === 'anthropic') && /^gpt-/i.test(String(model || ''))) {
|
|
74
|
+
throw new Error(`Codex model ${JSON.stringify(model)} cannot run on the built-in Claude provider`)
|
|
75
|
+
}
|
|
76
|
+
return true
|
|
77
|
+
}
|
package/flow-review.mjs
CHANGED
|
@@ -32,6 +32,13 @@ export const FLOW_REVIEWER_PROMPT = [
|
|
|
32
32
|
'EMIT YOUR VERDICT BY WRITING FLOW_REVIEW.json. Do NOT call mark_flow_done or ExitPlanMode (they hang here). As your LAST action, use the Write tool with file_path "FLOW_REVIEW.json" and content = a single JSON object { "pass": <boolean>, "reasons": ["<specific finding>", ...], "taskKey": "<the slice you reviewed>", "exhausted": <boolean, optional — true only when you have nothing left to check> }. `pass: false` reverts the reviewed slice so it rebuilds. The content of FLOW_REVIEW.json must be ONLY that JSON object (no prose, no markdown fences). Do not edit the slice — you review, you do not fix.',
|
|
33
33
|
].join(' ')
|
|
34
34
|
|
|
35
|
+
export const FLOW_CODEX_REVIEWER_PROMPT = [
|
|
36
|
+
'THINKPOOL FLOW REVIEW — you are a non-mutating adversarial reviewer. Your native sandbox lets you write only inside your OWN disposable reviewer worktree. The reviewed builder worktree is outside that writable root: never edit, commit, revert, install into, or repair it.',
|
|
37
|
+
'For any install/build/test that produces files, copy the reviewed source into a scratch directory under your own current worktree first, then run the write-producing command against that scratch copy. Read-only inspection may target the reviewed worktree directly.',
|
|
38
|
+
'Independently reproduce every acceptance criterion and hunt edge cases, reloads, repeated actions, concurrency, malformed input, and error paths. A green happy path is only the floor. If a claim cannot be reproduced, reject it with the exact command/output.',
|
|
39
|
+
'Submit each review round by calling the ThinkPool submit_flow_review MCP tool with a JSON verdict: {"pass":boolean,"reasons":["specific evidence"],"taskKey":"allowed reviewed task","exhausted":boolean}. Do not write FLOW_REVIEW.json and do not call mark_flow_done. A pass without exhausted=true may trigger another bounded hunt round; a failure triggers the room revert path.',
|
|
40
|
+
].join(' ')
|
|
41
|
+
|
|
35
42
|
// Parse a reviewer's raw verdict. Accepts an object OR a JSON string (optionally
|
|
36
43
|
// ```json-fenced). Tolerant of the fence the model sometimes wraps; throws on garbage.
|
|
37
44
|
// Returns { pass, reasons } — pass coerced to boolean, reasons normalized to string[].
|
package/flow-task-graph.mjs
CHANGED
|
@@ -173,6 +173,27 @@ export function normalizePlanOutput (raw) {
|
|
|
173
173
|
return { summary, tasks }
|
|
174
174
|
}
|
|
175
175
|
|
|
176
|
+
// Codex review protocol deliberately maps one reviewer lane to one builder target.
|
|
177
|
+
// Claude's legacy Flow plans may review several deps and remain unchanged.
|
|
178
|
+
export function validatePlanForRuntime (plan, runtime = 'claude') {
|
|
179
|
+
if (runtime !== 'codex') return plan
|
|
180
|
+
for (const task of plan?.tasks || []) {
|
|
181
|
+
if (task.sliceType === SLICE_TYPE.review && task.deps.length !== 1) {
|
|
182
|
+
throw new Error(`Codex review task "${task.key}" must depend on exactly one builder task`)
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
return plan
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
export function validReviewTargetShape (task, runtime = 'claude') {
|
|
189
|
+
if (!task || task.slice_type !== SLICE_TYPE.review) return true
|
|
190
|
+
return runtime !== 'codex' || (Array.isArray(task.deps) && task.deps.length === 1)
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
export function legacyBuilderCompletionAllowed ({ runtime, flowRole, eventKind, eventSubtype, interrupted = false } = {}) {
|
|
194
|
+
return runtime === 'claude' && flowRole === 'builder' && !interrupted && eventKind === 'result' && eventSubtype === 'success'
|
|
195
|
+
}
|
|
196
|
+
|
|
176
197
|
// ── Flow lane model tiers ───────────────────────────────────────────────────
|
|
177
198
|
// Spec: docs/specs/2026-07-03-flow-lane-model-tiers.md
|
|
178
199
|
// The conductor thinks; the worker lanes type. Running every lane on the same
|
package/interrupted-resume.mjs
CHANGED
|
@@ -11,6 +11,27 @@ export function sendInterruptedContinue(entry) {
|
|
|
11
11
|
catch { return false }
|
|
12
12
|
}
|
|
13
13
|
|
|
14
|
+
// A text recap is the fallback when the native thread cannot be resumed. Claude
|
|
15
|
+
// already has a live init event to trigger this; Codex does not -- sending the
|
|
16
|
+
// recap is what starts its fresh exec process. Consume atomically, but restore it
|
|
17
|
+
// if delivery was refused so the next human turn can still prepend the context.
|
|
18
|
+
export function dispatchPendingRecap(entry) {
|
|
19
|
+
const recap = typeof entry?.pendingRecap === 'string' ? entry.pendingRecap : ''
|
|
20
|
+
if (!recap.trim()) return false
|
|
21
|
+
entry.pendingRecap = null
|
|
22
|
+
try { entry.flush?.() } catch { /* best effort */ }
|
|
23
|
+
try {
|
|
24
|
+
if (entry.session?.sendTurn?.(recap) === false) {
|
|
25
|
+
entry.pendingRecap = recap
|
|
26
|
+
return false
|
|
27
|
+
}
|
|
28
|
+
return true
|
|
29
|
+
} catch {
|
|
30
|
+
entry.pendingRecap = recap
|
|
31
|
+
return false
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
|
|
14
35
|
export function armInterruptedResume(entry) {
|
|
15
36
|
if (!entry?.session) return 'unavailable'
|
|
16
37
|
if (entry.runtime === 'codex') return sendInterruptedContinue(entry) ? 'sent' : 'unavailable'
|
|
@@ -18,6 +39,37 @@ export function armInterruptedResume(entry) {
|
|
|
18
39
|
return 'armed'
|
|
19
40
|
}
|
|
20
41
|
|
|
42
|
+
export function recoverInterruptedTurn(entry, { resumable = false, recap = '' } = {}) {
|
|
43
|
+
if (!entry?.session) return 'unavailable'
|
|
44
|
+
if (resumable) return armInterruptedResume(entry)
|
|
45
|
+
if (!String(recap || '').trim()) return 'unavailable'
|
|
46
|
+
entry.pendingRecap = String(recap)
|
|
47
|
+
if (entry.runtime === 'codex') return dispatchPendingRecap(entry) ? 'recap-sent' : 'unavailable'
|
|
48
|
+
return 'recap-armed'
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
export function isMissingResumeError(message) {
|
|
52
|
+
return /No conversation found|no rollout found for thread id|thread\/resume[^\n]*failed[^\n]*no rollout/i.test(String(message || ''))
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
// Bound the stale-native-thread self-heal at the ownership seam. The old entry is
|
|
56
|
+
// marked before teardown/reopen so duplicate error delivery can never fork another
|
|
57
|
+
// process. `reopen` returns the new structured entry; Codex consumes the recap
|
|
58
|
+
// immediately because a fresh exec has no idle init event.
|
|
59
|
+
export function recoverMissingResumeOnce(entry, { message, recap = '', reopen } = {}) {
|
|
60
|
+
if (!entry || entry.recovered || !isMissingResumeError(message)) return { recovered: false, delivery: 'ignored' }
|
|
61
|
+
entry.recovered = true
|
|
62
|
+
try { entry.session?.end?.() } catch { /* already dead */ }
|
|
63
|
+
let recovered = null
|
|
64
|
+
try { recovered = reopen?.(String(recap || '')) || null } catch { /* caller surfaces reopen failure */ }
|
|
65
|
+
if (!recovered) return { recovered: true, delivery: 'unavailable', entry: null }
|
|
66
|
+
return {
|
|
67
|
+
recovered: true,
|
|
68
|
+
delivery: recoverInterruptedTurn(recovered, { resumable: false, recap }),
|
|
69
|
+
entry: recovered,
|
|
70
|
+
}
|
|
71
|
+
}
|
|
72
|
+
|
|
21
73
|
// A real person speaking always wins the race. Otherwise a human turn that starts a
|
|
22
74
|
// cold Claude/Codex runtime can trigger its init event and enqueue a stale second
|
|
23
75
|
// `continue` behind the person's actual request.
|
|
@@ -26,4 +78,3 @@ export function supersedeInterruptedResume(entry) {
|
|
|
26
78
|
entry.pendingAutoResume = false
|
|
27
79
|
return true
|
|
28
80
|
}
|
|
29
|
-
|
package/lane-worktree.mjs
CHANGED
|
@@ -12,11 +12,26 @@ export function createManagedLaneWorktree({ terminalId, cwd = process.cwd(), git
|
|
|
12
12
|
if (!root) throw new Error('not a git repository')
|
|
13
13
|
const dir = path.join(root, '.thinkpool', 'worktrees', `terminal-${short}`)
|
|
14
14
|
const branch = `thinkpool/terminal/${short}`
|
|
15
|
+
const base = resolveLaneBase({ root, git })
|
|
15
16
|
fsImpl.mkdirSync(path.dirname(dir), { recursive: true })
|
|
16
|
-
git(['worktree', 'add', '-b', branch, dir,
|
|
17
|
+
git(['worktree', 'add', '-b', branch, dir, base], root)
|
|
17
18
|
return { terminalId: id, root, dir, branch }
|
|
18
19
|
}
|
|
19
20
|
|
|
21
|
+
// Ordinary lanes should start from the repository's upstream default branch, not
|
|
22
|
+
// whichever parked/stale branch happens to be checked out in the bridge process.
|
|
23
|
+
// Repositories without a remote remain useful: fall back through conventional local
|
|
24
|
+
// defaults and finally HEAD.
|
|
25
|
+
function resolveLaneBase({ root, git }) {
|
|
26
|
+
for (const candidate of ['origin/HEAD', 'origin/main', 'origin/master', 'main', 'master', 'HEAD']) {
|
|
27
|
+
try {
|
|
28
|
+
const resolved = String(git(['rev-parse', '--verify', '--quiet', candidate], root) || '').trim()
|
|
29
|
+
if (resolved) return candidate
|
|
30
|
+
} catch { /* try the next base */ }
|
|
31
|
+
}
|
|
32
|
+
return 'HEAD'
|
|
33
|
+
}
|
|
34
|
+
|
|
20
35
|
export function removeManagedLaneWorktree({ terminalId, managedWorktree, git = runGit } = {}) {
|
|
21
36
|
const id = String(terminalId || '')
|
|
22
37
|
const rec = managedWorktree && typeof managedWorktree === 'object' ? managedWorktree : null
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "thinkpool-pair",
|
|
3
|
-
"version": "0.7.
|
|
3
|
+
"version": "0.7.222",
|
|
4
4
|
"description": "Share a local coding-agent CLI (Claude Code, Codex, Gemini, Aider, …) into a ThinkPool Code room, live.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -36,6 +36,8 @@
|
|
|
36
36
|
"flow-conductor.mjs",
|
|
37
37
|
"flow-worktree.mjs",
|
|
38
38
|
"flow-task-graph.mjs",
|
|
39
|
+
"flow-models.mjs",
|
|
40
|
+
"flow-host-revert.mjs",
|
|
39
41
|
"flow-preview.mjs",
|
|
40
42
|
"viewport.mjs",
|
|
41
43
|
"design-edit.mjs",
|