thinkpool-pair 0.7.246 → 0.7.248
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bridge.mjs +95 -20
- package/dispatch-lease.mjs +37 -0
- package/flow-models.mjs +26 -2
- package/flow-task-graph.mjs +3 -3
- package/hermes-acp-bootstrap.py +196 -0
- package/hermes-policy.mjs +85 -0
- package/hermes-probe.mjs +32 -2
- package/hermes-session.mjs +62 -13
- package/package.json +5 -1
- package/review-check.mjs +155 -0
- package/runtime-registry.mjs +1 -1
package/bridge.mjs
CHANGED
|
@@ -54,6 +54,7 @@ import { withMcpSessionFactory } from './codex-mcp-http.mjs'
|
|
|
54
54
|
import { startStructuredSession } from './runtime-session.mjs'
|
|
55
55
|
import { defaultStructuredMode, shouldDeferStructuredRuntime, structuredRuntimeForCommand, structuredRuntimeMetadata, structuredRuntimeSupportsMode } from './runtime-registry.mjs'
|
|
56
56
|
import { probeHermesRuntime } from './hermes-probe.mjs'
|
|
57
|
+
import { hermesRequiredMcpTools, hermesRoleFor } from './hermes-policy.mjs'
|
|
57
58
|
import { canonicalRoomFilePath, waitForNativeImages } from './codex-images.mjs'
|
|
58
59
|
import { createManagedLaneWorktree, removeManagedLaneWorktree } from './lane-worktree.mjs'
|
|
59
60
|
import { commandOnPath } from './agent-detect.mjs'
|
|
@@ -82,6 +83,7 @@ function stopFlowPreviews (flowId, laneId = null) {
|
|
|
82
83
|
}
|
|
83
84
|
import { FLOW_REVIEWER_PROMPT, FLOW_CODEX_REVIEWER_PROMPT, revertLane, parseReviewVerdict, reviewVerdictToReflection } from './flow-review.mjs'
|
|
84
85
|
import { reviewGateDecision } from './flow-review-gate.mjs'
|
|
86
|
+
import { readReviewFile, runReviewCheck } from './review-check.mjs'
|
|
85
87
|
import { pairAdjudicationPrompt, reviewReflectionDecision, REVIEW_DEFAULTS } from './flow-review-reflect.mjs'
|
|
86
88
|
import { mergeWorktrees, inlineSingleHtml, initRepo } from './flow-assembly.mjs'
|
|
87
89
|
import { canDispatch, FLOW_LIMITS, makeBudget, recordSpend, killSwitchEnv } from './flow-budget.mjs'
|
|
@@ -108,6 +110,7 @@ const flowRedispatch = new Map()
|
|
|
108
110
|
const flowBudgets = new Map()
|
|
109
111
|
import { formatPeek, PEEK, siblingsOf, resolveSibling, crossPostDecision, CROSSPOST, spawnDecision, SPAWN, CROSSROOM, formatPairRoster, crossRoomPostDecision, formatRoomNow, formatClosableHint, closeLaneDecision, laneStatusOf, nativeClaudeFallbackHint, buildDispatchPreview, authorizeDispatchApproval, verifyDurableDispatchAuthority } from './cross-terminal.mjs'
|
|
110
112
|
import { dispatchPendingRecap, recoverInterruptedTurn, recoverMissingResumeOnce, sendInterruptedContinue, shouldAttemptNativeResume, supersedeInterruptedResume } from './interrupted-resume.mjs'
|
|
113
|
+
import { beginDispatchLease, finishDispatchLease, isDispatchLeaseCurrent, supersedeDispatchLease } from './dispatch-lease.mjs'
|
|
111
114
|
import { turnInFlight } from './update-gate.mjs'
|
|
112
115
|
import { saveSession, flushSession, deleteSession, loadAll, canResume, loadPtyId, savePtyId, loadNames, saveNames, appendDurableEvents, seedDurableEvents, readDurablePage, readDurableOldestSeq, hasDurableArchive, servesHistoryPage } from './session-store.mjs'
|
|
113
116
|
import { stampEvent, makeSeqCounter, maxSeq, seqable, capReplayEvents, chunkReplayEvents, boundEventForBroadcast, inlineImageBlocks, ImageEventQueue, imageQueueConfig, uploadCodeImage as uploadCodeImageRequest, usageReportLine, codexUsageReportLine, buildRecapFromLog, RECAP_CAP, trimmedBeforeSeq, firstSeq } from './event-id.mjs'
|
|
@@ -1727,6 +1730,12 @@ function restoredTurnOpen(log) {
|
|
|
1727
1730
|
// minutes-scale. Plain `git worktree list` output (path · sha · [branch]) is the
|
|
1728
1731
|
// readable shape an agent acts on. Fail-quiet [] — a snapshot must never break a turn.
|
|
1729
1732
|
const _wtCache = new Map() // cwd → { at, list }
|
|
1733
|
+
function immutableReviewSnapshot(cwd, taskKey = 'parent') {
|
|
1734
|
+
const source = path.resolve(cwd || process.cwd())
|
|
1735
|
+
const sha = execFileSync('git', ['-C', source, 'rev-parse', '--verify', 'HEAD^{commit}'], { encoding: 'utf8', timeout: 3000, stdio: ['ignore', 'pipe', 'ignore'] }).trim()
|
|
1736
|
+
if (!/^[0-9a-f]{40}$/i.test(sha)) throw new Error('parent review source has no immutable HEAD')
|
|
1737
|
+
return { taskKey, cwd: source, sha }
|
|
1738
|
+
}
|
|
1730
1739
|
function worktreeSnapshot(cwd) {
|
|
1731
1740
|
const key = cwd || process.cwd()
|
|
1732
1741
|
const hit = _wtCache.get(key)
|
|
@@ -1744,7 +1753,7 @@ function worktreeSnapshot(cwd) {
|
|
|
1744
1753
|
// relay STRUCTURED events. onEvent → broadcast `code-event` + print locally +
|
|
1745
1754
|
// persist to the host file; tool calls round-trip through the perm card; the
|
|
1746
1755
|
// rolling log replays to joiners and survives bridge restarts (session-store).
|
|
1747
|
-
function openStructured({ id, runtime = 'claude', model, models, effort, resume, log, commands, mode, spawnedBy, spawnDepth, cascadeRole, hop, sideParent, sideTask, pendingSideContexts, rolePrompt, flowSessionId, flowTaskKey, flowRole, flowReviewTarget, flowReviewTargets, flowReviewRound, dispatchBaseSha, revertTarget, cwd, managedWorktree, reviewSliceRoots, openedAt, defer, provider, carryRecap, lastUsage }) {
|
|
1756
|
+
function openStructured({ id, runtime = 'claude', model, models, effort, resume, log, commands, mode, spawnedBy, spawnDepth, cascadeRole, hop, sideParent, sideTask, pendingSideContexts, rolePrompt, sliceType, flowSessionId, flowTaskKey, flowRole, flowReviewTarget, flowReviewTargets, flowReviewSnapshots, flowReviewRound, dispatchBaseSha, revertTarget, cwd, managedWorktree, reviewSliceRoots, openedAt, defer, provider, carryRecap, lastUsage }) {
|
|
1748
1757
|
if (sessions.has(id)) return
|
|
1749
1758
|
runtime = structuredRuntimeMetadata(runtime) ? runtime : 'claude'
|
|
1750
1759
|
// No explicit mode → a sensible default per runtime (see defaultModeForRuntime):
|
|
@@ -1797,9 +1806,11 @@ function openStructured({ id, runtime = 'claude', model, models, effort, resume,
|
|
|
1797
1806
|
? (laneModel || null)
|
|
1798
1807
|
: (laneModel || providerNameMap()[provider] || provider),
|
|
1799
1808
|
provider: provider || null, spawnedBy: spawnedBy || undefined, spawnDepth: structuralDepth, cascadeRole: cascadeRole === 'conductor' || cascadeRole === 'worker' ? cascadeRole : null, hop: initialHop, sideParent: sideParent || undefined, sideTask: sideTask || undefined, pendingSideContexts: Array.isArray(pendingSideContexts) ? pendingSideContexts.filter(Boolean).slice(-4) : [], flowSessionId: flowSessionId || null, flowTaskKey: flowTaskKey || null, cwd: cwd || null, managedWorktree: managedWorktree || null,
|
|
1809
|
+
sliceType: sliceType === 'review' ? 'review' : null,
|
|
1800
1810
|
flowRole: flowRole || (flowSessionId ? (flowTaskKey ? ((flowReviewTargets?.length || reviewSliceRoots?.length) ? 'reviewer' : 'builder') : 'conductor') : null),
|
|
1801
1811
|
flowReviewTarget: flowReviewTarget || null,
|
|
1802
1812
|
flowReviewTargets: Array.isArray(flowReviewTargets) && flowReviewTargets.length ? flowReviewTargets.filter(Boolean) : (flowReviewTarget ? [flowReviewTarget] : []),
|
|
1813
|
+
flowReviewSnapshots: Array.isArray(flowReviewSnapshots) ? flowReviewSnapshots.filter((item) => item?.taskKey && item?.sha && item?.cwd) : [],
|
|
1803
1814
|
flowReviewRound: Number.isInteger(flowReviewRound) && flowReviewRound >= 0 ? flowReviewRound : 0,
|
|
1804
1815
|
dispatchBaseSha: dispatchBaseSha || null,
|
|
1805
1816
|
revertTarget: revertTarget || null,
|
|
@@ -2046,7 +2057,7 @@ function openStructured({ id, runtime = 'claude', model, models, effort, resume,
|
|
|
2046
2057
|
// restart. Without this, sessionData omitted it → on restart the resumed session
|
|
2047
2058
|
// re-launched on the host default (Opus) regardless of the last switch, and the
|
|
2048
2059
|
// switch looked like it "never changed the model" (Max 2026-07-02). Restored below.
|
|
2049
|
-
const sessionData = () => ({ sessionId: entry.session?.sessionId || null, runtime: entry.runtime, log: entry.log, commands: entry.commands, mode: entry.mode, effort: entry.effort, model: entry.model || null, models: entry.runtime === 'hermes' ? (entry.models || []) : undefined, provider: entry.provider || null, spawnedBy: entry.spawnedBy, spawnDepth: entry.spawnDepth || 0, cascadeRole: entry.cascadeRole || null, hop: entry.hop || 0, sideParent: entry.sideParent, sideTask: entry.sideTask, pendingSideContexts: entry.pendingSideContexts, flowSessionId: entry.flowSessionId, flowTaskKey: entry.flowTaskKey, flowRole: entry.flowRole || null, flowReviewTargets: entry.flowReviewTargets || [], flowReviewRound: entry.flowReviewRound || 0, dispatchBaseSha: entry.dispatchBaseSha || null, cwd: entry.cwd, managedWorktree: entry.managedWorktree, rolePrompt: entry.rolePrompt, flowReviewTarget: entry.flowReviewTarget || null, revertTarget: entry.revertTarget || null, reviewSliceRoots: entry.reviewSliceRoots || [], openedAt: entry.openedAt || null, lastUsage: entry.lastUsage || null, carryRecap: entry.pendingRecap || null })
|
|
2060
|
+
const sessionData = () => ({ sessionId: entry.session?.sessionId || null, runtime: entry.runtime, log: entry.log, commands: entry.commands, mode: entry.mode, effort: entry.effort, model: entry.model || null, models: entry.runtime === 'hermes' ? (entry.models || []) : undefined, provider: entry.provider || null, spawnedBy: entry.spawnedBy, spawnDepth: entry.spawnDepth || 0, cascadeRole: entry.cascadeRole || null, hop: entry.hop || 0, sideParent: entry.sideParent, sideTask: entry.sideTask, pendingSideContexts: entry.pendingSideContexts, sliceType: entry.sliceType, flowSessionId: entry.flowSessionId, flowTaskKey: entry.flowTaskKey, flowRole: entry.flowRole || null, flowReviewTargets: entry.flowReviewTargets || [], flowReviewSnapshots: entry.flowReviewSnapshots || [], flowReviewRound: entry.flowReviewRound || 0, dispatchBaseSha: entry.dispatchBaseSha || null, cwd: entry.cwd, managedWorktree: entry.managedWorktree, rolePrompt: entry.rolePrompt, flowReviewTarget: entry.flowReviewTarget || null, revertTarget: entry.revertTarget || null, reviewSliceRoots: entry.reviewSliceRoots || [], openedAt: entry.openedAt || null, lastUsage: entry.lastUsage || null, carryRecap: entry.pendingRecap || null })
|
|
2050
2061
|
const persist = () => saveSession(room, id, sessionData())
|
|
2051
2062
|
// Synchronous flush of this session's record. Used on open (so a brand-new session
|
|
2052
2063
|
// has a file under its id BEFORE its first event — surviving a restart inside the
|
|
@@ -2074,7 +2085,7 @@ function openStructured({ id, runtime = 'claude', model, models, effort, resume,
|
|
|
2074
2085
|
return { error: `Could not open a Hermes worker on ${JSON.stringify(args.model)} — that exact model is not in this parent session's ACP catalog.` }
|
|
2075
2086
|
}
|
|
2076
2087
|
if (runtime === 'hermes' && args?.mode && !['default', 'acceptEdits'].includes(args.mode)) {
|
|
2077
|
-
return { error: `Hermes ACP
|
|
2088
|
+
return { error: `Hermes ACP only exposes default/acceptEdits as user modes; Flow roles use a bridge-owned process-local tool policy.` }
|
|
2078
2089
|
}
|
|
2079
2090
|
if (runtime === 'claude' && !args?.provider && args?.model && /^gpt-/i.test(args.model)) {
|
|
2080
2091
|
return { error: `Could not open a Claude terminal on Codex model ${JSON.stringify(args.model)}. Choose runtime="codex" or a Claude model.` }
|
|
@@ -2342,7 +2353,7 @@ function openStructured({ id, runtime = 'claude', model, models, effort, resume,
|
|
|
2342
2353
|
// not receive this tool, so the hierarchy is capability-enforced.
|
|
2343
2354
|
...(canSpawnWorkers ? [tool(
|
|
2344
2355
|
'spawn_terminal',
|
|
2345
|
-
'Open a visible WORKER SUB-TERMINAL (Claude, Codex, or—only from a Hermes parent—Hermes) in the ThinkPool Ensemble. spawn_terminal is worker-only: it can never create a Cascade conductor or main terminal.
|
|
2356
|
+
'Open a visible WORKER SUB-TERMINAL (Claude, Codex, or—only from a Hermes parent—Hermes) in the ThinkPool Ensemble. spawn_terminal is worker-only: it can never create a Cascade conductor or main terminal. For ordinary workers use an appropriate non-Sol tier: gpt-5.6-luna or gpt-5.4-mini for scaffold/search, gpt-5.6-terra for feature/fix, and a balanced tier for adversarial review; never select Sol for routine workers. A Hermes sliceType=review lane is structurally reviewer-scoped before ACP startup: it can inspect files and use the bridge review-check tool, but has no terminal, write, patch, delegate, browser, skill, memory, or session-history schema. Pass sliceType=scaffold for mechanical work, feature/fix for builders, and review for adversarial verification. Give the worker a bounded initial task, collect its result with read_terminal, then ALWAYS close_terminal it.',
|
|
2346
2357
|
{
|
|
2347
2358
|
name: z.string().max(80).optional().describe('a short label for the new lane so the room (and you) can find it, e.g. "Research: topologies"'),
|
|
2348
2359
|
task: z.string().optional().describe('an initial task to hand the new lane immediately; omit to open it idle'),
|
|
@@ -2391,9 +2402,16 @@ function openStructured({ id, runtime = 'claude', model, models, effort, resume,
|
|
|
2391
2402
|
})
|
|
2392
2403
|
} catch { return okText('Dispatch preview could not be built safely. No lane was created.') }
|
|
2393
2404
|
const permissionId = randomUUID()
|
|
2405
|
+
const dispatchAdmission = beginDispatchLease(entry, permissionId)
|
|
2406
|
+
if (!dispatchAdmission.ok) {
|
|
2407
|
+
return okText('A dispatch choice is already awaiting approval in the room. Do not retry or switch runtimes; wait for that choice to be approved or canceled, or for a new human turn to supersede it.')
|
|
2408
|
+
}
|
|
2409
|
+
const dispatchLease = dispatchAdmission.lease
|
|
2410
|
+
try {
|
|
2394
2411
|
let approval = { decision: 'deny' }
|
|
2395
2412
|
try { approval = await requestDispatchApproval({ id: permissionId, input: effectiveArgs, dispatchPreview: preview }) } catch { /* fail closed */ }
|
|
2396
2413
|
if (approval?.decision !== 'allow') return okText('Dispatch canceled. No lane or worktree was created.')
|
|
2414
|
+
if (!isDispatchLeaseCurrent(entry, dispatchLease)) return okText('That dispatch choice was superseded by a newer turn. No lane or worktree was created.')
|
|
2397
2415
|
const currentNow = Date.now()
|
|
2398
2416
|
const current = dispatchContext(currentNow)
|
|
2399
2417
|
const durable = await readDurableDispatchAuthority({
|
|
@@ -2406,6 +2424,7 @@ function openStructured({ id, runtime = 'claude', model, models, effort, resume,
|
|
|
2406
2424
|
},
|
|
2407
2425
|
})
|
|
2408
2426
|
if (!durable.ok) return okText('Dispatch approval could not be verified from the durable room authority. No lane was created.')
|
|
2427
|
+
if (!isDispatchLeaseCurrent(entry, dispatchLease)) return okText('That dispatch choice was superseded by a newer turn. No lane or worktree was created.')
|
|
2409
2428
|
const authorization = authorizeDispatchApproval({
|
|
2410
2429
|
preview,
|
|
2411
2430
|
approvedFingerprint: durable.fingerprint,
|
|
@@ -2438,10 +2457,18 @@ function openStructured({ id, runtime = 'claude', model, models, effort, resume,
|
|
|
2438
2457
|
// otherwise inherit, but fall plan → default (an autonomous worker doesn't plan-gate).
|
|
2439
2458
|
const childSpawnDepth = (entry.spawnDepth || 0) + 1
|
|
2440
2459
|
const childHop = (entry.hop || 0) + 1
|
|
2460
|
+
let manualReviewSnapshots = []
|
|
2461
|
+
if (args?.sliceType === 'review') {
|
|
2462
|
+
// A manual reviewer never accepts a model-supplied path/ref. It
|
|
2463
|
+
// reads this exact parent HEAD through read_review_file and runs
|
|
2464
|
+
// checks only through its archived `parent` target.
|
|
2465
|
+
try { manualReviewSnapshots = [immutableReviewSnapshot(entry.cwd || process.cwd())] }
|
|
2466
|
+
catch (error) { return okText(`Review lane was not opened: could not pin the spawning parent’s HEAD (${error?.message || error}).`) }
|
|
2467
|
+
}
|
|
2441
2468
|
// Stamp ownership/depth BEFORE the runtime starts so its first system
|
|
2442
2469
|
// preamble is truthful. Mutating ne.spawnedBy after openStructured was
|
|
2443
2470
|
// too late: Codex/Claude had already booted with the top-level wording.
|
|
2444
|
-
openStructured({ id: newId, runtime: resolved.runtime, model: resolved.model, provider: resolved.provider, mode: resolved.mode, spawnedBy: id, spawnDepth: childSpawnDepth, cascadeRole: 'worker', hop: childHop })
|
|
2471
|
+
openStructured({ id: newId, runtime: resolved.runtime, model: resolved.model, provider: resolved.provider, mode: resolved.mode, sliceType: args?.sliceType, flowReviewTargets: manualReviewSnapshots.map((item) => item.taskKey), flowReviewSnapshots: manualReviewSnapshots, spawnedBy: id, spawnDepth: childSpawnDepth, cascadeRole: 'worker', hop: childHop })
|
|
2445
2472
|
const ne = sessions.get(newId)
|
|
2446
2473
|
if (!ne) { if (args?.name) { delete termNames[newId]; saveNames(room, termNames) } return okText('Could not open a new lane — the terminal cap may have just been reached. Close one and retry.') }
|
|
2447
2474
|
ne.peekCount = 0; ne.postCount = 0; ne.spawnTimes = []
|
|
@@ -2449,13 +2476,19 @@ function openStructured({ id, runtime = 'claude', model, models, effort, resume,
|
|
|
2449
2476
|
if (args?.task) {
|
|
2450
2477
|
// The user-visible relay reinforces (but never defines) the structural
|
|
2451
2478
|
// role already injected by the system preamble above.
|
|
2452
|
-
const
|
|
2479
|
+
const reviewTarget = manualReviewSnapshots.length
|
|
2480
|
+
? `\n\n[Immutable review target: parent at spawn time — target key "parent", pinned HEAD ${manualReviewSnapshots[0].sha}. Use read_review_file with target="parent" for source and run_review_check with target="parent" for fixed checks. Do not infer a target path/ref or mutate anything.]`
|
|
2481
|
+
: ''
|
|
2482
|
+
const msg = `[Task from terminal ${fromRef}'s agent — relayed via ThinkPool Ensemble; you are its WORKER SUB-TERMINAL, never a main terminal or Cascade conductor]\n${args.task}${reviewTarget}`
|
|
2453
2483
|
const evt = { kind: 'you', text: msg, by: `terminal ${fromRef} (agent)`, crosspost: true, relaySourceName: termNames[id] || undefined }
|
|
2454
2484
|
stampEvent(evt); pushLog(ne, evt); bcast('code-event', { term: newId, evt })
|
|
2455
2485
|
try { ne.session.sendTurn(msg) } catch { return okText(`Opened lane ${newRef}${args?.name ? ` ("${args.name}")` : ''}, but it may still be starting — could not hand off the task. Try post_to_terminal shortly.`) }
|
|
2456
2486
|
return okText(`Opened agent lane ${newRef}${args?.name ? ` ("${args.name}")` : ''} and handed it the task. It runs in its own lane — check back with read_terminal, then close_terminal when done.`)
|
|
2457
2487
|
}
|
|
2458
2488
|
return okText(`Opened idle agent lane ${newRef}${args?.name ? ` ("${args.name}")` : ''}. Hand it work with post_to_terminal, or a person can type into it.`)
|
|
2489
|
+
} finally {
|
|
2490
|
+
finishDispatchLease(entry, dispatchLease)
|
|
2491
|
+
}
|
|
2459
2492
|
},
|
|
2460
2493
|
)] : []),
|
|
2461
2494
|
// Research lane — run a REAL multi-source search + adversarial verification and
|
|
@@ -2564,6 +2597,17 @@ function openStructured({ id, runtime = 'claude', model, models, effort, resume,
|
|
|
2564
2597
|
return { content: [{ type: 'text', text: result.message }] }
|
|
2565
2598
|
},
|
|
2566
2599
|
)] : []),
|
|
2600
|
+
...(entry.flowRole === 'reviewer' || entry.sliceType === 'review' ? [tool(
|
|
2601
|
+
'read_review_file',
|
|
2602
|
+
'Read one regular file from the declared immutable review target. This always reads the target’s pinned Git object, never a live worktree; target and path are validated and the response is byte-bounded.',
|
|
2603
|
+
{ target: z.string(), path: z.string() },
|
|
2604
|
+
async (args) => ({ content: [{ type: 'text', text: JSON.stringify(await readReviewFile({ target: args?.target, filePath: args?.path, snapshots: entry.flowReviewSnapshots })) }] }),
|
|
2605
|
+
), tool(
|
|
2606
|
+
'run_review_check',
|
|
2607
|
+
'Run one fixed verification check against an immutable git-archived review target. This accepts only the declared target and a fixed check id; it never accepts a shell command, cwd, environment, executable, or arguments. The check runs in a scratch archive, never in a builder worktree.',
|
|
2608
|
+
{ target: z.string(), checkId: z.string() },
|
|
2609
|
+
async (args) => ({ content: [{ type: 'text', text: JSON.stringify(await runReviewCheck({ target: args?.target, checkId: args?.checkId, snapshots: entry.flowReviewSnapshots })) }] }),
|
|
2610
|
+
)] : []),
|
|
2567
2611
|
],
|
|
2568
2612
|
})
|
|
2569
2613
|
const peekServer = withMcpSessionFactory(createPeekServer)
|
|
@@ -2663,8 +2707,9 @@ function openStructured({ id, runtime = 'claude', model, models, effort, resume,
|
|
|
2663
2707
|
// role-scoped tools through local /tools before a user turn. Claude/Codex
|
|
2664
2708
|
// already fail startup when their required MCP handshake fails.
|
|
2665
2709
|
requiredMcpTools: runtime === 'hermes'
|
|
2666
|
-
?
|
|
2710
|
+
? hermesRequiredMcpTools(hermesRoleFor({ flowRole: entry.flowRole, sliceType: entry.sliceType }), { canSpawnWorkers })
|
|
2667
2711
|
: undefined,
|
|
2712
|
+
hermesRole: runtime === 'hermes' ? hermesRoleFor({ flowRole: entry.flowRole, sliceType: entry.sliceType }) : undefined,
|
|
2668
2713
|
// Tier C precheck — the PreToolUse gate calls this BEFORE raising a card, so a
|
|
2669
2714
|
// disabled/looping/over-cap post never bothers a person. Closes over `entry`.
|
|
2670
2715
|
crossPostGate: () => crossPostDecision({ hop: entry.hop || 0, postCount: entry.postCount || 0, disabled: process.env.TP_CROSSPOST_OFF === '1' }),
|
|
@@ -2717,9 +2762,9 @@ function openStructured({ id, runtime = 'claude', model, models, effort, resume,
|
|
|
2717
2762
|
openStructured({
|
|
2718
2763
|
id, runtime: entry.runtime, model: entry.model || model, effort: entry.effort,
|
|
2719
2764
|
provider: entry.provider, log: entry.log, commands: entry.commands, mode: entry.mode,
|
|
2720
|
-
spawnedBy: entry.spawnedBy, spawnDepth: entry.spawnDepth, cascadeRole: entry.cascadeRole, hop: entry.hop, sideParent: entry.sideParent, sideTask: entry.sideTask, pendingSideContexts: entry.pendingSideContexts, flowSessionId: entry.flowSessionId,
|
|
2765
|
+
spawnedBy: entry.spawnedBy, spawnDepth: entry.spawnDepth, cascadeRole: entry.cascadeRole, hop: entry.hop, sideParent: entry.sideParent, sideTask: entry.sideTask, pendingSideContexts: entry.pendingSideContexts, sliceType: entry.sliceType, flowSessionId: entry.flowSessionId,
|
|
2721
2766
|
flowTaskKey: entry.flowTaskKey, flowRole: entry.flowRole, flowReviewTarget: entry.flowReviewTarget,
|
|
2722
|
-
flowReviewTargets: entry.flowReviewTargets, flowReviewRound: entry.flowReviewRound,
|
|
2767
|
+
flowReviewTargets: entry.flowReviewTargets, flowReviewSnapshots: entry.flowReviewSnapshots, flowReviewRound: entry.flowReviewRound,
|
|
2723
2768
|
dispatchBaseSha: entry.dispatchBaseSha, revertTarget: entry.revertTarget, cwd: entry.cwd,
|
|
2724
2769
|
managedWorktree: entry.managedWorktree, rolePrompt: entry.rolePrompt,
|
|
2725
2770
|
reviewSliceRoots: entry.reviewSliceRoots, openedAt: entry.openedAt,
|
|
@@ -3077,6 +3122,10 @@ function pendingResolution(pending, payload = {}) {
|
|
|
3077
3122
|
|
|
3078
3123
|
function drainPending(s) {
|
|
3079
3124
|
if (!s?.pending) return
|
|
3125
|
+
// Invalidate the execution fence before settling promises. A dispatch that
|
|
3126
|
+
// was already approved but is still re-reading durable authority must also
|
|
3127
|
+
// fail closed when this terminal stops or its turn settles.
|
|
3128
|
+
supersedeDispatchLease(s)
|
|
3080
3129
|
for (const [, p] of s.pending) {
|
|
3081
3130
|
if (p?.timer) clearTimeout(p.timer)
|
|
3082
3131
|
try { p.resolve(pendingResolution(p, { decision: 'deny' })) } catch { /* noop */ }
|
|
@@ -3086,6 +3135,17 @@ function drainPending(s) {
|
|
|
3086
3135
|
s.permNotifier?.clearAll()
|
|
3087
3136
|
}
|
|
3088
3137
|
|
|
3138
|
+
function supersedePendingDispatch(s) {
|
|
3139
|
+
const lease = supersedeDispatchLease(s)
|
|
3140
|
+
if (!lease) return null
|
|
3141
|
+
const pending = s.pending?.get(lease.permissionId)
|
|
3142
|
+
if (pending?.timer) clearTimeout(pending.timer)
|
|
3143
|
+
s.pending?.delete(lease.permissionId)
|
|
3144
|
+
s.permNotifier?.resolve(lease.permissionId)
|
|
3145
|
+
try { pending?.resolve(pendingResolution(pending, { decision: 'deny' })) } catch { /* noop */ }
|
|
3146
|
+
return lease.permissionId
|
|
3147
|
+
}
|
|
3148
|
+
|
|
3089
3149
|
// Switching a terminal to bypassPermissions must RETROACTIVELY clear the cards the
|
|
3090
3150
|
// current turn already raised — setPermissionMode is a streaming control request that
|
|
3091
3151
|
// only applies going forward, so without this a user who flips to bypass mid-turn keeps
|
|
@@ -3155,7 +3215,7 @@ function respawnStructured(id, provider) {
|
|
|
3155
3215
|
// openStructured seed from the TARGET provider's configured model, which is the
|
|
3156
3216
|
// only model this lane was ever asked for. A same-env model change never reaches
|
|
3157
3217
|
// here — that path is an in-place setModel (see provider-switch).
|
|
3158
|
-
const { runtime, log, commands, mode, spawnedBy, spawnDepth, cascadeRole, hop, sideParent, sideTask, pendingSideContexts, flowSessionId, flowTaskKey, flowRole, flowReviewTarget, flowReviewTargets, flowReviewRound, dispatchBaseSha, revertTarget, cwd, managedWorktree, rolePrompt, reviewSliceRoots, openedAt, lastUsage } = s
|
|
3218
|
+
const { runtime, log, commands, mode, spawnedBy, spawnDepth, cascadeRole, hop, sideParent, sideTask, pendingSideContexts, sliceType, flowSessionId, flowTaskKey, flowRole, flowReviewTarget, flowReviewTargets, flowReviewSnapshots, flowReviewRound, dispatchBaseSha, revertTarget, cwd, managedWorktree, rolePrompt, reviewSliceRoots, openedAt, lastUsage } = s
|
|
3159
3219
|
// Context-carry (2026-07-08): a provider switch is not an SDK resume — the new backend
|
|
3160
3220
|
// starts blank mid-conversation. Synthesize a plain-text recap from the VISIBLE log NOW
|
|
3161
3221
|
// (before teardown) and hand it to the fresh session as its first turn so the agent
|
|
@@ -3174,7 +3234,7 @@ function respawnStructured(id, provider) {
|
|
|
3174
3234
|
// sessionData() (provider included) synchronously on open, so a bridge restart
|
|
3175
3235
|
// restores the lane on its CURRENT provider, not the original — and its next
|
|
3176
3236
|
// announce carries the new provider badge (additive {id,name} projection).
|
|
3177
|
-
openStructured({ id, runtime, provider, log, commands, mode, spawnedBy, spawnDepth, cascadeRole, hop, sideParent, sideTask, pendingSideContexts, flowSessionId, flowTaskKey, flowRole, flowReviewTarget, flowReviewTargets, flowReviewRound, dispatchBaseSha, revertTarget, cwd, managedWorktree, rolePrompt, reviewSliceRoots, openedAt, lastUsage, carryRecap })
|
|
3237
|
+
openStructured({ id, runtime, provider, log, commands, mode, spawnedBy, spawnDepth, cascadeRole, hop, sideParent, sideTask, pendingSideContexts, sliceType, flowSessionId, flowTaskKey, flowRole, flowReviewTarget, flowReviewTargets, flowReviewSnapshots, flowReviewRound, dispatchBaseSha, revertTarget, cwd, managedWorktree, rolePrompt, reviewSliceRoots, openedAt, lastUsage, carryRecap })
|
|
3178
3238
|
return true
|
|
3179
3239
|
}
|
|
3180
3240
|
|
|
@@ -3584,6 +3644,15 @@ channel
|
|
|
3584
3644
|
s.hop = 0
|
|
3585
3645
|
s.roomHop = 0 // a human turn is room-hop 0 — clears any injected cross-room hop depth
|
|
3586
3646
|
const text = String(payload.text)
|
|
3647
|
+
// A fresh person-authored turn is a newer dispatch intent boundary. Retract
|
|
3648
|
+
// any older runtime choice and invalidate its lease before it can resume from
|
|
3649
|
+
// a late durable approval. Ordinary tool cards remain untouched: steering a
|
|
3650
|
+
// running turn must not silently answer unrelated permissions.
|
|
3651
|
+
const supersededDispatchId = supersedePendingDispatch(s)
|
|
3652
|
+
if (supersededDispatchId) {
|
|
3653
|
+
bcast('code-perm', { term: payload.term, id: supersededDispatchId, decision: 'deny', name: 'agent' })
|
|
3654
|
+
announce()
|
|
3655
|
+
}
|
|
3587
3656
|
// A human turn supersedes a restart's still-pending auto-continue. Without this,
|
|
3588
3657
|
// their message starts the cold runtime; its init event then queues a second stale
|
|
3589
3658
|
// `continue` behind the request. Codex normally resumes immediately at restore, while
|
|
@@ -3988,7 +4057,7 @@ channel
|
|
|
3988
4057
|
// FL-M6 — restore the flow context (id/role/cwd) so an in-flight flow survives a
|
|
3989
4058
|
// bridge restart: the conductor keeps its subagent-block + plan interception, and
|
|
3990
4059
|
// lanes keep their worktree cwd + the ability to mark done.
|
|
3991
|
-
openStructured({ id: rec.id, runtime: rec.runtime || 'claude', model: rec.model || undefined, models: rec.models, effort: rec.effort, provider: rec.runtime === 'claude' ? rec.provider || undefined : undefined, resume: resumable ? rec.sessionId : undefined, log: rec.log, commands: rec.commands, mode: rec.mode, spawnedBy: rec.spawnedBy, spawnDepth: rec.spawnDepth, cascadeRole: rec.cascadeRole, hop: rec.hop, sideParent: rec.sideParent, sideTask: rec.sideTask, pendingSideContexts: rec.pendingSideContexts, flowSessionId: rec.flowSessionId, flowTaskKey: rec.flowTaskKey, flowRole: rec.flowRole, flowReviewTarget: rec.flowReviewTarget, flowReviewTargets: rec.flowReviewTargets, flowReviewRound: rec.flowReviewRound, dispatchBaseSha: rec.dispatchBaseSha, revertTarget: rec.revertTarget, cwd: rec.cwd, managedWorktree: rec.managedWorktree, rolePrompt: rec.rolePrompt, reviewSliceRoots: rec.reviewSliceRoots, openedAt: rec.openedAt, lastUsage: rec.lastUsage, carryRecap: wasInterrupted ? recoveryRecap : rec.carryRecap,
|
|
4060
|
+
openStructured({ id: rec.id, runtime: rec.runtime || 'claude', model: rec.model || undefined, models: rec.models, effort: rec.effort, provider: rec.runtime === 'claude' ? rec.provider || undefined : undefined, resume: resumable ? rec.sessionId : undefined, log: rec.log, commands: rec.commands, mode: rec.mode, spawnedBy: rec.spawnedBy, spawnDepth: rec.spawnDepth, cascadeRole: rec.cascadeRole, hop: rec.hop, sideParent: rec.sideParent, sideTask: rec.sideTask, pendingSideContexts: rec.pendingSideContexts, sliceType: rec.sliceType, flowSessionId: rec.flowSessionId, flowTaskKey: rec.flowTaskKey, flowRole: rec.flowRole, flowReviewTarget: rec.flowReviewTarget, flowReviewTargets: rec.flowReviewTargets, flowReviewSnapshots: rec.flowReviewSnapshots, flowReviewRound: rec.flowReviewRound, dispatchBaseSha: rec.dispatchBaseSha, revertTarget: rec.revertTarget, cwd: rec.cwd, managedWorktree: rec.managedWorktree, rolePrompt: rec.rolePrompt, reviewSliceRoots: rec.reviewSliceRoots, openedAt: rec.openedAt, lastUsage: rec.lastUsage, carryRecap: wasInterrupted ? recoveryRecap : rec.carryRecap,
|
|
3992
4061
|
// Lazy-boot restored terminals that were IDLE + not part of a flow: their transcript
|
|
3993
4062
|
// shows immediately; the query boots on first turn. Mid-turn + flow terminals boot now
|
|
3994
4063
|
// (mid-turn needs auto-resume; flow needs its lane live).
|
|
@@ -4103,7 +4172,7 @@ flowChannel
|
|
|
4103
4172
|
if (payload.originTerm && !origin) return
|
|
4104
4173
|
const flowRuntime = origin ? normalizeFlowRuntime(origin.runtime, null) : 'claude'
|
|
4105
4174
|
if (!flowRuntime) return
|
|
4106
|
-
const flowCatalog = flowRuntime === 'codex' ? (origin?.models || readCodexModels()) : []
|
|
4175
|
+
const flowCatalog = flowRuntime === 'codex' ? (origin?.models || readCodexModels()) : flowRuntime === 'hermes' ? (origin?.models || []) : []
|
|
4107
4176
|
const conductorModel = flowConductorModelFor({ runtime: flowRuntime, originModel: origin?.model, catalog: flowCatalog })
|
|
4108
4177
|
const cid = randomUUID()
|
|
4109
4178
|
termNames[cid] = `Flow · ${String(payload.flowId).slice(0, 6)}`
|
|
@@ -4115,7 +4184,7 @@ flowChannel
|
|
|
4115
4184
|
// Model tiers (2026-07-03-flow-lane-model-tiers): the conductor keeps whatever brain it was
|
|
4116
4185
|
// given by default (TP_FLOW_CONDUCTOR_MODEL unset → undefined → today's behavior); set the env
|
|
4117
4186
|
// to pin a cheaper/smarter conductor. Lanes get tiered below via laneModelFor.
|
|
4118
|
-
openStructured({ id: cid, runtime: flowRuntime, model: conductorModel, mode: flowRuntime === 'codex' ? 'plan' : 'default', rolePrompt: flowRuntime === '
|
|
4187
|
+
openStructured({ id: cid, runtime: flowRuntime, model: conductorModel, mode: flowRuntime === 'codex' ? 'plan' : 'default', rolePrompt: flowRuntime === 'claude' ? FLOW_CONDUCTOR_PROMPT : FLOW_CODEX_CONDUCTOR_PROMPT, flowSessionId: payload.flowId, flowRole: 'conductor', spawnedBy: `flow:${payload.flowId}` })
|
|
4119
4188
|
const ce = sessions.get(cid)
|
|
4120
4189
|
if (ce?.session) { try { ce.session.sendTurn(payload.prompt || '') } catch { /* session still starting */ } }
|
|
4121
4190
|
process.stderr.write(`\n ${A.mag}◆ flow conductor launched — flow ${String(payload.flowId).slice(0, 8)} (plan mode).${A.rst}\n`)
|
|
@@ -4136,7 +4205,7 @@ flowChannel
|
|
|
4136
4205
|
}
|
|
4137
4206
|
const flowRuntime = normalizeFlowRuntime(conductor.runtime, null)
|
|
4138
4207
|
if (!flowRuntime) return
|
|
4139
|
-
const flowCatalog = flowRuntime === 'codex' ? (conductor.models || readCodexModels()) : []
|
|
4208
|
+
const flowCatalog = flowRuntime === 'codex' ? (conductor.models || readCodexModels()) : flowRuntime === 'hermes' ? (conductor.models || []) : []
|
|
4140
4209
|
const assignments = []
|
|
4141
4210
|
// Step 6 — cap concurrent Flow lanes (invariant: ≤ FLOW_LIMITS.maxConcurrentLanes).
|
|
4142
4211
|
// Overflow tasks stay pending; the room re-dispatches them in the next wave.
|
|
@@ -4166,7 +4235,7 @@ flowChannel
|
|
|
4166
4235
|
// every other slice gets the builder prompt.
|
|
4167
4236
|
const isReview = t.slice_type === 'review'
|
|
4168
4237
|
if (!validReviewTargetShape(t, flowRuntime)) {
|
|
4169
|
-
process.stderr.write(`\n ${A.yel}◆ flow dispatch held — Codex review ${t.task_key} must target exactly one dependency.${A.rst}\n`)
|
|
4238
|
+
process.stderr.write(`\n ${A.yel}◆ flow dispatch held — ${flowRuntime === 'hermes' ? 'Hermes' : 'Codex'} review ${t.task_key} must target exactly one dependency.${A.rst}\n`)
|
|
4170
4239
|
continue
|
|
4171
4240
|
}
|
|
4172
4241
|
const { dir } = createFlowWorktree({ flowId: payload.flowId, taskKey: t.task_key })
|
|
@@ -4194,6 +4263,12 @@ flowChannel
|
|
|
4194
4263
|
const reviewSliceRoots = isReview
|
|
4195
4264
|
? (t.deps || []).map((dep) => worktreeSpec({ flowId: payload.flowId, taskKey: dep }).dir)
|
|
4196
4265
|
: []
|
|
4266
|
+
const flowReviewSnapshots = isReview ? (t.deps || []).map((dep) => {
|
|
4267
|
+
const cwd = worktreeSpec({ flowId: payload.flowId, taskKey: dep }).dir
|
|
4268
|
+
let sha = null
|
|
4269
|
+
try { sha = execFileSync('git', ['-C', cwd, 'rev-parse', 'HEAD'], { encoding: 'utf8', timeout: 3000, stdio: ['ignore', 'pipe', 'ignore'] }).trim() } catch { /* target is not safely reviewable */ }
|
|
4270
|
+
return { taskKey: dep, cwd, sha }
|
|
4271
|
+
}).filter((item) => item.sha) : []
|
|
4197
4272
|
// Lanes build autonomously in their own worktree — bypassPermissions so they
|
|
4198
4273
|
// don't stall on a card for every write/bash (matches the user's expectation that
|
|
4199
4274
|
// a Flow summoned from a bypass terminal runs hands-off).
|
|
@@ -4204,14 +4279,14 @@ flowChannel
|
|
|
4204
4279
|
// a lane later, on demand, via activateLaneSkill — never the base prompt here.
|
|
4205
4280
|
// S4 — resume: on a re-dispatch, replay the killed lane's HEALED transcript (Heal-3'd,
|
|
4206
4281
|
// no dangling tool_use → no 400) instead of a cold start; undefined for a fresh lane.
|
|
4207
|
-
const laneBase = flowRuntime === 'codex'
|
|
4282
|
+
const laneBase = flowRuntime === 'codex' || flowRuntime === 'hermes'
|
|
4208
4283
|
? (isReview ? FLOW_CODEX_REVIEWER_PROMPT : FLOW_CODEX_LANE_PROMPT)
|
|
4209
4284
|
: (isReview ? FLOW_REVIEWER_PROMPT : FLOW_LANE_PROMPT)
|
|
4210
4285
|
const laneRolePrompt = buildLanePrompt({ base: laneBase })
|
|
4211
4286
|
// Model tiers (2026-07-03-flow-lane-model-tiers): pick the lane's brain by slice_type
|
|
4212
4287
|
// (scaffold→sonnet, feature/fix/review→opus; env can blanket-override or `inherit` to
|
|
4213
4288
|
// restore today's exact behavior). undefined → no model key passed (openStructured default).
|
|
4214
|
-
openStructured({ id: laneId, runtime: flowRuntime, cwd: dir, model: flowLaneModelFor({ sliceType: t.slice_type, runtime: flowRuntime, catalog: flowCatalog }), mode: flowRuntime === 'codex' && isReview ? 'review' : 'bypassPermissions', rolePrompt: laneRolePrompt, flowSessionId: payload.flowId, flowTaskKey: t.task_key, flowRole: isReview ? 'reviewer' : 'builder', flowReviewTargets: isReview ? (t.deps || []) : [], dispatchBaseSha, revertTarget: redispatch?.revertTarget || null, spawnedBy: `flow:${payload.flowId}`, resume: redispatch?.resumeSessionId || undefined, reviewSliceRoots })
|
|
4289
|
+
openStructured({ id: laneId, runtime: flowRuntime, cwd: dir, model: flowLaneModelFor({ sliceType: t.slice_type, runtime: flowRuntime, catalog: flowCatalog }), mode: flowRuntime === 'codex' && isReview ? 'review' : 'bypassPermissions', rolePrompt: laneRolePrompt, flowSessionId: payload.flowId, flowTaskKey: t.task_key, flowRole: isReview ? 'reviewer' : 'builder', flowReviewTargets: isReview ? (t.deps || []) : [], flowReviewSnapshots, dispatchBaseSha, revertTarget: redispatch?.revertTarget || null, spawnedBy: `flow:${payload.flowId}`, resume: redispatch?.resumeSessionId || undefined, reviewSliceRoots })
|
|
4215
4290
|
const le = sessions.get(laneId)
|
|
4216
4291
|
if (le) {
|
|
4217
4292
|
// S4 — stamp the surviving revert target on the resumed lane so a later reviewer still
|
|
@@ -4236,7 +4311,7 @@ flowChannel
|
|
|
4236
4311
|
: '') +
|
|
4237
4312
|
`ACCEPTANCE to INDEPENDENTLY verify: ${t.acceptance || t.title}\n` +
|
|
4238
4313
|
`\nProject (context): ${payload.flowPrompt || ''}\n\n` +
|
|
4239
|
-
(flowRuntime === 'codex'
|
|
4314
|
+
(flowRuntime === 'codex' || flowRuntime === 'hermes'
|
|
4240
4315
|
? `Review it without mutating the builder worktree. For write-producing install/build/test commands, copy its source into scratch under your own current worktree first. Submit the structured verdict with the ThinkPool submit_flow_review MCP tool; never write FLOW_REVIEW.json or call mark_flow_done.`
|
|
4241
4316
|
: `Review it — run it, hunt for failure. When done, use the Write tool to write FLOW_REVIEW.json with {"pass":<boolean>,"reasons":["<specific finding>"],"taskKey":"<the slice you reviewed>"}. That Write IS your verdict.`)
|
|
4242
4317
|
: `[Flow lane — slice "${t.task_key}" of flow ${String(payload.flowId).slice(0, 8)}]\n` +
|
|
@@ -4251,7 +4326,7 @@ flowChannel
|
|
|
4251
4326
|
assembleCrossWaveContext(payload.flowId, { baseDir: process.cwd(), deps: t.deps }).text) +
|
|
4252
4327
|
`\nProject (context): ${payload.flowPrompt || ''}\n\n` +
|
|
4253
4328
|
`Build your slice. Own only your files. Done = it runs + meets acceptance. Commit when done.` +
|
|
4254
|
-
(flowRuntime === 'codex' ? ` Then call the ThinkPool mark_flow_done MCP tool; do not write FLOW_DONE.` : '')
|
|
4329
|
+
(flowRuntime === 'codex' || flowRuntime === 'hermes' ? ` Then call the ThinkPool mark_flow_done MCP tool; do not write FLOW_DONE.` : '')
|
|
4255
4330
|
try { le.session.sendTurn(spec) } catch { /* session still starting */ }
|
|
4256
4331
|
}
|
|
4257
4332
|
}
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
// One live dispatch intent per structured terminal.
|
|
2
|
+
//
|
|
3
|
+
// Durable permission rows can outlive the agent turn that requested them. The
|
|
4
|
+
// lease is the bridge-local execution fence: arguments still need the durable
|
|
5
|
+
// fingerprint/authority checks, and the matching intent must also remain current
|
|
6
|
+
// until the lane is created. A newer human turn or abort supersedes the lease, so
|
|
7
|
+
// a late (otherwise valid) approval cannot resurrect an old runtime choice.
|
|
8
|
+
|
|
9
|
+
const nextGeneration = (entry) =>
|
|
10
|
+
(Number.isSafeInteger(entry?.dispatchGeneration) ? entry.dispatchGeneration : 0) + 1
|
|
11
|
+
|
|
12
|
+
export function beginDispatchLease (entry, permissionId) {
|
|
13
|
+
if (!entry || !permissionId) return Object.freeze({ ok: false, code: 'invalid_dispatch' })
|
|
14
|
+
if (entry.dispatchLease) return Object.freeze({ ok: false, code: 'dispatch_pending', active: entry.dispatchLease })
|
|
15
|
+
const lease = Object.freeze({ permissionId, generation: nextGeneration(entry) })
|
|
16
|
+
entry.dispatchGeneration = lease.generation
|
|
17
|
+
entry.dispatchLease = lease
|
|
18
|
+
return Object.freeze({ ok: true, lease })
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
export function isDispatchLeaseCurrent (entry, lease) {
|
|
22
|
+
return Boolean(entry && lease && entry.dispatchLease === lease && entry.dispatchGeneration === lease.generation)
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
export function finishDispatchLease (entry, lease) {
|
|
26
|
+
if (!isDispatchLeaseCurrent(entry, lease)) return false
|
|
27
|
+
entry.dispatchLease = null
|
|
28
|
+
return true
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export function supersedeDispatchLease (entry) {
|
|
32
|
+
const lease = entry?.dispatchLease || null
|
|
33
|
+
if (!entry || !lease) return null
|
|
34
|
+
entry.dispatchGeneration = nextGeneration(entry)
|
|
35
|
+
entry.dispatchLease = null
|
|
36
|
+
return lease
|
|
37
|
+
}
|
package/flow-models.mjs
CHANGED
|
@@ -11,10 +11,11 @@ const CLAUDE_TIERS = {
|
|
|
11
11
|
|
|
12
12
|
const CODEX_SCAFFOLD = ['gpt-5.6-luna', 'gpt-5.4-mini', 'gpt-5.3-codex-spark']
|
|
13
13
|
const CODEX_BALANCED = ['gpt-5.6-terra', 'gpt-5.4']
|
|
14
|
+
const HERMES_REVIEW = ['nous:anthropic/claude-sonnet-4.6', 'nous:anthropic/claude-sonnet-4.5', 'nous:anthropic/claude-sonnet-4']
|
|
14
15
|
|
|
15
16
|
export function normalizeFlowRuntime(runtime, fallback = null) {
|
|
16
|
-
if (runtime === 'claude' || runtime === 'codex') return runtime
|
|
17
|
-
return fallback === 'claude' || fallback === 'codex' ? fallback : null
|
|
17
|
+
if (runtime === 'claude' || runtime === 'codex' || runtime === 'hermes') return runtime
|
|
18
|
+
return fallback === 'claude' || fallback === 'codex' || fallback === 'hermes' ? fallback : null
|
|
18
19
|
}
|
|
19
20
|
|
|
20
21
|
export function modelCatalogValues(catalog = []) {
|
|
@@ -32,6 +33,12 @@ function firstVisible(candidates, catalog) {
|
|
|
32
33
|
return undefined
|
|
33
34
|
}
|
|
34
35
|
|
|
36
|
+
function firstVisibleMatching(catalog, patterns) {
|
|
37
|
+
const visible = modelCatalogValues(catalog)
|
|
38
|
+
for (const pattern of patterns) for (const model of visible) if (pattern.test(model)) return model
|
|
39
|
+
return undefined
|
|
40
|
+
}
|
|
41
|
+
|
|
35
42
|
export function flowLaneModelFor({ sliceType, runtime = 'claude', catalog = [], env = process.env } = {}) {
|
|
36
43
|
if (normalizeFlowRuntime(runtime, 'claude') === 'codex') {
|
|
37
44
|
const override = env.TP_FLOW_CODEX_LANE_MODEL
|
|
@@ -39,6 +46,16 @@ export function flowLaneModelFor({ sliceType, runtime = 'claude', catalog = [],
|
|
|
39
46
|
if (override) return modelCatalogValues(catalog).has(override) ? override : undefined
|
|
40
47
|
return firstVisible(sliceType === 'scaffold' ? CODEX_SCAFFOLD : CODEX_BALANCED, catalog)
|
|
41
48
|
}
|
|
49
|
+
if (normalizeFlowRuntime(runtime, 'claude') === 'hermes') {
|
|
50
|
+
const override = env.TP_FLOW_HERMES_LANE_MODEL
|
|
51
|
+
const visible = modelCatalogValues(catalog)
|
|
52
|
+
if (override === 'inherit') return undefined
|
|
53
|
+
if (override) return visible.has(override) ? override : undefined
|
|
54
|
+
if (sliceType === 'review') return firstVisible(HERMES_REVIEW, catalog)
|
|
55
|
+
|| firstVisibleMatching(catalog, [/terra/i, /sonnet/i, /balanced/i])
|
|
56
|
+
if (sliceType === 'scaffold') return firstVisibleMatching(catalog, [/(?:luna|mini)/i])
|
|
57
|
+
return firstVisibleMatching(catalog, [/terra/i, /(?:luna|mini)/i])
|
|
58
|
+
}
|
|
42
59
|
const override = env.TP_FLOW_CLAUDE_LANE_MODEL || env.TP_FLOW_LANE_MODEL
|
|
43
60
|
if (override === 'inherit') return undefined
|
|
44
61
|
if (override) return override
|
|
@@ -53,6 +70,13 @@ export function flowConductorModelFor({ runtime = 'claude', originModel, catalog
|
|
|
53
70
|
if (override) return visible.has(override) ? override : undefined
|
|
54
71
|
return originModel && visible.has(originModel) ? originModel : undefined
|
|
55
72
|
}
|
|
73
|
+
if (normalizeFlowRuntime(runtime, 'claude') === 'hermes') {
|
|
74
|
+
const override = env.TP_FLOW_HERMES_CONDUCTOR_MODEL
|
|
75
|
+
const visible = modelCatalogValues(catalog)
|
|
76
|
+
if (override === 'inherit') return undefined
|
|
77
|
+
if (override) return visible.has(override) ? override : undefined
|
|
78
|
+
return originModel && visible.has(originModel) ? originModel : undefined
|
|
79
|
+
}
|
|
56
80
|
const override = env.TP_FLOW_CLAUDE_CONDUCTOR_MODEL || env.TP_FLOW_CONDUCTOR_MODEL
|
|
57
81
|
if (override === 'inherit') return undefined
|
|
58
82
|
// Claude non-regression: the conductor historically inherited the host default,
|
package/flow-task-graph.mjs
CHANGED
|
@@ -176,10 +176,10 @@ export function normalizePlanOutput (raw) {
|
|
|
176
176
|
// Codex review protocol deliberately maps one reviewer lane to one builder target.
|
|
177
177
|
// Claude's legacy Flow plans may review several deps and remain unchanged.
|
|
178
178
|
export function validatePlanForRuntime (plan, runtime = 'claude') {
|
|
179
|
-
if (runtime !== 'codex') return plan
|
|
179
|
+
if (runtime !== 'codex' && runtime !== 'hermes') return plan
|
|
180
180
|
for (const task of plan?.tasks || []) {
|
|
181
181
|
if (task.sliceType === SLICE_TYPE.review && task.deps.length !== 1) {
|
|
182
|
-
throw new Error(
|
|
182
|
+
throw new Error(`${runtime === 'hermes' ? 'Hermes' : 'Codex'} review task "${task.key}" must depend on exactly one builder task`)
|
|
183
183
|
}
|
|
184
184
|
}
|
|
185
185
|
return plan
|
|
@@ -187,7 +187,7 @@ export function validatePlanForRuntime (plan, runtime = 'claude') {
|
|
|
187
187
|
|
|
188
188
|
export function validReviewTargetShape (task, runtime = 'claude') {
|
|
189
189
|
if (!task || task.slice_type !== SLICE_TYPE.review) return true
|
|
190
|
-
return runtime !== 'codex' || (Array.isArray(task.deps) && task.deps.length === 1)
|
|
190
|
+
return (runtime !== 'codex' && runtime !== 'hermes') || (Array.isArray(task.deps) && task.deps.length === 1)
|
|
191
191
|
}
|
|
192
192
|
|
|
193
193
|
export function legacyBuilderCompletionAllowed ({ runtime, flowRole, eventKind, eventSubtype, interrupted = false } = {}) {
|
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Process-local ThinkPool Hermes ACP tool policy.
|
|
3
|
+
|
|
4
|
+
Never import this through the user profile. The bridge starts this file with
|
|
5
|
+
Hermes' installed venv interpreter and passes a validated role policy in env.
|
|
6
|
+
"""
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
import sys
|
|
10
|
+
|
|
11
|
+
POLICY_ENV = "THINKPOOL_HERMES_ACP_POLICY"
|
|
12
|
+
|
|
13
|
+
CODING_TOOLS = frozenset({
|
|
14
|
+
"web_search", "web_extract", "terminal", "process", "read_file", "write_file",
|
|
15
|
+
"patch", "search_files", "vision_analyze", "skills_list", "skill_view",
|
|
16
|
+
"skill_manage", "browser_navigate", "browser_snapshot", "browser_click",
|
|
17
|
+
"browser_type", "browser_scroll", "browser_back", "browser_press",
|
|
18
|
+
"browser_get_images", "browser_vision", "browser_console", "browser_cdp",
|
|
19
|
+
"browser_dialog", "todo", "memory", "execute_code",
|
|
20
|
+
})
|
|
21
|
+
READ_ONLY_TOOLS = frozenset({"read_file", "search_files"})
|
|
22
|
+
ESSENTIAL_CODING_TOOLS = frozenset({
|
|
23
|
+
"terminal", "process", "read_file", "write_file", "patch", "search_files",
|
|
24
|
+
})
|
|
25
|
+
|
|
26
|
+
def die(message):
|
|
27
|
+
print("ThinkPool Hermes ACP policy error: " + message, file=sys.stderr)
|
|
28
|
+
raise SystemExit(78)
|
|
29
|
+
|
|
30
|
+
def policy():
|
|
31
|
+
raw = os.environ.get(POLICY_ENV)
|
|
32
|
+
try:
|
|
33
|
+
value = json.loads(raw)
|
|
34
|
+
except Exception:
|
|
35
|
+
die("missing or malformed policy")
|
|
36
|
+
if not isinstance(value, dict) or value.get("version") != 1:
|
|
37
|
+
die("unsupported policy")
|
|
38
|
+
role = value.get("role")
|
|
39
|
+
builtin = value.get("builtinTools")
|
|
40
|
+
required_builtin = value.get("requiredBuiltinTools")
|
|
41
|
+
tools = value.get("mcpTools")
|
|
42
|
+
if role not in {"ordinary", "builder", "conductor", "reviewer", "manual-review"}:
|
|
43
|
+
die("unknown role")
|
|
44
|
+
if value.get("mcpServer") != "thinkpool" or not isinstance(builtin, list) or not isinstance(required_builtin, list) or not isinstance(tools, list):
|
|
45
|
+
die("invalid tool policy")
|
|
46
|
+
if not all(isinstance(x, str) and x and x.replace("_", "").isalnum() for x in builtin + required_builtin + tools):
|
|
47
|
+
die("invalid tool name")
|
|
48
|
+
forbidden = {"delegate_task", "session_search"}
|
|
49
|
+
if forbidden.intersection(builtin) or forbidden.intersection(required_builtin) or forbidden.intersection(tools):
|
|
50
|
+
die("delegation and session search are forbidden")
|
|
51
|
+
required_mcp = {
|
|
52
|
+
"ordinary": {"read_terminal"}, "builder": {"mark_flow_done"},
|
|
53
|
+
"conductor": {"submit_flow_plan"},
|
|
54
|
+
"reviewer": {"submit_flow_review", "run_review_check", "read_review_file"},
|
|
55
|
+
"manual-review": {"run_review_check", "read_review_file"},
|
|
56
|
+
}
|
|
57
|
+
if not required_mcp[role].issubset(tools):
|
|
58
|
+
die("incomplete role MCP policy")
|
|
59
|
+
restricted = role in {"conductor", "reviewer", "manual-review"}
|
|
60
|
+
allowed_builtin = READ_ONLY_TOOLS if restricted else CODING_TOOLS
|
|
61
|
+
required = READ_ONLY_TOOLS if restricted else ESSENTIAL_CODING_TOOLS
|
|
62
|
+
if set(builtin) != allowed_builtin:
|
|
63
|
+
die("restricted role requires exact read-only builtins" if restricted else "coding role requires the full approved builtin allowlist")
|
|
64
|
+
if set(required_builtin) != required:
|
|
65
|
+
die("invalid required builtin policy")
|
|
66
|
+
return role, tuple(dict.fromkeys(builtin)), tuple(dict.fromkeys(required_builtin)), tuple(dict.fromkeys(tools))
|
|
67
|
+
|
|
68
|
+
ROLE, BUILTIN, REQUIRED_BUILTIN, MCP_TOOLS = policy()
|
|
69
|
+
ALLOWED = frozenset(BUILTIN) | frozenset("mcp__thinkpool__" + x for x in MCP_TOOLS) | frozenset("mcp_thinkpool_" + x for x in MCP_TOOLS)
|
|
70
|
+
|
|
71
|
+
def name_of(schema):
|
|
72
|
+
if not isinstance(schema, dict): return ""
|
|
73
|
+
fn = schema.get("function")
|
|
74
|
+
return fn.get("name", "") if isinstance(fn, dict) else schema.get("name", "")
|
|
75
|
+
|
|
76
|
+
def filter_schemas(items):
|
|
77
|
+
return [item for item in (items or []) if name_of(item) in ALLOWED]
|
|
78
|
+
|
|
79
|
+
def assert_exact_inventory(agent):
|
|
80
|
+
"""Reject an ACP lifecycle that lost the bridge-owned MCP surface.
|
|
81
|
+
|
|
82
|
+
Hermes treats registration errors as non-fatal. That is acceptable for a
|
|
83
|
+
standalone CLI, but never for a ThinkPool role: the bridge must not let a
|
|
84
|
+
reconstructed agent accept a prompt with a partial policy.
|
|
85
|
+
"""
|
|
86
|
+
tools = list(getattr(agent, "tools", []) or [])
|
|
87
|
+
names = {name_of(item) for item in tools}
|
|
88
|
+
# Allowed is deliberately broader than required for ordinary/builder:
|
|
89
|
+
# browser/provider/vision integrations are availability-gated upstream.
|
|
90
|
+
missing_builtin = set(REQUIRED_BUILTIN) - names
|
|
91
|
+
missing_mcp = [name for name in MCP_TOOLS if not ({"mcp__thinkpool__" + name, "mcp_thinkpool_" + name} & names)]
|
|
92
|
+
extras = names - ALLOWED
|
|
93
|
+
if missing_builtin or missing_mcp or extras:
|
|
94
|
+
details = []
|
|
95
|
+
if missing_builtin: details.append("missing builtins " + ", ".join(sorted(missing_builtin)))
|
|
96
|
+
if missing_mcp: details.append("missing MCP " + ", ".join(sorted(missing_mcp)))
|
|
97
|
+
if extras: details.append("forbidden extras " + ", ".join(sorted(extras)))
|
|
98
|
+
raise RuntimeError("ThinkPool exact inventory is incomplete: " + "; ".join(details))
|
|
99
|
+
agent.tools = tools
|
|
100
|
+
agent.valid_tool_names = names
|
|
101
|
+
|
|
102
|
+
# Patch before importing ACP server. Each new/resumed/reset ACP process reloads
|
|
103
|
+
# this exact policy; no mutable shell alias or profile config participates.
|
|
104
|
+
import toolsets
|
|
105
|
+
toolsets.TOOLSETS["hermes-acp"] = {
|
|
106
|
+
"description": "ThinkPool process-local ACP policy",
|
|
107
|
+
"tools": list(BUILTIN), "includes": []
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
# entry.main() normally discovers profile-configured MCP servers before it
|
|
111
|
+
# creates the ACP server. ThinkPool ACP never inherits those profile servers;
|
|
112
|
+
# the only permitted registration is the bridge's per-session `thinkpool` one.
|
|
113
|
+
import tools.mcp_tool
|
|
114
|
+
tools.mcp_tool.discover_mcp_tools = lambda *args, **kwargs: []
|
|
115
|
+
|
|
116
|
+
import model_tools
|
|
117
|
+
_get_definitions = model_tools.get_tool_definitions
|
|
118
|
+
def constrained_definitions(*args, **kwargs):
|
|
119
|
+
return filter_schemas(_get_definitions(*args, **kwargs))
|
|
120
|
+
model_tools.get_tool_definitions = constrained_definitions
|
|
121
|
+
|
|
122
|
+
import agent.memory_manager
|
|
123
|
+
_inject_memory = agent.memory_manager.inject_memory_provider_tools
|
|
124
|
+
def constrained_memory(agent):
|
|
125
|
+
result = _inject_memory(agent)
|
|
126
|
+
if hasattr(agent, "tools"):
|
|
127
|
+
agent.tools = filter_schemas(agent.tools)
|
|
128
|
+
agent.valid_tool_names = {name_of(x) for x in agent.tools}
|
|
129
|
+
return result
|
|
130
|
+
agent.memory_manager.inject_memory_provider_tools = constrained_memory
|
|
131
|
+
|
|
132
|
+
import acp_adapter.session
|
|
133
|
+
_expand = acp_adapter.session._expand_acp_enabled_toolsets
|
|
134
|
+
def constrained_expand(toolsets_arg=None, mcp_server_names=None):
|
|
135
|
+
requested = list(toolsets_arg or ["hermes-acp"])
|
|
136
|
+
if any(name not in {"hermes-acp", "mcp-thinkpool"} for name in requested):
|
|
137
|
+
raise RuntimeError("ThinkPool ACP only permits hermes-acp and mcp-thinkpool toolsets")
|
|
138
|
+
names = list(mcp_server_names or [])
|
|
139
|
+
if any(name != "thinkpool" for name in names):
|
|
140
|
+
raise RuntimeError("ThinkPool ACP only permits dynamic MCP server thinkpool")
|
|
141
|
+
# Hermes 0.18.2 calls this from /tools with an already-expanded
|
|
142
|
+
# ["hermes-acp", "mcp-thinkpool"] list and no mcp_server_names. Preserve
|
|
143
|
+
# that exact legal expansion; otherwise /tools silently omits ThinkPool.
|
|
144
|
+
return ["hermes-acp"] + (["mcp-thinkpool"] if names or "mcp-thinkpool" in requested else [])
|
|
145
|
+
acp_adapter.session._expand_acp_enabled_toolsets = constrained_expand
|
|
146
|
+
|
|
147
|
+
import acp_adapter.server
|
|
148
|
+
_register = acp_adapter.server.HermesACPAgent._register_session_mcp_servers
|
|
149
|
+
async def constrained_register(self, state, mcp_servers):
|
|
150
|
+
if any(getattr(server, "name", None) != "thinkpool" for server in (mcp_servers or [])):
|
|
151
|
+
raise RuntimeError("ThinkPool ACP only permits dynamic MCP server thinkpool")
|
|
152
|
+
if mcp_servers:
|
|
153
|
+
# SessionState is process-local. Keep only the validated descriptors so
|
|
154
|
+
# a subsequent set_model reconstruction can re-register the same MCP.
|
|
155
|
+
state._thinkpool_mcp_servers = tuple(mcp_servers)
|
|
156
|
+
await _register(self, state, mcp_servers)
|
|
157
|
+
assert_exact_inventory(state.agent)
|
|
158
|
+
acp_adapter.server.HermesACPAgent._register_session_mcp_servers = constrained_register
|
|
159
|
+
|
|
160
|
+
# Hermes 0.18.2's session/set_model creates a fresh state.agent. Upstream
|
|
161
|
+
# does not re-run ACP MCP registration, so the new agent can expose only its
|
|
162
|
+
# built-ins while the request still returns success. Keep the old state until
|
|
163
|
+
# the replacement has re-registered and passed the same exact policy check.
|
|
164
|
+
_set_model = acp_adapter.server.HermesACPAgent.set_session_model
|
|
165
|
+
async def constrained_set_model(self, model_id, session_id, **kwargs):
|
|
166
|
+
state = self.session_manager.get_session(session_id)
|
|
167
|
+
if state is None:
|
|
168
|
+
return await _set_model(self, model_id, session_id, **kwargs)
|
|
169
|
+
old_agent, old_model = state.agent, getattr(state, "model", None)
|
|
170
|
+
try:
|
|
171
|
+
result = await _set_model(self, model_id, session_id, **kwargs)
|
|
172
|
+
if result is None:
|
|
173
|
+
raise RuntimeError("Hermes did not acknowledge model switch")
|
|
174
|
+
servers = getattr(state, "_thinkpool_mcp_servers", ())
|
|
175
|
+
if MCP_TOOLS and not servers:
|
|
176
|
+
raise RuntimeError("ThinkPool MCP registration is unavailable after model switch")
|
|
177
|
+
await constrained_register(self, state, list(servers))
|
|
178
|
+
assert_exact_inventory(state.agent)
|
|
179
|
+
self.session_manager.save_session(session_id)
|
|
180
|
+
return result
|
|
181
|
+
except Exception:
|
|
182
|
+
# Fail closed and restore the prior usable agent/model. A later bridge
|
|
183
|
+
# /tools probe is the external acknowledgement before UI persistence.
|
|
184
|
+
state.agent, state.model = old_agent, old_model
|
|
185
|
+
try:
|
|
186
|
+
servers = getattr(state, "_thinkpool_mcp_servers", ())
|
|
187
|
+
if servers:
|
|
188
|
+
await constrained_register(self, state, list(servers))
|
|
189
|
+
self.session_manager.save_session(session_id)
|
|
190
|
+
except Exception:
|
|
191
|
+
pass
|
|
192
|
+
raise
|
|
193
|
+
acp_adapter.server.HermesACPAgent.set_session_model = constrained_set_model
|
|
194
|
+
|
|
195
|
+
from acp_adapter.entry import main
|
|
196
|
+
main()
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
// Bridge-owned Hermes ACP schema policy. This is intentionally a small, pure
|
|
2
|
+
// contract: Python receives only this JSON and fails closed for anything else.
|
|
3
|
+
export const HERMES_POLICY_VERSION = 1
|
|
4
|
+
|
|
5
|
+
export const CODING_TOOLS = Object.freeze([
|
|
6
|
+
'web_search', 'web_extract', 'terminal', 'process', 'read_file', 'write_file',
|
|
7
|
+
'patch', 'search_files', 'vision_analyze', 'skills_list', 'skill_view',
|
|
8
|
+
'skill_manage', 'browser_navigate', 'browser_snapshot', 'browser_click',
|
|
9
|
+
'browser_type', 'browser_scroll', 'browser_back', 'browser_press',
|
|
10
|
+
'browser_get_images', 'browser_vision', 'browser_console', 'browser_cdp',
|
|
11
|
+
'browser_dialog', 'todo', 'memory', 'execute_code',
|
|
12
|
+
])
|
|
13
|
+
|
|
14
|
+
const READ_ONLY_TOOLS = Object.freeze(['read_file', 'search_files'])
|
|
15
|
+
// Hermes providers advertise capabilities conditionally (for example browser
|
|
16
|
+
// CDP/dialog support depends on the installed browser integration). These are
|
|
17
|
+
// the only builtins an ordinary or builder lane must have to do useful coding
|
|
18
|
+
// work; the complete coding allowlist above remains permitted when available.
|
|
19
|
+
export const ESSENTIAL_CODING_TOOLS = Object.freeze([
|
|
20
|
+
'terminal', 'process', 'read_file', 'write_file', 'patch', 'search_files',
|
|
21
|
+
])
|
|
22
|
+
const ROLE_REQUIRED = Object.freeze({
|
|
23
|
+
ordinary: ['read_terminal', 'spawn_terminal', 'close_terminal'],
|
|
24
|
+
conductor: ['submit_flow_plan'],
|
|
25
|
+
builder: ['mark_flow_done'],
|
|
26
|
+
reviewer: ['submit_flow_review', 'run_review_check', 'read_review_file'],
|
|
27
|
+
'manual-review': ['run_review_check', 'read_review_file'],
|
|
28
|
+
})
|
|
29
|
+
|
|
30
|
+
export function hermesRoleFor({ flowRole, sliceType } = {}) {
|
|
31
|
+
if (flowRole === 'conductor') return 'conductor'
|
|
32
|
+
if (flowRole === 'reviewer') return 'reviewer'
|
|
33
|
+
if (sliceType === 'review') return 'manual-review'
|
|
34
|
+
if (flowRole === 'builder') return 'builder'
|
|
35
|
+
return 'ordinary'
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
export function hermesPolicyForRole(role, { mcpTools } = {}) {
|
|
39
|
+
if (!Object.hasOwn(ROLE_REQUIRED, role)) throw new Error(`Unknown Hermes role policy: ${role}`)
|
|
40
|
+
const supplied = Array.isArray(mcpTools) ? mcpTools.map(String).filter(Boolean) : ROLE_REQUIRED[role]
|
|
41
|
+
const required = ROLE_REQUIRED[role]
|
|
42
|
+
// Ordinary worker leaves deliberately lack spawn/close. Main-lane proof is
|
|
43
|
+
// enforced by requiredMcpTools at dispatch; keep this schema usable for a
|
|
44
|
+
// non-delegating ordinary child without widening it.
|
|
45
|
+
const minimum = role === 'ordinary' ? ['read_terminal'] : required
|
|
46
|
+
for (const tool of minimum) if (!supplied.includes(tool)) throw new Error(`Hermes ${role} policy is missing required ThinkPool tool ${tool}`)
|
|
47
|
+
const restricted = role === 'conductor' || role === 'reviewer' || role === 'manual-review'
|
|
48
|
+
const builtinTools = restricted ? READ_ONLY_TOOLS : CODING_TOOLS
|
|
49
|
+
const requiredBuiltinTools = restricted ? READ_ONLY_TOOLS : ESSENTIAL_CODING_TOOLS
|
|
50
|
+
return Object.freeze({
|
|
51
|
+
version: HERMES_POLICY_VERSION,
|
|
52
|
+
role,
|
|
53
|
+
// `builtinTools` is the role's complete approved surface, not a promise
|
|
54
|
+
// that an availability-gated upstream provider implements every tool.
|
|
55
|
+
builtinTools,
|
|
56
|
+
requiredBuiltinTools,
|
|
57
|
+
mcpServer: 'thinkpool',
|
|
58
|
+
mcpTools: [...new Set(supplied)].sort(),
|
|
59
|
+
})
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export function hermesRequiredMcpTools(role, { canSpawnWorkers = false } = {}) {
|
|
63
|
+
if (role === 'ordinary') return ['read_terminal', ...(canSpawnWorkers ? ['spawn_terminal', 'close_terminal'] : [])]
|
|
64
|
+
return [...ROLE_REQUIRED[role]]
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
// Keep the bridge's local /tools proof on the same canonical inventory as the
|
|
68
|
+
// process-local Python bootstrap. `allBuiltinTools` lets that proof reject a
|
|
69
|
+
// known built-in which is not part of this role, rather than merely looking
|
|
70
|
+
// for a couple of required MCP names.
|
|
71
|
+
export function hermesExactInventory(role, options = {}) {
|
|
72
|
+
const policy = hermesPolicyForRole(role, options)
|
|
73
|
+
return Object.freeze({
|
|
74
|
+
builtinTools: [...policy.builtinTools].sort(),
|
|
75
|
+
requiredBuiltinTools: [...policy.requiredBuiltinTools].sort(),
|
|
76
|
+
mcpTools: [...policy.mcpTools].sort(),
|
|
77
|
+
// The two excluded upstream capabilities must be checked as forbidden too;
|
|
78
|
+
// they are intentionally absent from CODING_TOOLS, not unknown to policy.
|
|
79
|
+
allBuiltinTools: [...CODING_TOOLS, 'delegate_task', 'session_search'].sort(),
|
|
80
|
+
})
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
export function hermesPolicyEnv(role, options = {}) {
|
|
84
|
+
return JSON.stringify(hermesPolicyForRole(role, options))
|
|
85
|
+
}
|
package/hermes-probe.mjs
CHANGED
|
@@ -1,8 +1,36 @@
|
|
|
1
1
|
import { execFileSync } from 'node:child_process'
|
|
2
|
+
import fs from 'node:fs'
|
|
3
|
+
import path from 'node:path'
|
|
2
4
|
|
|
3
5
|
const clean = (value) => String(value || '').replace(/[\r\n]+/g, ' ').trim()
|
|
4
6
|
|
|
5
|
-
|
|
7
|
+
const INSTALL = /Install directory:\s*(.+?)(?:\r?\n|$)/i
|
|
8
|
+
const PROFILE = /Config:\s*(.+?)(?:\r?\n|$)/i
|
|
9
|
+
|
|
10
|
+
// Resolve the installed venv and isolated profile once, then launch ACP through
|
|
11
|
+
// bridge-owned code. `thinkpool` is only queried for inventory; it is never the
|
|
12
|
+
// executable that serves an ACP lane.
|
|
13
|
+
export function resolveHermesAcpRuntime({ command = 'thinkpool', execFile = execFileSync, exists = fs.existsSync, bootstrap = new URL('./hermes-acp-bootstrap.py', import.meta.url) } = {}) {
|
|
14
|
+
try {
|
|
15
|
+
const versionOutput = execFile(command, ['--version'], { encoding: 'utf8', timeout: 10_000, stdio: ['ignore', 'pipe', 'pipe'] })
|
|
16
|
+
const install = clean(String(versionOutput).match(INSTALL)?.[1])
|
|
17
|
+
if (!install || !path.isAbsolute(install)) throw new Error('Hermes did not report an absolute install directory')
|
|
18
|
+
const python = path.join(install, 'venv', 'bin', 'python')
|
|
19
|
+
const bootstrapPath = bootstrap instanceof URL ? bootstrap.pathname : String(bootstrap)
|
|
20
|
+
if (!exists(python) || !exists(bootstrapPath)) throw new Error('Hermes venv Python or ThinkPool ACP bootstrap is missing')
|
|
21
|
+
// `thinkpool` is the dedicated profile wrapper on supported installs. Its
|
|
22
|
+
// config output is evidence, not the ACP launch path.
|
|
23
|
+
const configOutput = execFile(command, ['config', 'show'], { encoding: 'utf8', timeout: 10_000, stdio: ['ignore', 'pipe', 'pipe'] })
|
|
24
|
+
const config = clean(String(configOutput).match(PROFILE)?.[1])
|
|
25
|
+
const profile = config ? path.dirname(config) : ''
|
|
26
|
+
if (!profile || !path.isAbsolute(profile) || path.basename(profile) !== 'thinkpool') throw new Error('Hermes did not report the isolated thinkpool profile')
|
|
27
|
+
return { python, bootstrap: bootstrapPath, profile, install }
|
|
28
|
+
} catch (error) {
|
|
29
|
+
return { error: clean(error?.stderr || error?.message || error) || 'could not resolve Hermes ACP runtime' }
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export function probeHermesRuntime({ command = 'thinkpool', prefixArgs = [], execFile = execFileSync, env = process.env, strictBootstrap = false, exists = fs.existsSync } = {}) {
|
|
6
34
|
try {
|
|
7
35
|
const versionOutput = execFile(command, [...prefixArgs, '--version'], { encoding: 'utf8', env, timeout: 10_000, stdio: ['ignore', 'pipe', 'pipe'] })
|
|
8
36
|
const version = clean(versionOutput).match(/Hermes Agent v([^\s]+)/i)?.[1] || null
|
|
@@ -16,7 +44,9 @@ export function probeHermesRuntime({ command = 'thinkpool', prefixArgs = [], exe
|
|
|
16
44
|
&& /ThinkPool blocks hidden Hermes delegation/i.test(guardOutput)
|
|
17
45
|
&& /pre_tool_call[\s\S]*delegate_task[\s\S]*(?:✓ allowed|allowed)/i.test(hooksOutput)
|
|
18
46
|
if (!doctorHealthy || !delegationBlocked) return { available: false, version, reason: 'delegate_task guard is not healthy, unchanged, allowlisted, and structurally blocking in the dedicated Hermes profile' }
|
|
19
|
-
|
|
47
|
+
const runtime = strictBootstrap ? resolveHermesAcpRuntime({ command, execFile, exists }) : null
|
|
48
|
+
if (runtime?.error) return { available: false, version, reason: `Hermes ACP bootstrap unavailable: ${runtime.error}` }
|
|
49
|
+
return { available: true, version, acpProtocol: 1, delegationBlocked: true, ...(runtime || {}) }
|
|
20
50
|
} catch (error) {
|
|
21
51
|
return { available: false, version: null, reason: clean(error?.stderr || error?.message || error) }
|
|
22
52
|
}
|
package/hermes-session.mjs
CHANGED
|
@@ -6,6 +6,7 @@ import { randomUUID } from 'node:crypto'
|
|
|
6
6
|
import { AcpClient } from './acp-client.mjs'
|
|
7
7
|
import { HermesEventMapper, hermesToolFor } from './hermes-event-mapper.mjs'
|
|
8
8
|
import { probeHermesRuntime } from './hermes-probe.mjs'
|
|
9
|
+
import { hermesExactInventory, hermesPolicyEnv, hermesRoleFor } from './hermes-policy.mjs'
|
|
9
10
|
import { startCodexMcpHttp } from './codex-mcp-http.mjs'
|
|
10
11
|
import { classifyRisk } from './claude-session.mjs'
|
|
11
12
|
|
|
@@ -44,7 +45,7 @@ export function startHermesSession({
|
|
|
44
45
|
cwd, model, resume, env = process.env, mode = 'default', onEvent, requestPermission,
|
|
45
46
|
roomContext, terminalRolePrompt, rolePrompt, mcpServers, requiredMcpTools = [], prepareCwd = null,
|
|
46
47
|
command = HERMES_COMMAND, args = ['acp'], clientFactory = createAcpClient,
|
|
47
|
-
mcpHttpFactory = startCodexMcpHttp, lazy = false,
|
|
48
|
+
mcpHttpFactory = startCodexMcpHttp, lazy = false, hermesRole = null,
|
|
48
49
|
} = {}) {
|
|
49
50
|
let activeCwd = cwd
|
|
50
51
|
const requestedModel = model || null
|
|
@@ -68,6 +69,7 @@ export function startHermesSession({
|
|
|
68
69
|
let resuming = false
|
|
69
70
|
let inventoryProbe = null
|
|
70
71
|
let modelSwitchPending = false
|
|
72
|
+
const policyRole = hermesRole || hermesRoleFor({})
|
|
71
73
|
|
|
72
74
|
const emit = (event) => { try { onEvent?.(event) } catch { /* consumer isolation */ } }
|
|
73
75
|
|
|
@@ -101,7 +103,14 @@ export function startHermesSession({
|
|
|
101
103
|
async function probeMcpTools() {
|
|
102
104
|
const required = [...new Set((Array.isArray(requiredMcpTools) ? requiredMcpTools : [])
|
|
103
105
|
.map((name) => String(name || '').trim()).filter(Boolean))]
|
|
104
|
-
|
|
106
|
+
// Direct runtime tests and unscoped upstream callers have no bridge role
|
|
107
|
+
// contract to prove. Every bridge-created Hermes lane supplies its required
|
|
108
|
+
// MCP list; only those lanes enter the exact-inventory transaction.
|
|
109
|
+
if (!required.length) return { inventory: '', missing: [], forbidden: [] }
|
|
110
|
+
const exact = hermesExactInventory(policyRole, { mcpTools: required })
|
|
111
|
+
const allowedBuiltins = exact.builtinTools
|
|
112
|
+
const requiredBuiltins = exact.requiredBuiltinTools
|
|
113
|
+
const expectedMcp = exact.mcpTools
|
|
105
114
|
const chunks = []
|
|
106
115
|
inventoryProbe = chunks
|
|
107
116
|
try {
|
|
@@ -112,7 +121,32 @@ export function startHermesSession({
|
|
|
112
121
|
}, 30_000)
|
|
113
122
|
} finally { inventoryProbe = null }
|
|
114
123
|
const inventory = chunks.join('')
|
|
115
|
-
|
|
124
|
+
const has = (name) => new RegExp(`(?:^|[^a-z0-9_])${name}(?:$|[^a-z0-9_])`, 'i').test(inventory)
|
|
125
|
+
const mcpNames = [...inventory.matchAll(/\bmcp(?:__|_)[a-z0-9_]+/gi)].map((item) => item[0])
|
|
126
|
+
const expectedMcpNames = new Set(expectedMcp.flatMap((name) => [`mcp__thinkpool__${name}`, `mcp_thinkpool_${name}`]))
|
|
127
|
+
const unexpectedMcp = [...new Set(mcpNames.filter((name) => !expectedMcpNames.has(name)))]
|
|
128
|
+
// This is a zero-inference ACP inventory command. It proves the complete
|
|
129
|
+
// role schema (required built-ins plus exactly this lane's ThinkPool MCP
|
|
130
|
+
// set). Optional provider/browser capabilities may be absent, but any
|
|
131
|
+
// presented builtin outside the role allowlist is still a hard failure.
|
|
132
|
+
return {
|
|
133
|
+
inventory,
|
|
134
|
+
missing: [
|
|
135
|
+
...requiredBuiltins.filter((name) => !has(name)),
|
|
136
|
+
...expectedMcp.filter((name) => !has(`mcp__thinkpool__${name}`) && !has(`mcp_thinkpool_${name}`)),
|
|
137
|
+
],
|
|
138
|
+
forbidden: [
|
|
139
|
+
...exact.allBuiltinTools.filter((name) => !allowedBuiltins.includes(name) && has(name)),
|
|
140
|
+
...unexpectedMcp,
|
|
141
|
+
],
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
async function assertMcpReadiness() {
|
|
146
|
+
const probe = await probeMcpTools()
|
|
147
|
+
if (!probe.missing.length && !probe.forbidden.length) return probe
|
|
148
|
+
const details = [probe.missing.length ? `missing ${probe.missing.join(', ')}` : '', probe.forbidden.length ? `forbidden ${probe.forbidden.join(', ')}` : ''].filter(Boolean).join('; ')
|
|
149
|
+
throw new Error(`Hermes ThinkPool MCP readiness failed; ${details}`)
|
|
116
150
|
}
|
|
117
151
|
|
|
118
152
|
async function boot() {
|
|
@@ -126,18 +160,25 @@ export function startHermesSession({
|
|
|
126
160
|
}
|
|
127
161
|
const thinkpool = mcpServers?.thinkpool
|
|
128
162
|
if (thinkpool && !mcpHttp) mcpHttp = await mcpHttpFactory({ sdkServer: thinkpool })
|
|
129
|
-
|
|
163
|
+
let childEnv = hermesChildEnv(env)
|
|
164
|
+
let launchCommand = command
|
|
165
|
+
let launchArgs = args
|
|
130
166
|
// The bridge advertises Hermes only after this probe, but lazy lanes may
|
|
131
167
|
// boot much later. Re-check immediately before every real ACP process so
|
|
132
168
|
// a removed/changed delegation hook cannot ride a stale startup verdict.
|
|
133
169
|
if (command === HERMES_COMMAND && clientFactory === createAcpClient) {
|
|
134
|
-
const profile = probeHermesRuntime({ command, env: childEnv })
|
|
170
|
+
const profile = probeHermesRuntime({ command, env: childEnv, strictBootstrap: true })
|
|
135
171
|
if (!profile.available) throw new Error(`Hermes profile safety check failed: ${profile.reason || 'runtime unavailable'}`)
|
|
172
|
+
// The executable is the installed venv Python, never the mutable
|
|
173
|
+
// profile wrapper. HERMES_HOME is the probe-verified isolated profile.
|
|
174
|
+
launchCommand = profile.python
|
|
175
|
+
launchArgs = [profile.bootstrap]
|
|
176
|
+
childEnv = { ...childEnv, HERMES_HOME: profile.profile, THINKPOOL_HERMES_ACP_POLICY: hermesPolicyEnv(policyRole, { mcpTools: requiredMcpTools }) }
|
|
136
177
|
}
|
|
137
178
|
let retired = false
|
|
138
179
|
retireClient = () => { retired = true }
|
|
139
180
|
client = clientFactory({
|
|
140
|
-
command, args, cwd: activeCwd, env: childEnv,
|
|
181
|
+
command: launchCommand, args: launchArgs, cwd: activeCwd, env: childEnv,
|
|
141
182
|
onNotification,
|
|
142
183
|
onRequest,
|
|
143
184
|
onStderr: (text) => { stderrTail = (stderrTail + text).slice(-2000) },
|
|
@@ -175,7 +216,8 @@ export function startHermesSession({
|
|
|
175
216
|
} else state = await client.request('session/new', params, 30_000)
|
|
176
217
|
if (!sessionId) sessionId = state?.sessionId || null
|
|
177
218
|
if (!sessionId) throw new Error('Hermes ACP did not return a session id')
|
|
178
|
-
let
|
|
219
|
+
let toolProbe = await probeMcpTools()
|
|
220
|
+
let missingMcpTools = toolProbe.missing
|
|
179
221
|
// Hermes treats ACP-provided MCP registration as non-fatal. A transient
|
|
180
222
|
// first connection can therefore leave an otherwise healthy session with
|
|
181
223
|
// only built-in tools. Re-running the released resume lifecycle retries
|
|
@@ -186,10 +228,12 @@ export function startHermesSession({
|
|
|
186
228
|
resuming = true
|
|
187
229
|
state = await client.request('session/resume', { ...params, sessionId }, 30_000)
|
|
188
230
|
} finally { resuming = false }
|
|
189
|
-
|
|
231
|
+
toolProbe = await probeMcpTools()
|
|
232
|
+
missingMcpTools = toolProbe.missing
|
|
190
233
|
}
|
|
191
|
-
if (missingMcpTools.length) {
|
|
192
|
-
|
|
234
|
+
if (missingMcpTools.length || toolProbe.forbidden.length) {
|
|
235
|
+
const details = [missingMcpTools.length ? `missing ${missingMcpTools.join(', ')}` : '', toolProbe.forbidden.length ? `forbidden ${toolProbe.forbidden.join(', ')}` : ''].filter(Boolean).join('; ')
|
|
236
|
+
throw new Error(`Hermes ThinkPool MCP readiness failed; ${details}`)
|
|
193
237
|
}
|
|
194
238
|
const serverModel = state?.models?.currentModelId || null
|
|
195
239
|
activeModel = requestedModel || serverModel || activeModel
|
|
@@ -197,8 +241,9 @@ export function startHermesSession({
|
|
|
197
241
|
// ACP catalog. Acknowledge that model BEFORE publishing the initial system/
|
|
198
242
|
// models events; otherwise the worker runs the requested model while the UI
|
|
199
243
|
// and persisted resume record lie that it still uses the profile default.
|
|
200
|
-
if (requestedModel
|
|
244
|
+
if (requestedModel) {
|
|
201
245
|
await client.request('session/set_model', { sessionId, modelId: activeModel })
|
|
246
|
+
await assertMcpReadiness()
|
|
202
247
|
}
|
|
203
248
|
const publishedModels = state?.models
|
|
204
249
|
? { ...state.models, currentModelId: activeModel }
|
|
@@ -209,7 +254,7 @@ export function startHermesSession({
|
|
|
209
254
|
if (state?.modes?.availableModes?.some((item) => item.id === acpMode) && state.modes.currentModeId !== acpMode) {
|
|
210
255
|
await client.request('session/set_mode', { sessionId, modeId: acpMode })
|
|
211
256
|
}
|
|
212
|
-
emit({ kind: 'capabilities', runtime: 'hermes', protocol: 'acp', protocolVersion: initialized?.protocolVersion, capabilities: initialized?.agentCapabilities || {}, models: modelList(state?.models), flow:
|
|
257
|
+
emit({ kind: 'capabilities', runtime: 'hermes', protocol: 'acp', protocolVersion: initialized?.protocolVersion, capabilities: initialized?.agentCapabilities || {}, models: modelList(state?.models), flow: command === HERMES_COMMAND && clientFactory === createAcpClient })
|
|
213
258
|
})().catch((error) => {
|
|
214
259
|
crashed = true
|
|
215
260
|
client?.end()
|
|
@@ -328,7 +373,11 @@ export function startHermesSession({
|
|
|
328
373
|
if (!nextModel || turnActive || modelSwitchPending) return false
|
|
329
374
|
const requested = String(nextModel)
|
|
330
375
|
modelSwitchPending = true
|
|
331
|
-
void boot().then(() => client.request('session/set_model', { sessionId, modelId: requested })).then(() => {
|
|
376
|
+
void boot().then(() => client.request('session/set_model', { sessionId, modelId: requested })).then(async () => {
|
|
377
|
+
// 0.18.2 reconstructs state.agent during set_model. The bootstrap
|
|
378
|
+
// re-registers its MCP; this independent zero-inference /tools check is
|
|
379
|
+
// the transaction's acknowledgement boundary.
|
|
380
|
+
await assertMcpReadiness()
|
|
332
381
|
if (ended) return
|
|
333
382
|
activeModel = requested
|
|
334
383
|
if (mapper) mapper.model = activeModel
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "thinkpool-pair",
|
|
3
|
-
"version": "0.7.
|
|
3
|
+
"version": "0.7.248",
|
|
4
4
|
"description": "Share a local coding-agent CLI (Claude Code, Codex, Gemini, Aider, …) into a ThinkPool Code room, live.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -24,6 +24,8 @@
|
|
|
24
24
|
"codex-event-mapper.mjs",
|
|
25
25
|
"acp-client.mjs",
|
|
26
26
|
"hermes-session.mjs",
|
|
27
|
+
"hermes-policy.mjs",
|
|
28
|
+
"hermes-acp-bootstrap.py",
|
|
27
29
|
"hermes-event-mapper.mjs",
|
|
28
30
|
"hermes-probe.mjs",
|
|
29
31
|
"hermes-setup.mjs",
|
|
@@ -47,6 +49,7 @@
|
|
|
47
49
|
"cross-terminal.mjs",
|
|
48
50
|
"lane-lifecycle.mjs",
|
|
49
51
|
"interrupted-resume.mjs",
|
|
52
|
+
"dispatch-lease.mjs",
|
|
50
53
|
"flow-conductor.mjs",
|
|
51
54
|
"flow-worktree.mjs",
|
|
52
55
|
"flow-task-graph.mjs",
|
|
@@ -56,6 +59,7 @@
|
|
|
56
59
|
"viewport.mjs",
|
|
57
60
|
"design-edit.mjs",
|
|
58
61
|
"flow-review.mjs",
|
|
62
|
+
"review-check.mjs",
|
|
59
63
|
"flow-review-gate.mjs",
|
|
60
64
|
"flow-review-reflect.mjs",
|
|
61
65
|
"flow-assembly.mjs",
|
package/review-check.mjs
ADDED
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
// Fixed, bridge-owned review checks. Models receive only { target, checkId }.
|
|
2
|
+
// This is host execution parity with Claude/Codex Flow, not a malicious-code sandbox.
|
|
3
|
+
import fs from 'node:fs'
|
|
4
|
+
import fsp from 'node:fs/promises'
|
|
5
|
+
import os from 'node:os'
|
|
6
|
+
import path from 'node:path'
|
|
7
|
+
import { spawn } from 'node:child_process'
|
|
8
|
+
import { execFile } from 'node:child_process'
|
|
9
|
+
import { promisify } from 'node:util'
|
|
10
|
+
|
|
11
|
+
const execFileAsync = promisify(execFile)
|
|
12
|
+
|
|
13
|
+
const CHECKS = Object.freeze({
|
|
14
|
+
'node-test': ['node', ['--test']], 'npm-test': ['npm', ['test']],
|
|
15
|
+
'npm-build': ['npm', ['run', 'build']], pytest: ['pytest', []],
|
|
16
|
+
'cargo-test': ['cargo', ['test']], 'go-test': ['go', ['test', './...']],
|
|
17
|
+
})
|
|
18
|
+
const OUTPUT_CAP = 512 * 1024
|
|
19
|
+
|
|
20
|
+
export function reviewCheckCommand(checkId) {
|
|
21
|
+
if (!Object.hasOwn(CHECKS, checkId)) throw new Error('unapproved review check')
|
|
22
|
+
const [command, args] = CHECKS[checkId]
|
|
23
|
+
return { command, args: [...args] }
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
export async function reviewCheckEnv(env = process.env, root = null) {
|
|
27
|
+
root ||= await fsp.mkdtemp(path.join(os.tmpdir(), 'tp-review-env-'))
|
|
28
|
+
const keptPath = env.PATH || process.env.PATH || ''
|
|
29
|
+
const home = path.join(root, 'home')
|
|
30
|
+
const npmCache = path.join(root, 'npm-cache')
|
|
31
|
+
const tmp = path.join(root, 'tmp')
|
|
32
|
+
await Promise.all([fsp.mkdir(home, { recursive: true }), fsp.mkdir(npmCache, { recursive: true }), fsp.mkdir(tmp, { recursive: true })])
|
|
33
|
+
return { PATH: keptPath, HOME: home, TMPDIR: path.join(root, 'tmp'), npm_config_cache: npmCache, npm_config_userconfig: path.join(home, '.npmrc'), GIT_CONFIG_NOSYSTEM: '1' }
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export function validateReviewTarget(target, snapshots) {
|
|
37
|
+
if (typeof target !== 'string' || !/^[a-z0-9-]{1,64}$/.test(target)) throw new Error('invalid reviewed target')
|
|
38
|
+
const snapshot = (Array.isArray(snapshots) ? snapshots : []).find((item) => item?.taskKey === target)
|
|
39
|
+
if (!snapshot || !/^[0-9a-f]{40}$/i.test(snapshot.sha) || !path.isAbsolute(snapshot.cwd || '')) throw new Error('reviewed target is not pinned')
|
|
40
|
+
return snapshot
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export function validateReviewPath(filePath) {
|
|
44
|
+
if (typeof filePath !== 'string' || !filePath || filePath.length > 512 || path.posix.isAbsolute(filePath)) throw new Error('invalid review path')
|
|
45
|
+
const normalized = path.posix.normalize(filePath)
|
|
46
|
+
if (normalized === '.' || normalized === '..' || normalized.startsWith('../') || normalized.split('/').some((part) => !part || part === '.')) throw new Error('invalid review path')
|
|
47
|
+
return normalized
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
// Read the pinned Git blob, never the reviewer worktree. git ls-tree refuses
|
|
51
|
+
// directories and symlinks before git show is allowed to materialize bytes.
|
|
52
|
+
export async function readReviewFile({ target, filePath, snapshots, maxBytes = 512 * 1024 } = {}) {
|
|
53
|
+
const snapshot = validateReviewTarget(target, snapshots)
|
|
54
|
+
const normalized = validateReviewPath(filePath)
|
|
55
|
+
const cap = Number.isInteger(maxBytes) && maxBytes > 0 && maxBytes <= 512 * 1024 ? maxBytes : 512 * 1024
|
|
56
|
+
const listed = await execFileAsync('git', ['-C', snapshot.cwd, 'ls-tree', '-z', snapshot.sha, '--', normalized], { encoding: 'buffer', maxBuffer: 1024 * 1024 })
|
|
57
|
+
const record = Buffer.from(listed.stdout).toString('utf8').split('\0').filter(Boolean)[0] || ''
|
|
58
|
+
const match = /^(100[0-7]{3}) blob ([0-9a-f]{40})\t(.+)$/.exec(record)
|
|
59
|
+
if (!match || match[3] !== normalized) throw new Error('review file is not a pinned regular file')
|
|
60
|
+
const shown = await execFileAsync('git', ['-C', snapshot.cwd, 'show', `${snapshot.sha}:${normalized}`], { encoding: 'buffer', maxBuffer: cap + 1 })
|
|
61
|
+
const bytes = Buffer.from(shown.stdout)
|
|
62
|
+
if (bytes.length > cap) throw new Error('review file exceeds byte limit')
|
|
63
|
+
return { target, sha: snapshot.sha, path: normalized, content: bytes.toString('utf8'), bytes: bytes.length }
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
function boundedPush(parts, value) {
|
|
67
|
+
const text = String(value || '')
|
|
68
|
+
parts.push(text)
|
|
69
|
+
let total = parts.reduce((n, part) => n + part.length, 0)
|
|
70
|
+
while (total > OUTPUT_CAP && parts.length) total -= parts.shift().length
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export async function killTree(child, signal = 'SIGTERM', { platform = process.platform, spawnImpl = spawn, taskkillTimeoutMs = 1_000 } = {}) {
|
|
74
|
+
if (!child?.pid) return true
|
|
75
|
+
if (platform === 'win32' && signal === 'SIGKILL') {
|
|
76
|
+
// Windows has no process-group equivalent. This is trusted-host parity,
|
|
77
|
+
// not unescapable process-tree isolation. taskkill itself can hang on a
|
|
78
|
+
// hostile/broken host, so race it: cleanup remains best-effort and the
|
|
79
|
+
// review request always settles.
|
|
80
|
+
return await new Promise((resolve) => {
|
|
81
|
+
let done = false
|
|
82
|
+
let timer = null
|
|
83
|
+
const finish = (ok) => {
|
|
84
|
+
if (done) return
|
|
85
|
+
done = true
|
|
86
|
+
if (timer) clearTimeout(timer)
|
|
87
|
+
resolve(ok)
|
|
88
|
+
}
|
|
89
|
+
try {
|
|
90
|
+
const taskkill = spawnImpl('taskkill', ['/pid', String(child.pid), '/T', '/F'], { stdio: 'ignore', windowsHide: true })
|
|
91
|
+
taskkill.once('close', (code) => finish(code === 0))
|
|
92
|
+
taskkill.once('error', () => finish(false))
|
|
93
|
+
timer = setTimeout(() => { try { taskkill.kill?.() } catch {} finish(false) }, taskkillTimeoutMs)
|
|
94
|
+
} catch { finish(false) }
|
|
95
|
+
})
|
|
96
|
+
}
|
|
97
|
+
try { platform === 'win32' ? child.kill(signal) : process.kill(-child.pid, signal); return true } catch { try { return child.kill(signal) } catch { return false } }
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
export function runProcess(command, args, { cwd, env, timeoutMs = 120_000, killGraceMs = 250 } = {}) {
|
|
101
|
+
return new Promise((resolve) => {
|
|
102
|
+
const stdout = [], stderr = []
|
|
103
|
+
let timedOut = false
|
|
104
|
+
let settled = false
|
|
105
|
+
let child
|
|
106
|
+
let timer = null
|
|
107
|
+
let forceTimer = null
|
|
108
|
+
let settleTimer = null
|
|
109
|
+
const settle = (code) => {
|
|
110
|
+
if (settled) return
|
|
111
|
+
settled = true
|
|
112
|
+
if (timer) clearTimeout(timer)
|
|
113
|
+
if (forceTimer) clearTimeout(forceTimer)
|
|
114
|
+
if (settleTimer) clearTimeout(settleTimer)
|
|
115
|
+
resolve({ exitCode: Number.isInteger(code) ? code : null, stdout: stdout.join('').slice(-OUTPUT_CAP), stderr: stderr.join('').slice(-OUTPUT_CAP), timedOut })
|
|
116
|
+
}
|
|
117
|
+
try { child = spawn(command, args, { cwd, env, detached: process.platform !== 'win32', stdio: ['ignore', 'pipe', 'pipe'] }) }
|
|
118
|
+
catch (error) { stderr.push(String(error?.message || error)); settle(null); return }
|
|
119
|
+
child.stdout.on('data', (value) => boundedPush(stdout, value))
|
|
120
|
+
child.stderr.on('data', (value) => boundedPush(stderr, value))
|
|
121
|
+
timer = setTimeout(() => {
|
|
122
|
+
timedOut = true; void killTree(child)
|
|
123
|
+
forceTimer = setTimeout(async () => {
|
|
124
|
+
if (!await killTree(child, 'SIGKILL')) boundedPush(stderr, 'review check cleanup could not confirm process-tree termination')
|
|
125
|
+
// SIGKILL is unconditional on POSIX, but a platform/process failure must
|
|
126
|
+
// not leave the MCP call pending forever. A second best-effort kill then
|
|
127
|
+
// settles the caller; the scratch finally block can run.
|
|
128
|
+
settleTimer = setTimeout(() => {
|
|
129
|
+
void killTree(child, 'SIGKILL')
|
|
130
|
+
boundedPush(stderr, 'review check process did not report close after forced termination')
|
|
131
|
+
settle(null)
|
|
132
|
+
}, 1_000)
|
|
133
|
+
}, killGraceMs)
|
|
134
|
+
}, timeoutMs)
|
|
135
|
+
child.once('error', (error) => boundedPush(stderr, error?.message || error))
|
|
136
|
+
child.once('close', (code) => settle(code))
|
|
137
|
+
})
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
export async function runReviewCheck({ target, checkId, snapshots, env = process.env, tmpdir = os.tmpdir(), timeoutMs } = {}) {
|
|
141
|
+
const snapshot = validateReviewTarget(target, snapshots)
|
|
142
|
+
const archiveDir = await fsp.mkdtemp(path.join(tmpdir, 'tp-review-'))
|
|
143
|
+
const archive = path.join(archiveDir, 'source.tar')
|
|
144
|
+
const scratch = path.join(archiveDir, 'tree')
|
|
145
|
+
try {
|
|
146
|
+
await fsp.mkdir(scratch)
|
|
147
|
+
const archiveResult = await runProcess('git', ['-C', snapshot.cwd, 'archive', '--format=tar', `--output=${archive}`, snapshot.sha], { env: await reviewCheckEnv(env, archiveDir), timeoutMs: 15_000 })
|
|
148
|
+
if (archiveResult.exitCode !== 0 || archiveResult.timedOut) return { pass: false, target, sha: snapshot.sha, command: ['git', 'archive'], exitCode: archiveResult.exitCode, ...archiveResult }
|
|
149
|
+
const extract = await runProcess('tar', ['-xf', archive, '-C', scratch], { env: await reviewCheckEnv(env, archiveDir), timeoutMs: 15_000 })
|
|
150
|
+
if (extract.exitCode !== 0 || extract.timedOut) return { pass: false, target, sha: snapshot.sha, command: ['tar', '-xf'], exitCode: extract.exitCode, ...extract }
|
|
151
|
+
const spec = reviewCheckCommand(checkId)
|
|
152
|
+
const result = await runProcess(spec.command, spec.args, { cwd: scratch, env: await reviewCheckEnv(env, archiveDir), timeoutMs })
|
|
153
|
+
return { pass: result.exitCode === 0 && !result.timedOut, target, sha: snapshot.sha, command: [spec.command, ...spec.args], ...result }
|
|
154
|
+
} finally { await fsp.rm(archiveDir, { recursive: true, force: true }) }
|
|
155
|
+
}
|
package/runtime-registry.mjs
CHANGED
|
@@ -13,7 +13,7 @@ const RUNTIMES = Object.freeze({
|
|
|
13
13
|
}),
|
|
14
14
|
hermes: Object.freeze({
|
|
15
15
|
id: 'hermes', command: 'thinkpool', label: 'Hermes Agent', protocol: 'acp',
|
|
16
|
-
structured: true, flow:
|
|
16
|
+
structured: true, flow: true, canSteer: true, images: true, nativeModelCatalog: true, catalogRequiresSession: true, effortControl: false, defaultMode: 'default',
|
|
17
17
|
modes: Object.freeze(['default', 'acceptEdits']),
|
|
18
18
|
beta: true,
|
|
19
19
|
}),
|