thinkpool-pair 0.7.245 → 0.7.247

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,6 +6,7 @@ import { randomUUID } from 'node:crypto'
6
6
  import { AcpClient } from './acp-client.mjs'
7
7
  import { HermesEventMapper, hermesToolFor } from './hermes-event-mapper.mjs'
8
8
  import { probeHermesRuntime } from './hermes-probe.mjs'
9
+ import { hermesExactInventory, hermesPolicyEnv, hermesRoleFor } from './hermes-policy.mjs'
9
10
  import { startCodexMcpHttp } from './codex-mcp-http.mjs'
10
11
  import { classifyRisk } from './claude-session.mjs'
11
12
 
@@ -44,7 +45,7 @@ export function startHermesSession({
44
45
  cwd, model, resume, env = process.env, mode = 'default', onEvent, requestPermission,
45
46
  roomContext, terminalRolePrompt, rolePrompt, mcpServers, requiredMcpTools = [], prepareCwd = null,
46
47
  command = HERMES_COMMAND, args = ['acp'], clientFactory = createAcpClient,
47
- mcpHttpFactory = startCodexMcpHttp, lazy = false,
48
+ mcpHttpFactory = startCodexMcpHttp, lazy = false, hermesRole = null,
48
49
  } = {}) {
49
50
  let activeCwd = cwd
50
51
  const requestedModel = model || null
@@ -68,6 +69,7 @@ export function startHermesSession({
68
69
  let resuming = false
69
70
  let inventoryProbe = null
70
71
  let modelSwitchPending = false
72
+ const policyRole = hermesRole || hermesRoleFor({})
71
73
 
72
74
  const emit = (event) => { try { onEvent?.(event) } catch { /* consumer isolation */ } }
73
75
 
@@ -101,7 +103,14 @@ export function startHermesSession({
101
103
  async function probeMcpTools() {
102
104
  const required = [...new Set((Array.isArray(requiredMcpTools) ? requiredMcpTools : [])
103
105
  .map((name) => String(name || '').trim()).filter(Boolean))]
104
- if (!required.length) return []
106
+ // Direct runtime tests and unscoped upstream callers have no bridge role
107
+ // contract to prove. Every bridge-created Hermes lane supplies its required
108
+ // MCP list; only those lanes enter the exact-inventory transaction.
109
+ if (!required.length) return { inventory: '', missing: [], forbidden: [] }
110
+ const exact = hermesExactInventory(policyRole, { mcpTools: required })
111
+ const allowedBuiltins = exact.builtinTools
112
+ const requiredBuiltins = exact.requiredBuiltinTools
113
+ const expectedMcp = exact.mcpTools
105
114
  const chunks = []
106
115
  inventoryProbe = chunks
107
116
  try {
@@ -112,7 +121,32 @@ export function startHermesSession({
112
121
  }, 30_000)
113
122
  } finally { inventoryProbe = null }
114
123
  const inventory = chunks.join('')
115
- return required.filter((name) => !inventory.includes(`mcp__thinkpool__${name}`))
124
+ const has = (name) => new RegExp(`(?:^|[^a-z0-9_])${name}(?:$|[^a-z0-9_])`, 'i').test(inventory)
125
+ const mcpNames = [...inventory.matchAll(/\bmcp(?:__|_)[a-z0-9_]+/gi)].map((item) => item[0])
126
+ const expectedMcpNames = new Set(expectedMcp.flatMap((name) => [`mcp__thinkpool__${name}`, `mcp_thinkpool_${name}`]))
127
+ const unexpectedMcp = [...new Set(mcpNames.filter((name) => !expectedMcpNames.has(name)))]
128
+ // This is a zero-inference ACP inventory command. It proves the complete
129
+ // role schema (required built-ins plus exactly this lane's ThinkPool MCP
130
+ // set). Optional provider/browser capabilities may be absent, but any
131
+ // presented builtin outside the role allowlist is still a hard failure.
132
+ return {
133
+ inventory,
134
+ missing: [
135
+ ...requiredBuiltins.filter((name) => !has(name)),
136
+ ...expectedMcp.filter((name) => !has(`mcp__thinkpool__${name}`) && !has(`mcp_thinkpool_${name}`)),
137
+ ],
138
+ forbidden: [
139
+ ...exact.allBuiltinTools.filter((name) => !allowedBuiltins.includes(name) && has(name)),
140
+ ...unexpectedMcp,
141
+ ],
142
+ }
143
+ }
144
+
145
+ async function assertMcpReadiness() {
146
+ const probe = await probeMcpTools()
147
+ if (!probe.missing.length && !probe.forbidden.length) return probe
148
+ const details = [probe.missing.length ? `missing ${probe.missing.join(', ')}` : '', probe.forbidden.length ? `forbidden ${probe.forbidden.join(', ')}` : ''].filter(Boolean).join('; ')
149
+ throw new Error(`Hermes ThinkPool MCP readiness failed; ${details}`)
116
150
  }
117
151
 
118
152
  async function boot() {
@@ -126,18 +160,25 @@ export function startHermesSession({
126
160
  }
127
161
  const thinkpool = mcpServers?.thinkpool
128
162
  if (thinkpool && !mcpHttp) mcpHttp = await mcpHttpFactory({ sdkServer: thinkpool })
129
- const childEnv = hermesChildEnv(env)
163
+ let childEnv = hermesChildEnv(env)
164
+ let launchCommand = command
165
+ let launchArgs = args
130
166
  // The bridge advertises Hermes only after this probe, but lazy lanes may
131
167
  // boot much later. Re-check immediately before every real ACP process so
132
168
  // a removed/changed delegation hook cannot ride a stale startup verdict.
133
169
  if (command === HERMES_COMMAND && clientFactory === createAcpClient) {
134
- const profile = probeHermesRuntime({ command, env: childEnv })
170
+ const profile = probeHermesRuntime({ command, env: childEnv, strictBootstrap: true })
135
171
  if (!profile.available) throw new Error(`Hermes profile safety check failed: ${profile.reason || 'runtime unavailable'}`)
172
+ // The executable is the installed venv Python, never the mutable
173
+ // profile wrapper. HERMES_HOME is the probe-verified isolated profile.
174
+ launchCommand = profile.python
175
+ launchArgs = [profile.bootstrap]
176
+ childEnv = { ...childEnv, HERMES_HOME: profile.profile, THINKPOOL_HERMES_ACP_POLICY: hermesPolicyEnv(policyRole, { mcpTools: requiredMcpTools }) }
136
177
  }
137
178
  let retired = false
138
179
  retireClient = () => { retired = true }
139
180
  client = clientFactory({
140
- command, args, cwd: activeCwd, env: childEnv,
181
+ command: launchCommand, args: launchArgs, cwd: activeCwd, env: childEnv,
141
182
  onNotification,
142
183
  onRequest,
143
184
  onStderr: (text) => { stderrTail = (stderrTail + text).slice(-2000) },
@@ -175,7 +216,8 @@ export function startHermesSession({
175
216
  } else state = await client.request('session/new', params, 30_000)
176
217
  if (!sessionId) sessionId = state?.sessionId || null
177
218
  if (!sessionId) throw new Error('Hermes ACP did not return a session id')
178
- let missingMcpTools = await probeMcpTools()
219
+ let toolProbe = await probeMcpTools()
220
+ let missingMcpTools = toolProbe.missing
179
221
  // Hermes treats ACP-provided MCP registration as non-fatal. A transient
180
222
  // first connection can therefore leave an otherwise healthy session with
181
223
  // only built-in tools. Re-running the released resume lifecycle retries
@@ -186,10 +228,12 @@ export function startHermesSession({
186
228
  resuming = true
187
229
  state = await client.request('session/resume', { ...params, sessionId }, 30_000)
188
230
  } finally { resuming = false }
189
- missingMcpTools = await probeMcpTools()
231
+ toolProbe = await probeMcpTools()
232
+ missingMcpTools = toolProbe.missing
190
233
  }
191
- if (missingMcpTools.length) {
192
- throw new Error(`Hermes ThinkPool MCP readiness failed; missing ${missingMcpTools.join(', ')}`)
234
+ if (missingMcpTools.length || toolProbe.forbidden.length) {
235
+ const details = [missingMcpTools.length ? `missing ${missingMcpTools.join(', ')}` : '', toolProbe.forbidden.length ? `forbidden ${toolProbe.forbidden.join(', ')}` : ''].filter(Boolean).join('; ')
236
+ throw new Error(`Hermes ThinkPool MCP readiness failed; ${details}`)
193
237
  }
194
238
  const serverModel = state?.models?.currentModelId || null
195
239
  activeModel = requestedModel || serverModel || activeModel
@@ -197,8 +241,9 @@ export function startHermesSession({
197
241
  // ACP catalog. Acknowledge that model BEFORE publishing the initial system/
198
242
  // models events; otherwise the worker runs the requested model while the UI
199
243
  // and persisted resume record lie that it still uses the profile default.
200
- if (requestedModel && requestedModel !== serverModel) {
244
+ if (requestedModel) {
201
245
  await client.request('session/set_model', { sessionId, modelId: activeModel })
246
+ await assertMcpReadiness()
202
247
  }
203
248
  const publishedModels = state?.models
204
249
  ? { ...state.models, currentModelId: activeModel }
@@ -209,7 +254,7 @@ export function startHermesSession({
209
254
  if (state?.modes?.availableModes?.some((item) => item.id === acpMode) && state.modes.currentModeId !== acpMode) {
210
255
  await client.request('session/set_mode', { sessionId, modeId: acpMode })
211
256
  }
212
- emit({ kind: 'capabilities', runtime: 'hermes', protocol: 'acp', protocolVersion: initialized?.protocolVersion, capabilities: initialized?.agentCapabilities || {}, models: modelList(state?.models), flow: false })
257
+ emit({ kind: 'capabilities', runtime: 'hermes', protocol: 'acp', protocolVersion: initialized?.protocolVersion, capabilities: initialized?.agentCapabilities || {}, models: modelList(state?.models), flow: command === HERMES_COMMAND && clientFactory === createAcpClient })
213
258
  })().catch((error) => {
214
259
  crashed = true
215
260
  client?.end()
@@ -328,7 +373,11 @@ export function startHermesSession({
328
373
  if (!nextModel || turnActive || modelSwitchPending) return false
329
374
  const requested = String(nextModel)
330
375
  modelSwitchPending = true
331
- void boot().then(() => client.request('session/set_model', { sessionId, modelId: requested })).then(() => {
376
+ void boot().then(() => client.request('session/set_model', { sessionId, modelId: requested })).then(async () => {
377
+ // 0.18.2 reconstructs state.agent during set_model. The bootstrap
378
+ // re-registers its MCP; this independent zero-inference /tools check is
379
+ // the transaction's acknowledgement boundary.
380
+ await assertMcpReadiness()
332
381
  if (ended) return
333
382
  activeModel = requested
334
383
  if (mapper) mapper.model = activeModel
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "thinkpool-pair",
3
- "version": "0.7.245",
3
+ "version": "0.7.247",
4
4
  "description": "Share a local coding-agent CLI (Claude Code, Codex, Gemini, Aider, …) into a ThinkPool Code room, live.",
5
5
  "type": "module",
6
6
  "bin": {
@@ -9,6 +9,8 @@
9
9
  "files": [
10
10
  "bridge.mjs",
11
11
  "sdk-smoke.mjs",
12
+ "sdk-admission.mjs",
13
+ "sdk-admission.mjs",
12
14
  "launcher.mjs",
13
15
  "byok-detect.mjs",
14
16
  "context-windows.mjs",
@@ -22,6 +24,8 @@
22
24
  "codex-event-mapper.mjs",
23
25
  "acp-client.mjs",
24
26
  "hermes-session.mjs",
27
+ "hermes-policy.mjs",
28
+ "hermes-acp-bootstrap.py",
25
29
  "hermes-event-mapper.mjs",
26
30
  "hermes-probe.mjs",
27
31
  "hermes-setup.mjs",
@@ -54,6 +58,7 @@
54
58
  "viewport.mjs",
55
59
  "design-edit.mjs",
56
60
  "flow-review.mjs",
61
+ "review-check.mjs",
57
62
  "flow-review-gate.mjs",
58
63
  "flow-review-reflect.mjs",
59
64
  "flow-assembly.mjs",
@@ -0,0 +1,155 @@
1
+ // Fixed, bridge-owned review checks. Models receive only { target, checkId }.
2
+ // This is host execution parity with Claude/Codex Flow, not a malicious-code sandbox.
3
+ import fs from 'node:fs'
4
+ import fsp from 'node:fs/promises'
5
+ import os from 'node:os'
6
+ import path from 'node:path'
7
+ import { spawn } from 'node:child_process'
8
+ import { execFile } from 'node:child_process'
9
+ import { promisify } from 'node:util'
10
+
11
+ const execFileAsync = promisify(execFile)
12
+
13
+ const CHECKS = Object.freeze({
14
+ 'node-test': ['node', ['--test']], 'npm-test': ['npm', ['test']],
15
+ 'npm-build': ['npm', ['run', 'build']], pytest: ['pytest', []],
16
+ 'cargo-test': ['cargo', ['test']], 'go-test': ['go', ['test', './...']],
17
+ })
18
+ const OUTPUT_CAP = 512 * 1024
19
+
20
+ export function reviewCheckCommand(checkId) {
21
+ if (!Object.hasOwn(CHECKS, checkId)) throw new Error('unapproved review check')
22
+ const [command, args] = CHECKS[checkId]
23
+ return { command, args: [...args] }
24
+ }
25
+
26
+ export async function reviewCheckEnv(env = process.env, root = null) {
27
+ root ||= await fsp.mkdtemp(path.join(os.tmpdir(), 'tp-review-env-'))
28
+ const keptPath = env.PATH || process.env.PATH || ''
29
+ const home = path.join(root, 'home')
30
+ const npmCache = path.join(root, 'npm-cache')
31
+ const tmp = path.join(root, 'tmp')
32
+ await Promise.all([fsp.mkdir(home, { recursive: true }), fsp.mkdir(npmCache, { recursive: true }), fsp.mkdir(tmp, { recursive: true })])
33
+ return { PATH: keptPath, HOME: home, TMPDIR: path.join(root, 'tmp'), npm_config_cache: npmCache, npm_config_userconfig: path.join(home, '.npmrc'), GIT_CONFIG_NOSYSTEM: '1' }
34
+ }
35
+
36
+ export function validateReviewTarget(target, snapshots) {
37
+ if (typeof target !== 'string' || !/^[a-z0-9-]{1,64}$/.test(target)) throw new Error('invalid reviewed target')
38
+ const snapshot = (Array.isArray(snapshots) ? snapshots : []).find((item) => item?.taskKey === target)
39
+ if (!snapshot || !/^[0-9a-f]{40}$/i.test(snapshot.sha) || !path.isAbsolute(snapshot.cwd || '')) throw new Error('reviewed target is not pinned')
40
+ return snapshot
41
+ }
42
+
43
+ export function validateReviewPath(filePath) {
44
+ if (typeof filePath !== 'string' || !filePath || filePath.length > 512 || path.posix.isAbsolute(filePath)) throw new Error('invalid review path')
45
+ const normalized = path.posix.normalize(filePath)
46
+ if (normalized === '.' || normalized === '..' || normalized.startsWith('../') || normalized.split('/').some((part) => !part || part === '.')) throw new Error('invalid review path')
47
+ return normalized
48
+ }
49
+
50
+ // Read the pinned Git blob, never the reviewer worktree. git ls-tree refuses
51
+ // directories and symlinks before git show is allowed to materialize bytes.
52
+ export async function readReviewFile({ target, filePath, snapshots, maxBytes = 512 * 1024 } = {}) {
53
+ const snapshot = validateReviewTarget(target, snapshots)
54
+ const normalized = validateReviewPath(filePath)
55
+ const cap = Number.isInteger(maxBytes) && maxBytes > 0 && maxBytes <= 512 * 1024 ? maxBytes : 512 * 1024
56
+ const listed = await execFileAsync('git', ['-C', snapshot.cwd, 'ls-tree', '-z', snapshot.sha, '--', normalized], { encoding: 'buffer', maxBuffer: 1024 * 1024 })
57
+ const record = Buffer.from(listed.stdout).toString('utf8').split('\0').filter(Boolean)[0] || ''
58
+ const match = /^(100[0-7]{3}) blob ([0-9a-f]{40})\t(.+)$/.exec(record)
59
+ if (!match || match[3] !== normalized) throw new Error('review file is not a pinned regular file')
60
+ const shown = await execFileAsync('git', ['-C', snapshot.cwd, 'show', `${snapshot.sha}:${normalized}`], { encoding: 'buffer', maxBuffer: cap + 1 })
61
+ const bytes = Buffer.from(shown.stdout)
62
+ if (bytes.length > cap) throw new Error('review file exceeds byte limit')
63
+ return { target, sha: snapshot.sha, path: normalized, content: bytes.toString('utf8'), bytes: bytes.length }
64
+ }
65
+
66
+ function boundedPush(parts, value) {
67
+ const text = String(value || '')
68
+ parts.push(text)
69
+ let total = parts.reduce((n, part) => n + part.length, 0)
70
+ while (total > OUTPUT_CAP && parts.length) total -= parts.shift().length
71
+ }
72
+
73
+ export async function killTree(child, signal = 'SIGTERM', { platform = process.platform, spawnImpl = spawn, taskkillTimeoutMs = 1_000 } = {}) {
74
+ if (!child?.pid) return true
75
+ if (platform === 'win32' && signal === 'SIGKILL') {
76
+ // Windows has no process-group equivalent. This is trusted-host parity,
77
+ // not unescapable process-tree isolation. taskkill itself can hang on a
78
+ // hostile/broken host, so race it: cleanup remains best-effort and the
79
+ // review request always settles.
80
+ return await new Promise((resolve) => {
81
+ let done = false
82
+ let timer = null
83
+ const finish = (ok) => {
84
+ if (done) return
85
+ done = true
86
+ if (timer) clearTimeout(timer)
87
+ resolve(ok)
88
+ }
89
+ try {
90
+ const taskkill = spawnImpl('taskkill', ['/pid', String(child.pid), '/T', '/F'], { stdio: 'ignore', windowsHide: true })
91
+ taskkill.once('close', (code) => finish(code === 0))
92
+ taskkill.once('error', () => finish(false))
93
+ timer = setTimeout(() => { try { taskkill.kill?.() } catch {} finish(false) }, taskkillTimeoutMs)
94
+ } catch { finish(false) }
95
+ })
96
+ }
97
+ try { platform === 'win32' ? child.kill(signal) : process.kill(-child.pid, signal); return true } catch { try { return child.kill(signal) } catch { return false } }
98
+ }
99
+
100
+ export function runProcess(command, args, { cwd, env, timeoutMs = 120_000, killGraceMs = 250 } = {}) {
101
+ return new Promise((resolve) => {
102
+ const stdout = [], stderr = []
103
+ let timedOut = false
104
+ let settled = false
105
+ let child
106
+ let timer = null
107
+ let forceTimer = null
108
+ let settleTimer = null
109
+ const settle = (code) => {
110
+ if (settled) return
111
+ settled = true
112
+ if (timer) clearTimeout(timer)
113
+ if (forceTimer) clearTimeout(forceTimer)
114
+ if (settleTimer) clearTimeout(settleTimer)
115
+ resolve({ exitCode: Number.isInteger(code) ? code : null, stdout: stdout.join('').slice(-OUTPUT_CAP), stderr: stderr.join('').slice(-OUTPUT_CAP), timedOut })
116
+ }
117
+ try { child = spawn(command, args, { cwd, env, detached: process.platform !== 'win32', stdio: ['ignore', 'pipe', 'pipe'] }) }
118
+ catch (error) { stderr.push(String(error?.message || error)); settle(null); return }
119
+ child.stdout.on('data', (value) => boundedPush(stdout, value))
120
+ child.stderr.on('data', (value) => boundedPush(stderr, value))
121
+ timer = setTimeout(() => {
122
+ timedOut = true; void killTree(child)
123
+ forceTimer = setTimeout(async () => {
124
+ if (!await killTree(child, 'SIGKILL')) boundedPush(stderr, 'review check cleanup could not confirm process-tree termination')
125
+ // SIGKILL is unconditional on POSIX, but a platform/process failure must
126
+ // not leave the MCP call pending forever. A second best-effort kill then
127
+ // settles the caller; the scratch finally block can run.
128
+ settleTimer = setTimeout(() => {
129
+ void killTree(child, 'SIGKILL')
130
+ boundedPush(stderr, 'review check process did not report close after forced termination')
131
+ settle(null)
132
+ }, 1_000)
133
+ }, killGraceMs)
134
+ }, timeoutMs)
135
+ child.once('error', (error) => boundedPush(stderr, error?.message || error))
136
+ child.once('close', (code) => settle(code))
137
+ })
138
+ }
139
+
140
+ export async function runReviewCheck({ target, checkId, snapshots, env = process.env, tmpdir = os.tmpdir(), timeoutMs } = {}) {
141
+ const snapshot = validateReviewTarget(target, snapshots)
142
+ const archiveDir = await fsp.mkdtemp(path.join(tmpdir, 'tp-review-'))
143
+ const archive = path.join(archiveDir, 'source.tar')
144
+ const scratch = path.join(archiveDir, 'tree')
145
+ try {
146
+ await fsp.mkdir(scratch)
147
+ const archiveResult = await runProcess('git', ['-C', snapshot.cwd, 'archive', '--format=tar', `--output=${archive}`, snapshot.sha], { env: await reviewCheckEnv(env, archiveDir), timeoutMs: 15_000 })
148
+ if (archiveResult.exitCode !== 0 || archiveResult.timedOut) return { pass: false, target, sha: snapshot.sha, command: ['git', 'archive'], exitCode: archiveResult.exitCode, ...archiveResult }
149
+ const extract = await runProcess('tar', ['-xf', archive, '-C', scratch], { env: await reviewCheckEnv(env, archiveDir), timeoutMs: 15_000 })
150
+ if (extract.exitCode !== 0 || extract.timedOut) return { pass: false, target, sha: snapshot.sha, command: ['tar', '-xf'], exitCode: extract.exitCode, ...extract }
151
+ const spec = reviewCheckCommand(checkId)
152
+ const result = await runProcess(spec.command, spec.args, { cwd: scratch, env: await reviewCheckEnv(env, archiveDir), timeoutMs })
153
+ return { pass: result.exitCode === 0 && !result.timedOut, target, sha: snapshot.sha, command: [spec.command, ...spec.args], ...result }
154
+ } finally { await fsp.rm(archiveDir, { recursive: true, force: true }) }
155
+ }
@@ -13,7 +13,7 @@ const RUNTIMES = Object.freeze({
13
13
  }),
14
14
  hermes: Object.freeze({
15
15
  id: 'hermes', command: 'thinkpool', label: 'Hermes Agent', protocol: 'acp',
16
- structured: true, flow: false, canSteer: true, images: true, nativeModelCatalog: true, catalogRequiresSession: true, effortControl: false, defaultMode: 'default',
16
+ structured: true, flow: true, canSteer: true, images: true, nativeModelCatalog: true, catalogRequiresSession: true, effortControl: false, defaultMode: 'default',
17
17
  modes: Object.freeze(['default', 'acceptEdits']),
18
18
  beta: true,
19
19
  }),
@@ -0,0 +1,9 @@
1
+ // SDK checks protect agent-serving room processes only. Account/service commands
2
+ // remain available so an operator can inspect or repair a bad SDK installation.
3
+ export function requiresSdkAdmission(argv) {
4
+ return Boolean(argv?.[0]) && argv[0] !== 'verify-sdk'
5
+ }
6
+
7
+ export function sdkSmokePassed(output) {
8
+ return /^SMOKE:PASS:[^:\n]+(?:\n|$)/.test(String(output || ''))
9
+ }
package/session-store.mjs CHANGED
@@ -11,6 +11,7 @@
11
11
  import os from 'node:os'
12
12
  import fs from 'node:fs'
13
13
  import path from 'node:path'
14
+ import { SPAWN } from './cross-terminal.mjs'
14
15
 
15
16
  // TP_PAIR_ROOT override exists for tests + sandboxes so they never touch the real
16
17
  // ~/.thinkpool-pair store. Read LIVE per call (not captured at module load) so a test
@@ -35,9 +36,51 @@ const archiveDir = (room) => path.join(dir(room), '.archive')
35
36
  // Only resume the live SDK context for recent sessions — an expired session id
36
37
  // fails ("No conversation found"); past this window we restore transcript only.
37
38
  const RESUME_MAX_AGE_MS = 12 * 60 * 60 * 1000
38
- const KEEP = 8 // newest N UNNAMED session files per room (named are immune — see prune)
39
+ // A live snapshot exists precisely so a live structured lane can survive a bridge
40
+ // restart. Keep every lane the bridge can admit. Explicitly closed lanes move to
41
+ // .archive/, so closed/history retention stays separate from this live set.
42
+ export const LIVE_SESSION_MAX = SPAWN.machineMax
39
43
 
40
- function ensureDir(room) { try { fs.mkdirSync(dir(room), { recursive: true }) } catch { /* noop */ } }
44
+ function ensureDir(room) { fs.mkdirSync(dir(room), { recursive: true }) }
45
+
46
+ // Commit a replacement through a sibling file so a restart never observes a torn
47
+ // final record. The temp file is fsynced before the same-directory rename: readers
48
+ // see the complete previous value or the complete replacement, never a half-write.
49
+ // This is shared by session JSON plus the restart-critical names and PTY metadata.
50
+ let tempSerial = 0
51
+ let beforeAtomicRenameForTest = null
52
+ function atomicCommit(file, contents) {
53
+ const tmp = path.join(path.dirname(file), `.${path.basename(file)}.tmp.${process.pid}.${Date.now()}.${tempSerial++}`)
54
+ let fd = null
55
+ try {
56
+ fd = fs.openSync(tmp, 'wx', 0o600)
57
+ // fs.writeSync may legally write fewer bytes than requested. Never fsync and
58
+ // promote a partial temp file over the last good snapshot.
59
+ const bytes = Buffer.from(contents)
60
+ for (let offset = 0; offset < bytes.length;) {
61
+ const written = fs.writeSync(fd, bytes, offset, bytes.length - offset)
62
+ if (written <= 0) throw new Error(`short write while committing ${file}`)
63
+ offset += written
64
+ }
65
+ fs.fsyncSync(fd)
66
+ fs.closeSync(fd); fd = null
67
+ beforeAtomicRenameForTest?.({ file, tmp })
68
+ fs.renameSync(tmp, file)
69
+ return true
70
+ } catch (error) {
71
+ try { if (fd != null) fs.closeSync(fd) } catch { /* best effort */ }
72
+ try { fs.rmSync(tmp, { force: true }) } catch { /* best effort */ }
73
+ throw error
74
+ }
75
+ }
76
+ // Direct persistence tests inject a crash-equivalent immediately before rename.
77
+ // Production never configures this hook.
78
+ export function setAtomicCommitHookForTest(hook) {
79
+ beforeAtomicRenameForTest = typeof hook === 'function' ? hook : null
80
+ }
81
+ function persistenceError(action, file, error) {
82
+ console.error(`[thinkpool-pair session-store] ${action} failed for ${file}: ${error?.message || error}`)
83
+ }
41
84
  // Move a session file into <room>/.archive/ instead of unlinking it. Archived files
42
85
  // are OUTSIDE the readdirSync(dir) path loadAll/listRecs read, so a bridge restart
43
86
  // never resurrects them — but they survive on disk for manual recovery. This is the
@@ -50,7 +93,11 @@ function archive(room, file) {
50
93
  const base = path.basename(file)
51
94
  const dst = path.join(archiveDir(room), `${base}.${Date.now()}`)
52
95
  fs.renameSync(file, dst)
53
- } catch { /* noop */ }
96
+ return true
97
+ } catch (error) {
98
+ persistenceError('archive session snapshot', file, error)
99
+ return false
100
+ }
54
101
  }
55
102
  function listRecs(room) {
56
103
  try {
@@ -65,7 +112,12 @@ function listRecs(room) {
65
112
  // truncated log's first event ts drifts far from real creation. Load-time only.
66
113
  try { const s = fs.statSync(p); r._bt = s.birthtimeMs || s.mtimeMs || 0 } catch { r._bt = 0 }
67
114
  return r
68
- } catch { return null }
115
+ } catch (error) {
116
+ // Keep the damaged file in place for recovery; never silently make a
117
+ // restart look like a deliberately closed lane.
118
+ persistenceError('parse session snapshot', p, error)
119
+ return null
120
+ }
69
121
  })
70
122
  .filter(Boolean)
71
123
  } catch { return [] }
@@ -80,8 +132,8 @@ function prune(room) {
80
132
  // Only UNNAMED sessions are eligible for prune. A named terminal is deliberate
81
133
  // user intent (a project) and must never be silently destroyed by churn.
82
134
  const prunable = files.filter((x) => { const id = namedId(x.p); return !id || !names[id] })
83
- prunable.slice(KEEP).forEach((x) => archive(room, x.p))
84
- } catch { /* noop */ }
135
+ prunable.slice(LIVE_SESSION_MAX).forEach((x) => archive(room, x.p))
136
+ } catch (error) { persistenceError('prune live session snapshots', dir(room), error) }
85
137
  }
86
138
 
87
139
  // ── Durable append-only event log (unbounded history; the <id>.json snapshot is
@@ -210,14 +262,23 @@ export function readDurablePage(room, id, beforeSeq, limit = 200) {
210
262
 
211
263
  const timers = new Map()
212
264
  function write(room, id, data) {
213
- try { ensureDir(room); fs.writeFileSync(path.join(dir(room), `${id}.json`), JSON.stringify({ ...data, id, savedAt: Date.now() })); prune(room) } catch { /* noop */ }
265
+ const file = path.join(dir(room), `${id}.json`)
266
+ try {
267
+ ensureDir(room)
268
+ atomicCommit(file, JSON.stringify({ ...data, id, savedAt: Date.now() }))
269
+ prune(room)
270
+ return true
271
+ } catch (error) {
272
+ persistenceError('write session snapshot', file, error)
273
+ return false
274
+ }
214
275
  }
215
276
  // Debounced — events arrive in bursts; one write per ~1.5s is plenty.
216
277
  export function saveSession(room, id, data) {
217
278
  clearTimeout(timers.get(id))
218
279
  timers.set(id, setTimeout(() => write(room, id, data), 1500))
219
280
  }
220
- export function flushSession(room, id, data) { clearTimeout(timers.get(id)); write(room, id, data) }
281
+ export function flushSession(room, id, data) { clearTimeout(timers.get(id)); return write(room, id, data) }
221
282
  // User-initiated close: cancel any pending debounced save and move the on-disk
222
283
  // record to .archive/ so a later loadAll() (bridge restart) can't resurrect a
223
284
  // closed terminal — but the transcript stays recoverable. PTY ids never have an
@@ -225,7 +286,8 @@ export function flushSession(room, id, data) { clearTimeout(timers.get(id)); wri
225
286
  export function deleteSession(room, id) {
226
287
  clearTimeout(timers.get(id)); timers.delete(id)
227
288
  const p = path.join(dir(room), `${id}.json`)
228
- try { if (fs.existsSync(p)) archive(room, p) } catch { /* noop */ }
289
+ try { return !fs.existsSync(p) || archive(room, p) }
290
+ catch (error) { persistenceError('archive closed session snapshot', p, error); return false }
229
291
  }
230
292
 
231
293
  // Most-recently-saved structured session for the room (drives attached restore).
@@ -240,7 +302,7 @@ export function canResume(rec) {
240
302
  // EVERY saved session for the room, oldest→newest (so tabs reappear in order).
241
303
  // Drives resume-ALL on a bridge restart — without this only the latest session
242
304
  // came back and every backgrounded terminal died on restart despite its state
243
- // sitting on disk. Bounded by KEEP (prune keeps the newest 8).
305
+ // sitting on disk. Bounded by LIVE_SESSION_MAX, exactly the admission capacity.
244
306
  export function loadAll(room) {
245
307
  return listRecs(room).sort((a, b) => (a.savedAt || 0) - (b.savedAt || 0))
246
308
  }
@@ -254,7 +316,9 @@ export function loadPtyId(room) {
254
316
  try { return fs.readFileSync(ptyFile(room), 'utf8').trim() || null } catch { return null }
255
317
  }
256
318
  export function savePtyId(room, id) {
257
- try { ensureDir(room); fs.writeFileSync(ptyFile(room), String(id)) } catch { /* noop */ }
319
+ const file = ptyFile(room)
320
+ try { ensureDir(room); atomicCommit(file, String(id)); return true }
321
+ catch (error) { persistenceError('write PTY id', file, error); return false }
258
322
  }
259
323
 
260
324
  // Per-room terminal display names (terminal id -> label), set via the web's
@@ -265,8 +329,15 @@ export function savePtyId(room, id) {
265
329
  // (NOT *.json) so it never lands in listRecs/loadAll.
266
330
  const namesFile = (room) => path.join(dir(room), '.names')
267
331
  export function loadNames(room) {
268
- try { return JSON.parse(fs.readFileSync(namesFile(room), 'utf8')) || {} } catch { return {} }
332
+ const file = namesFile(room)
333
+ try { return JSON.parse(fs.readFileSync(file, 'utf8')) || {} }
334
+ catch (error) {
335
+ if (error?.code !== 'ENOENT') persistenceError('parse terminal names', file, error)
336
+ return {}
337
+ }
269
338
  }
270
339
  export function saveNames(room, names) {
271
- try { ensureDir(room); fs.writeFileSync(namesFile(room), JSON.stringify(names || {})) } catch { /* noop */ }
340
+ const file = namesFile(room)
341
+ try { ensureDir(room); atomicCommit(file, JSON.stringify(names || {})); return true }
342
+ catch (error) { persistenceError('write terminal names', file, error); return false }
272
343
  }