thinkpool-pair 0.7.41 → 0.7.43

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bridge.mjs CHANGED
@@ -151,7 +151,9 @@ if (argv[0] === 'install-service' || argv[0] === 'uninstall-service') {
151
151
  // shell export can't reach a serviced bridge — provider.json is its own source
152
152
  // of truth. Idempotent; a no-op for the default Anthropic provider. See provider.mjs.
153
153
  const { applyProviderEnv } = await import('./provider.mjs')
154
- applyProviderEnv()
154
+ // Captured so the room announce can report the active LLM provider (anthropic,
155
+ // or a custom Anthropic-compatible endpoint) to the web welcome banner.
156
+ const PROVIDER_INFO = applyProviderEnv()
155
157
 
156
158
  if (argv[0] === 'login') { const { runLogin } = await import('./account.mjs'); await runLogin(SUPABASE_URL, SUPABASE_ANON, WEB_BASE) }
157
159
  if (argv[0] === 'bind') { const { runBind } = await import('./account.mjs'); runBind(argv[1], argv[2]) }
@@ -415,6 +417,14 @@ const announce = () =>
415
417
  // cwd + version: the host's working dir + thinkpool-pair version, shown in
416
418
  // the room's welcome banner. Re-sent per announce so late joiners get them.
417
419
  cwd, version: VERSION,
420
+ // provider: the LLM endpoint this bridge drives Claude Code through —
421
+ // 'anthropic' (the default, the host's regular Claude login) or 'custom'
422
+ // (any Anthropic-compatible base url: GLM/Z.ai, OpenRouter, a proxy). Shown
423
+ // in the room banner so both drivers see whose model they're on.
424
+ // providerBaseUrl rides along only for custom so the banner can show the
425
+ // host (e.g. openrouter.ai). Older web clients ignore unknown fields.
426
+ provider: PROVIDER_INFO.provider,
427
+ providerBaseUrl: PROVIDER_INFO.baseUrl || undefined,
418
428
  // pendingUpdate: a newer thinkpool-pair is out — drives the room's "update ready"
419
429
  // chip; re-sent per announce so reloads/late-joiners see it (Slice 3).
420
430
  pendingUpdate,
@@ -451,7 +461,7 @@ const flushAll = () => {
451
461
  bcast('pty-out', { term: id, b64 })
452
462
  }
453
463
  }
454
- const flushTimer = setInterval(flushAll, 35)
464
+ const flushTimer = setInterval(flushAll, 35); flushTimer.unref?.() // don't keep the event loop alive past shutdown
455
465
 
456
466
  // ── mockup hand-off ────────────────────────────────────────────────────
457
467
  // render.sh (mockup-iterate) drops a manifest JSON into TP_MOCKUP_OUTBOX after
@@ -548,6 +558,11 @@ const attachedDims = () => ({
548
558
 
549
559
  function openTerm({ id, cmd, args = [], attached = false, cols, rows }) {
550
560
  if (terms.has(id)) return
561
+ if (terms.size >= 8) { // cap live PTYs (matches session-store KEEP=8) — no unbounded host leak
562
+ process.stderr.write(`\n ⚠ can't open "${cmd}" — terminal cap (8) reached; close one first.\n`)
563
+ bcast('term-exit', { id })
564
+ return
565
+ }
551
566
  if (!pty) { // raw-PTY path needs node-pty; structured Claude never lands here
552
567
  process.stderr.write(`\n ⚠ can't open "${cmd}" — node-pty isn't built (PTY mode disabled).\n`)
553
568
  bcast('term-exit', { id })
@@ -697,7 +712,9 @@ function openStructured({ id, model, resume, log, commands, mode }) {
697
712
  // 'stalled' is transient turn-liveness state (would replay as a stale frozen
698
713
  // banner if persisted); 'compaction' is a real transcript milestone (the recap
699
714
  // card) and is intentionally NOT chrome, so it persists + replays.
700
- const chrome = evt.kind === 'mode' || evt.kind === 'usage' || evt.kind === 'clear' || evt.kind === 'compact' || evt.kind === 'stalled'
715
+ // 'suggestion' is Claude Code's predicted-next-prompt ghost text — ephemeral
716
+ // composer UI, broadcast to both clients but never logged/persisted/replayed.
717
+ const chrome = evt.kind === 'mode' || evt.kind === 'usage' || evt.kind === 'clear' || evt.kind === 'compact' || evt.kind === 'stalled' || evt.kind === 'suggestion'
701
718
  // compact turn finished (or errored) → clear the pulsing indicator.
702
719
  // No "done" ctl line — the SDK's own output in the transcript is the
703
720
  // signal (compact summary on success, AbortError text on abort/too-small).
@@ -782,7 +799,7 @@ if (attachedCmd && process.stdin.isTTY) process.stdin.setRawMode(true)
782
799
  process.stdin.resume()
783
800
  process.stdin.on('data', d => {
784
801
  const t = attachedId && terms.get(attachedId)
785
- if (t) t.term.write(d.toString('utf8'))
802
+ if (t) { try { t.term.write(d.toString('utf8')) } catch { /* term just exited */ } }
786
803
  else if (d.includes(3)) shutdown() // Ctrl-C once detached/headless
787
804
  })
788
805
  // attached terminal follows the host's TTY size — but never below the
@@ -805,7 +822,7 @@ channel
805
822
  if (!payload?.data) return
806
823
  markActivity()
807
824
  const t = terms.get(payload.term) || (payload.term == null && attachedId ? terms.get(attachedId) : null)
808
- if (t) t.term.write(payload.data)
825
+ if (t) { try { t.term.write(payload.data) } catch { /* term just exited */ } }
809
826
  })
810
827
  .on('broadcast', { event: 'resize' }, ({ payload }) => {
811
828
  // headless terms are web-sized (last writer wins); the attached term is
@@ -84,8 +84,10 @@ function makeInputStream() {
84
84
  return {
85
85
  stream: gen(),
86
86
  push(content) {
87
+ if (ended) return false // session ended — caller (sendTurn) can surface this
87
88
  queue.push({ type: 'user', message: { role: 'user', content } })
88
89
  if (wake) { wake(); wake = null }
90
+ return true
89
91
  },
90
92
  end() { ended = true; if (wake) { wake(); wake = null } },
91
93
  }
@@ -120,9 +122,17 @@ const MODES = new Set(['default', 'acceptEdits', 'plan', 'bypassPermissions'])
120
122
 
121
123
  export function startClaudeSession({ cwd, model, resume, env, mode: initialMode = 'default', onEvent, requestPermission }) {
122
124
  const ac = new AbortController()
123
- const input = makeInputStream()
125
+ let input = makeInputStream() // `let`: auto-restart swaps in a fresh stream
124
126
  let sessionId = resume || null
125
127
  let closed = false
128
+ // Cold-start measurement (2026-06-21, spec 2026-06-21-code-session-deploy-stability):
129
+ // stamp when the open begins so we can log spawn→ready (the SDK's first `init`
130
+ // system message — when the terminal becomes usable). New room terminals "take a
131
+ // long time"; settingSources:['user','project','local'] boots every host MCP
132
+ // server + SessionStart hook + plugin before init. This is the A-side baseline;
133
+ // open a terminal with TP_MCP_STRICT=1 (below) for the B-side to isolate MCP's share.
134
+ const spawnT0 = Date.now()
135
+ let readyLogged = false
126
136
  let q = null // the live Query — control requests (interrupt /
127
137
  // setPermissionMode) route through it once streaming.
128
138
  // The session is CREATED in the caller's chosen mode (not hard-coded default):
@@ -150,12 +160,39 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
150
160
  let turnActive = false // true between a sent turn and its `result`
151
161
  let lastEvtTs = Date.now() // wall-clock of the most recent emitted event
152
162
  let stalledSent = false // one `stalled` per stall, not a storm
163
+ // Auto-restart (item 1) — a transient upstream stream drop (e.g. Anthropic
164
+ // "Connection closed mid-response") used to leave the session half-dead:
165
+ // query loop gone, turnActive stuck, later sendTurns pushed into a dead
166
+ // stream (silent freeze). The loop now self-heals: a recoverable throw
167
+ // re-runs query() with resume, bounded by RESTART_MAX consecutive attempts
168
+ // (reset to 0 on every successful `result`). Exhaustion/non-recoverable →
169
+ // closed, so sendTurn stops silently queuing.
170
+ const RESTART_MAX = 3
171
+ const RECOVERABLE = /connection closed|connection reset|econnreset|etimedout|socket hang up|fetch failed|network error|socket destroyed/i
172
+ let restartCount = 0
173
+ // Force-stop a true wedge (item 3): no result, no error, just silence past
174
+ // FORCE_STOP_MS. Clears turnActive so between-turns updates unblock, and
175
+ // surfaces a recoverable error. The wedged loop is left in place (a hung
176
+ // native call can't be safely restarted mid-flight); if it later throws,
177
+ // the catch self-heals.
178
+ const FORCE_STOP_MS = Math.max(STALL_MS * 3, parseInt(process.env.TP_FORCE_STOP_MS, 10) || 300000)
179
+ let forceStopped = false
153
180
 
154
181
  const emit = (evt) => { lastEvtTs = Date.now(); if (evt && evt.kind !== 'stalled') stalledSent = false; try { onEvent?.(evt) } catch { /* never let a consumer throw into the loop */ } }
155
182
  const stallTimer = setInterval(() => {
156
- if (turnActive && !stalledSent && Date.now() - lastEvtTs > STALL_MS) {
183
+ if (!turnActive) return
184
+ const quiet = Date.now() - lastEvtTs
185
+ if (quiet > FORCE_STOP_MS && !forceStopped) {
186
+ // True wedge — no result, no error, just silence. Unblocks between-turns
187
+ // updates + surfaces a recoverable error. The wedged loop is left alone.
188
+ forceStopped = true
189
+ turnActive = false
190
+ emit({ kind: 'error', message: `agent went silent for ${Math.round(quiet / 1000)}s — force-stopped the turn; send again to resume`, recoverable: true })
191
+ return
192
+ }
193
+ if (!stalledSent && quiet > STALL_MS) {
157
194
  stalledSent = true
158
- emit({ kind: 'stalled', sinceMs: Date.now() - lastEvtTs })
195
+ emit({ kind: 'stalled', sinceMs: quiet })
159
196
  }
160
197
  }, 5000)
161
198
  stallTimer.unref?.()
@@ -301,7 +338,22 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
301
338
  // flow during a turn. We ignore the fine-grained stream_event partials in
302
339
  // the loop; only the coarse thinking_tokens system message is surfaced.
303
340
  includePartialMessages: true,
341
+ // Claude Code's own "predicted next prompt" — the SDK emits at most one
342
+ // `prompt_suggestion` per turn, AFTER the `result` message, suppressed on the
343
+ // first turn / after API errors / in plan mode, and it piggybacks the parent's
344
+ // prompt cache (nearly free). We surface it verbatim to the room composer as a
345
+ // ghost-text chip; nothing is generated on our side. The loop keeps iterating
346
+ // past `result`, so the post-result suggestion is received. Global off switch:
347
+ // CLAUDE_CODE_ENABLE_PROMPT_SUGGESTION=false.
348
+ promptSuggestions: true,
304
349
  }
350
+ // Measurement lever (default OFF — no behavior change). TP_MCP_STRICT=1 makes the
351
+ // room session skip booting the host's filesystem-configured MCP servers
352
+ // (strictMcpConfig + empty mcpServers map), isolating the MCP share of cold-start
353
+ // in the spawn→ready log below. settingSources still loads slash commands /
354
+ // CLAUDE.md / agents either way — strict only governs MCP server sourcing. Flip
355
+ // this to the default (or curate a slim allowlist) once the numbers justify it.
356
+ if (process.env.TP_MCP_STRICT === '1') { opts.strictMcpConfig = true; opts.mcpServers = {} }
305
357
  if (cwd) opts.cwd = cwd
306
358
  if (model) opts.model = model
307
359
  if (resume) opts.resume = resume
@@ -317,9 +369,9 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
317
369
  if (healed.blocks) process.stderr.write(`\n ◆ healed ${healed.blocks} cross-provider tool block(s) in the transcript before resume.\n`)
318
370
  }
319
371
 
320
- ;(async () => {
372
+ const runQuery = async () => {
321
373
  try {
322
- q = query({ prompt: input.stream, options: opts })
374
+ q = query({ prompt: input.stream, options: { ...opts, ...(sessionId ? { resume: sessionId } : {}) } })
323
375
  for await (const m of q) {
324
376
  if (closed) break
325
377
  switch (m.type) {
@@ -343,6 +395,14 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
343
395
  break
344
396
  }
345
397
  if (m.session_id) sessionId = m.session_id
398
+ // Cold-start timing — the init system message is when the terminal is
399
+ // usable, so spawn→here IS the latency the user waits on. One line per
400
+ // open. Compare a normal open vs one with TP_MCP_STRICT=1 to read off how
401
+ // many ms the host MCP servers cost. (resume opens also pay sanitizeSession.)
402
+ if (!readyLogged) {
403
+ readyLogged = true
404
+ process.stderr.write(`\n ◆ session ready in ${Date.now() - spawnT0}ms — MCP ${process.env.TP_MCP_STRICT === '1' ? 'OFF' : 'on'}${resume ? ', resume' : ', fresh'}.\n`)
405
+ }
346
406
  // m.slash_commands (init message) — the commands this session really
347
407
  // supports: built-ins + the host's custom .claude/commands. Surfaced
348
408
  // so the room composer's autocomplete lists what ACTUALLY exists.
@@ -388,6 +448,8 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
388
448
  if (m.session_id) sessionId = m.session_id
389
449
  turnBaseOut = 0; curMsgOut = 0 // reset the live token count for the next turn
390
450
  turnActive = false // turn settled → stall watchdog stands down
451
+ restartCount = 0 // a successful turn refills the auto-restart budget
452
+ forceStopped = false
391
453
  emit({ kind: 'result', subtype: m.subtype, sessionId, costUsd: m.total_cost_usd, usage: m.usage, numTurns: m.num_turns, durationMs: m.duration_ms ?? null, denials: Array.isArray(m.permission_denials) ? m.permission_denials.length : 0 })
392
454
  // Surface a usage/context meter (chrome, not a transcript line). The
393
455
  // context window % comes from the control request; cost is cumulative.
@@ -401,14 +463,40 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
401
463
  emit({ kind: 'usage', costUsd: m.total_cost_usd ?? null, tokens: (u.input_tokens || 0) + (u.output_tokens || 0), ctx })
402
464
  })()
403
465
  break
466
+ case 'prompt_suggestion':
467
+ // Claude Code's own predicted next prompt (promptSuggestions opt-in).
468
+ // Arrives once per turn AFTER `result`; surfaced verbatim as ephemeral
469
+ // chrome (a ghost-text chip in the room composer), never persisted.
470
+ if (m.suggestion) emit({ kind: 'suggestion', text: m.suggestion })
471
+ break
404
472
  default:
405
473
  break
406
474
  }
407
475
  }
408
476
  } catch (e) {
409
- if (!closed) emit({ kind: 'error', message: e?.message || String(e) })
477
+ if (closed) return
478
+ const msg = e?.message || String(e)
479
+ emit({ kind: 'error', message: msg })
480
+ // Auto-restart a transient upstream stream drop: re-run query() with
481
+ // resume so the conversation continues. Bounded by RESTART_MAX (reset on
482
+ // each successful result) so a persistent outage can't loop forever; a
483
+ // fresh input stream is needed because the throw killed the old iterator.
484
+ if (RECOVERABLE.test(msg) && restartCount < RESTART_MAX) {
485
+ restartCount++
486
+ turnActive = false
487
+ emit({ kind: 'note', text: `connection dropped — reconnecting (${restartCount}/${RESTART_MAX})` })
488
+ input = makeInputStream()
489
+ setTimeout(runQuery, Math.min(1000 * restartCount, 4000)) // 1s, 2s, 4s backoff
490
+ return
491
+ }
492
+ // Non-recoverable, or budget exhausted: mark done so sendTurn stops
493
+ // silently queuing into a dead stream. The bridge's onEvent self-heal
494
+ // still fires for "No conversation found" (it reopens fresh).
495
+ closed = true
496
+ turnActive = false
410
497
  }
411
- })()
498
+ }
499
+ runQuery()
412
500
 
413
501
  return {
414
502
  sendTurn(text) { if (!closed) { turnActive = true; lastEvtTs = Date.now(); stalledSent = false; input.push(String(text)) } },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "thinkpool-pair",
3
- "version": "0.7.41",
3
+ "version": "0.7.43",
4
4
  "description": "Share a local coding-agent CLI (Claude Code, Codex, Gemini, Aider, …) into a ThinkPool Code room, live.",
5
5
  "type": "module",
6
6
  "bin": {