thinkpool-pair 0.7.53 → 0.7.55

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/claude-session.mjs +57 -2
  2. package/package.json +1 -1
@@ -177,6 +177,16 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
177
177
  // the catch self-heals.
178
178
  const FORCE_STOP_MS = Math.max(STALL_MS * 3, parseInt(process.env.TP_FORCE_STOP_MS, 10) || 300000)
179
179
  let forceStopped = false
180
+ // ── Haiku suggestion fallback ──
181
+ // Claude Code's own `prompt_suggestion` rarely fires in /code — the binary suppresses it
182
+ // per-turn (cache_cold when a turn's tokens > 10k, rate_limit when throttled, first turn).
183
+ // When it doesn't arrive shortly after `result`, we generate one ourselves with a cheap
184
+ // one-shot Haiku call on the SAME subscription auth (no API key), fed ONLY the last
185
+ // assistant reply (suggestions are short: "proceed" / "go with A") — fast + ~free.
186
+ let sawSuggestion = false // did Claude's own prompt_suggestion fire this turn?
187
+ let lastAssistantText = '' // most recent assistant prose — the only context the fallback needs
188
+ let sugTimer = null // pending fallback timer
189
+ const SUGGEST_FALLBACK_MS = Math.max(1200, parseInt(process.env.TP_SUGGEST_FALLBACK_MS, 10) || 2500)
180
190
 
181
191
  const emit = (evt) => { lastEvtTs = Date.now(); if (evt && evt.kind !== 'stalled') stalledSent = false; try { onEvent?.(evt) } catch { /* never let a consumer throw into the loop */ } }
182
192
  const stallTimer = setInterval(() => {
@@ -374,6 +384,44 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
374
384
  if (healed.blocks) process.stderr.write(`\n ◆ healed ${healed.blocks} cross-provider tool block(s) in the transcript before resume.\n`)
375
385
  }
376
386
 
387
+ // One-shot Haiku fallback suggestion. Runs on the SAME subscription auth as the room
388
+ // (opts.env carries the OAuth — no API key needed), but as a BARE model call: no
389
+ // settingSources, no MCP, no tools — so it's a fast cold call, not a full session boot.
390
+ // Fed only `lastAssistantText` (trimmed). Never blocks a turn; any failure is silent.
391
+ const haikuSuggest = async () => {
392
+ const seed = (lastAssistantText || '').trim().slice(-1500)
393
+ if (!seed || closed) return
394
+ const ac2 = new AbortController()
395
+ const t = setTimeout(() => { try { ac2.abort() } catch { /* noop */ } }, 8000)
396
+ try {
397
+ const hq = query({
398
+ prompt: `You are predicting the user's NEXT chat message in a live coding session, to prefill their composer. Given the assistant's latest reply below, output the single most likely next user message — short and natural (often just "proceed", "go with option A", "yes do that", or a brief follow-up). One line, <=12 words, imperative, no preamble, no quotes, no markdown.\n\nAssistant's latest reply:\n"""\n${seed}\n"""\n\nNext user message:`,
399
+ options: {
400
+ model: 'claude-haiku-4-5',
401
+ ...(cwd ? { cwd } : {}),
402
+ env: opts.env,
403
+ maxTurns: 1,
404
+ permissionMode: 'bypassPermissions',
405
+ settingSources: [], // no CLAUDE.md / commands / agents — raw call
406
+ strictMcpConfig: true,
407
+ mcpServers: {}, // no MCP servers — fast cold start
408
+ abortController: ac2,
409
+ },
410
+ })
411
+ let out = ''
412
+ for await (const mm of hq) {
413
+ if (mm.type === 'assistant') for (const b of (mm.message?.content || [])) if (b.type === 'text') out += b.text
414
+ if (mm.type === 'result') break
415
+ }
416
+ out = out.trim().split('\n')[0].replace(/^["'`]+|["'`]+$/g, '').trim().slice(0, 140)
417
+ if (out && !sawSuggestion && !closed && !/^(sure|of course|certainly|let me know|i can help|happy to)\b/i.test(out)) {
418
+ emit({ kind: 'suggestion', text: out, source: 'haiku' })
419
+ process.stderr.write(` ◆ [suggestion] haiku fallback: ${JSON.stringify(out)}\n`)
420
+ }
421
+ } catch { /* fallback failed (rate limit / abort / model error) — silent */ }
422
+ finally { clearTimeout(t) }
423
+ }
424
+
377
425
  const runQuery = async () => {
378
426
  try {
379
427
  q = query({ prompt: input.stream, options: { ...opts, ...(sessionId ? { resume: sessionId } : {}) } })
@@ -421,6 +469,8 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
421
469
  // parentToolUseId: non-null when this assistant message comes from a
422
470
  // sub-agent (Task tool) — the universal nesting spine. Thread it so the
423
471
  // room can group sub-agent activity under its parent Task card.
472
+ // Capture the latest assistant prose — the only context the Haiku fallback needs.
473
+ { const txt = (m.message?.content || []).filter((b) => b?.type === 'text').map((b) => b.text).join('\n').trim(); if (txt) lastAssistantText = txt }
424
474
  emit({ kind: 'assistant', blocks: simplifyBlocks(m.message?.content), parentToolUseId: m.parent_tool_use_id || null })
425
475
  break
426
476
  case 'user':
@@ -456,6 +506,10 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
456
506
  restartCount = 0 // a successful turn refills the auto-restart budget
457
507
  forceStopped = false
458
508
  emit({ kind: 'result', subtype: m.subtype, sessionId, costUsd: m.total_cost_usd, usage: m.usage, numTurns: m.num_turns, durationMs: m.duration_ms ?? null, denials: Array.isArray(m.permission_denials) ? m.permission_denials.length : 0 })
509
+ // Arm the Haiku fallback: if Claude's own prompt_suggestion doesn't arrive within
510
+ // the window, generate one from the last assistant reply. (Skip aborted turns.)
511
+ if (sugTimer) clearTimeout(sugTimer)
512
+ if (m.subtype !== 'aborted') sugTimer = setTimeout(() => { sugTimer = null; if (!sawSuggestion && !closed) haikuSuggest() }, SUGGEST_FALLBACK_MS)
459
513
  // Surface a usage/context meter (chrome, not a transcript line). The
460
514
  // context window % comes from the control request; cost is cumulative.
461
515
  ;(async () => {
@@ -472,7 +526,8 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
472
526
  // Claude Code's own predicted next prompt (promptSuggestions opt-in).
473
527
  // Arrives once per turn AFTER `result`; surfaced verbatim as ephemeral
474
528
  // chrome (a ghost-text chip in the room composer), never persisted.
475
- if (m.suggestion) emit({ kind: 'suggestion', text: m.suggestion })
529
+ if (sugTimer) { clearTimeout(sugTimer); sugTimer = null } // built-in won — cancel the Haiku fallback
530
+ if (m.suggestion) { sawSuggestion = true; emit({ kind: 'suggestion', text: m.suggestion }) }
476
531
  break
477
532
  default:
478
533
  break
@@ -504,7 +559,7 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
504
559
  runQuery()
505
560
 
506
561
  return {
507
- sendTurn(text) { if (!closed) { turnActive = true; lastEvtTs = Date.now(); stalledSent = false; input.push(String(text)) } },
562
+ sendTurn(text) { if (!closed) { turnActive = true; sawSuggestion = false; if (sugTimer) { clearTimeout(sugTimer); sugTimer = null } lastEvtTs = Date.now(); stalledSent = false; input.push(String(text)) } },
508
563
  // Set the permission mode — Claude Code's ⇧⇥ cycle. setPermissionMode is a
509
564
  // streaming control request (drives plan-mode behaviour SDK-side); the local
510
565
  // `mode` drives our PreToolUse auto-approve policy. Echo so the room syncs.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "thinkpool-pair",
3
- "version": "0.7.53",
3
+ "version": "0.7.55",
4
4
  "description": "Share a local coding-agent CLI (Claude Code, Codex, Gemini, Aider, …) into a ThinkPool Code room, live.",
5
5
  "type": "module",
6
6
  "bin": {