thinkpool-pair 0.7.144 → 0.7.146

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bridge.mjs CHANGED
@@ -353,6 +353,18 @@ if (!argv[0] || argv[0].startsWith('-')) { const { runAccount } = await import('
353
353
 
354
354
  const room = (argv[0] || '').toUpperCase().trim()
355
355
  if (!room) { console.error('usage: npx thinkpool-pair <ROOM> [--headless] [--continue|--fresh] [-- <command…>] | npx thinkpool-pair (account mode)'); process.exit(1) }
356
+ // Security (audit 2026-07-02): `room` becomes a filesystem path segment under
357
+ // ~/.thinkpool-pair (session-store) AND is interpolated into launchctl/systemd
358
+ // service labels. sessions.id is unconstrained TEXT, so refuse path-unsafe / shell-
359
+ // active characters at the gate instead of trusting the upstream session id.
360
+ // Mirrors service.mjs's ROOM_CODE_RE, broadened for the dash in code_sessions
361
+ // codes and slightly longer ids. Every room-interpolating path is now guarded.
362
+ // NOTE: pure startup gate — runs before any realtime channel exists; does not
363
+ // alter emit topic/event/presence (BR4 surface untouched).
364
+ if (!/^[A-Z0-9_-]{2,32}$/.test(room)) {
365
+ console.error(`refused: invalid room code ${JSON.stringify(room)} (expected 2–32 alphanumerics, dash, or underscore)`)
366
+ process.exit(1)
367
+ }
356
368
  // ── Supervisor mode (--supervise / --keep-alive): keep the bridge alive across
357
369
  // crashes. We re-exec ourselves without the flag and respawn the child on any
358
370
  // non-clean exit with exponential backoff. Zero-dependency, cross-platform. With
@@ -1602,6 +1614,41 @@ function openStructured({ id, model, resume, log, commands, mode, spawnedBy, rol
1602
1614
  return okText(`Opened idle agent lane ${newRef}${args?.name ? ` ("${args.name}")` : ''}. Hand it work with post_to_terminal, or a person can type into it.`)
1603
1615
  },
1604
1616
  ),
1617
+ // Research lane — run a REAL multi-source search + adversarial verification and
1618
+ // return sourced, verdict-tagged findings. Calls the verified web backend
1619
+ // (/api/research-run) as the room owner; plan-gated + budget-capped server-side.
1620
+ tool(
1621
+ 'research',
1622
+ 'Run a REAL multi-source web research + verification on a factual question and surface sourced, verdict-tagged findings to the room. Use when the people would genuinely benefit from looking something up or settling an external-fact question — pricing, "is X still maintained / deprecated", "is that benchmark real", a debate over current facts. OFFER it first in plain language ("want me to spawn a research lane on that?") and only call it once they agree — it spends (plan-gated: Free 5 / Plus 100 runs per month) and takes ~1 minute. It searches the web, reads sources, and returns each claim marked HELD or REJECTED with citations. Present the findings clearly and let both people weigh the sources.',
1623
+ {
1624
+ question: z.string().min(4).max(400).describe('the question to research, phrased as a clear factual query'),
1625
+ },
1626
+ async (args) => {
1627
+ const okText = (t) => ({ content: [{ type: 'text', text: t }] })
1628
+ const q = String(args?.question || '').trim()
1629
+ if (q.length < 4) return okText('Give a clearer question to research.')
1630
+ if (!codeAuthToken) return okText('Research needs the room signed in — no auth token available on this bridge.')
1631
+ let data
1632
+ try {
1633
+ const res = await fetch(`${WEB_BASE}/api/research-run`, {
1634
+ method: 'POST',
1635
+ headers: { 'Content-Type': 'application/json', Authorization: `Bearer ${codeAuthToken}` },
1636
+ body: JSON.stringify({ code: room, question: q }),
1637
+ })
1638
+ if (res.status === 402) { const b = await res.json().catch(() => ({})); return okText(b.hint || 'You have used your research runs for this month.') }
1639
+ if (!res.ok) return okText(`Research could not run (HTTP ${res.status}). Often a transient model spike — offer to retry in a moment.`)
1640
+ data = await res.json()
1641
+ } catch (e) { return okText(`Research failed to run: ${e?.message || e}`) }
1642
+ const claims = Array.isArray(data?.claims) ? data.claims : []
1643
+ const held = claims.filter((c) => c.verdict === 'held').length
1644
+ const lines = claims.map((c) => `[${String(c.verdict || 'rejected').toUpperCase()}] ${c.text}${(c.sources && c.sources.length) ? `\n sources: ${c.sources.join(', ')}` : ''}`)
1645
+ return okText(
1646
+ `Research complete on "${q}" — ${data.searchCount || 0} searches, ${data.sourceCount || 0} sources, ${held}/${claims.length} claims held${data.capped ? ' (capped)' : ''}.\n\n` +
1647
+ (lines.length ? lines.join('\n\n') : 'No verifiable claims found.') +
1648
+ `\n\nPresent these to the room and let both people weigh the sources — flag which held claims rest on a source they might not trust.`
1649
+ )
1650
+ },
1651
+ ),
1605
1652
  // Tier C+ — CLOSE a lane the caller spawned. Restricted to self-spawned agent
1606
1653
  // lanes (spawnedBy === this id): an agent must never be able to kill a sibling
1607
1654
  // someone else is working in, nor the host/attached terminal. The fan-out
@@ -635,6 +635,7 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
635
635
  'CROSS-SESSION AWARENESS: the Ensemble reaches across your SESSIONS, not just the terminals in this room. list_sessions() lists your OTHER ThinkPool Code rooms — both your own rooms running on this machine AND your partner\'s rooms in the same pair, reachable over the per-pair bus (a room on the partner\'s machine shows its host). read_session(session, terminal?) reads recent activity inside one (omit `terminal` to list that room\'s terminals, or pass a ref/name to read that lane). Both are READ-ONLY — they never change another session, and they reach ONLY your own rooms and rooms you share with your partner, never a stranger\'s. Reach for them when work spans rooms — "what\'s the other project up to", "pick up where the other session left off", or to check a long-running task elsewhere before you act here.',
636
636
  'CROSS-SESSION HAND-OFF: post_to_session(session, text, terminal?) sends a task or message to an agent in ANOTHER of your rooms — your own, or your partner\'s over the pair bus. Use it sparingly and only when the people clearly want the rooms to coordinate — e.g. hand the API room\'s agent a concrete follow-up once the frontend is ready. It is dual-consent: a person in YOUR room approves sending, and a person in the TARGET room approves receiving, before anything is delivered — so never rely on it for chit-chat or loops, and an agent that was itself reached via a cross-room post cannot post onward to a third room. It spends real model tokens in the other room (maybe on the other person\'s machine), so prefer read_session to understand a room before you ever post into it, and only post one concrete hand-off at a time. All of this works only under the ThinkPool account bridge; a standalone room sees just its own terminals.',
637
637
  'SUBAGENT POLICY: in this room, delegated work goes through the Ensemble. When you fan out research, parallel sub-tasks, reviews, or any substantive multi-step delegated work, open spawn_terminal lanes — they are visible in the room, the people can watch and steer them, and every other lane can peer at their work. Do NOT reach for the built-in Task/Agent subagents for that work: an in-process subagent is invisible to the room, cannot be peered at or steered, and its work is lost to the Ensemble. The only exception is a trivial, seconds-scale read-only lookup where a room lane would be pure overhead.',
638
+ 'RESEARCH LANE: you have a `research` tool that runs a REAL multi-source web search + adversarial verification and returns each claim marked HELD or REJECTED with citations. Reach for it when the people would genuinely benefit from looking something external up or settling a question of current fact — pricing, "is X still maintained / deprecated", "is that benchmark real", a debate over facts you are not sure of. Do NOT run it unprompted or for things you already know: first OFFER in plain language ("want me to spawn a research lane on that and check it?"), and only call `research(question)` once they agree — it spends real budget (plan-gated Free 5 / Plus 100 runs a month) and takes ~a minute. When it returns, present the held/rejected findings clearly and invite both people to weigh the sources, flagging any held claim that rests on a source they might not trust — that shared scrutiny is the point.',
638
639
  'PEER FIRST: other lanes may be working in the same repo as you, right now. Before starting substantive work — and before any code edit that could overlap another lane — check what the room is doing: the ROOM NOW snapshot appended to your latest turn, or read_terminal for detail; list_sessions/read_session when the question spans your other rooms. If a sibling is touching the same files or branch, coordinate (read its lane, or raise it in chat) instead of colliding.',
639
640
  'WORKTREES: parallel lanes share one machine and usually one repo. Run `git worktree list` before your first code edit; if linked worktrees exist, the shared main checkout is contended (and may be guard-blocked) — do your work in your OWN worktree on your OWN branch (`git worktree add <dir> -b <branch>`), and never edit a checkout or ride a branch another lane is using.',
640
641
  'REMOTE USER: the people driving this room may be on a phone, with NO terminal and no access to this host. Never ask them to run a local command, open a local file, or "go check" something on the machine — anything that must run on the host, you run yourself and show the output in the room. Only suggest actions they can actually do from the room UI or a browser.',
@@ -1006,7 +1007,18 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
1006
1007
  return {
1007
1008
  // A lazy (restored-idle) session cold-boots the query on its first turn. input.push
1008
1009
  // is queue-backed, so the pushed turn buffers and runs once the query is ready.
1009
- sendTurn(text) { if (!closed) { if (!started) runQuery(); turnActive = true; sawSuggestion = false; lastTurnText = String(text); if (sugTimer) { clearTimeout(sugTimer); sugTimer = null } lastEvtTs = Date.now(); stalledSent = false; input.push([{ type: 'text', text: String(text) }, { type: 'text', text: roomReminder() }]) } },
1010
+ sendTurn(text) { if (!closed) { if (!started) runQuery(); turnActive = true; sawSuggestion = false; lastTurnText = String(text); if (sugTimer) { clearTimeout(sugTimer); sugTimer = null } lastEvtTs = Date.now(); stalledSent = false;
1011
+ const t = String(text)
1012
+ // SDK 0.3.198 only recognizes a slash command (/compact, /clear, custom /cmds) when the
1013
+ // user message is a SINGLE text block. Appending the per-turn ROOM NOW <system-reminder>
1014
+ // as a 2nd block made 0.3.198 treat /compact as a plain turn → compaction silently
1015
+ // no-op'd (Max, 2026-07-02; 0.3.185 tolerated the extra block, 0.3.198 tightened it —
1016
+ // verified against the SDK: single-block → the compact command RUNS, two-block → not
1017
+ // recognized). So a slash command goes CLEAN; conversational turns keep the reminder.
1018
+ // Normal turns never start with "/" (composeAgentStdin prepends the preamble), and the
1019
+ // web already routes "/"-prefixed input as a command (pane.jsx), so this matches intent.
1020
+ input.push(/^\s*\//.test(t) ? [{ type: 'text', text: t }] : [{ type: 'text', text: t }, { type: 'text', text: roomReminder() }])
1021
+ } },
1010
1022
  // Cold-boot the query WITHOUT sending a turn — the background warmer calls this on
1011
1023
  // lazily-restored idle terminals so they're ready before the user clicks them.
1012
1024
  warm() { if (!started && !closed) runQuery() },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "thinkpool-pair",
3
- "version": "0.7.144",
3
+ "version": "0.7.146",
4
4
  "description": "Share a local coding-agent CLI (Claude Code, Codex, Gemini, Aider, …) into a ThinkPool Code room, live.",
5
5
  "type": "module",
6
6
  "bin": {
package/session-store.mjs CHANGED
@@ -17,7 +17,20 @@ import path from 'node:path'
17
17
  // can point each case at a fresh tempdir after importing this module. Production leaves
18
18
  // it unset → the real home-dir store.
19
19
  const ROOT = () => process.env.TP_PAIR_ROOT || path.join(os.homedir(), '.thinkpool-pair')
20
- const dir = (room) => path.join(ROOT(), room || 'default')
20
+ // Security (audit 2026-07-02): `room` is an external value (a session/code id)
21
+ // used as a filesystem path segment. sessions.id is unconstrained TEXT in the
22
+ // DB and client-generated, so a crafted id like '../../foo' would otherwise let
23
+ // a hostile room write session files OUTSIDE ~/.thinkpool-pair. Contain EVERY
24
+ // derived path under ROOT: resolve and refuse if the result escapes. Pair of
25
+ // the bridge.mjs startup format check — belt + suspenders.
26
+ const dir = (room) => {
27
+ const root = path.resolve(ROOT())
28
+ const target = path.resolve(root, room || 'default')
29
+ if (target !== root && !target.startsWith(root + path.sep)) {
30
+ throw new Error(`refused: room path escapes sandbox: ${JSON.stringify(room)}`)
31
+ }
32
+ return target
33
+ }
21
34
  const archiveDir = (room) => path.join(dir(room), '.archive')
22
35
  // Only resume the live SDK context for recent sessions — an expired session id
23
36
  // fails ("No conversation found"); past this window we restore transcript only.