thinkpool-pair 0.7.155 → 0.7.157

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -153,5 +153,22 @@ Provider config is stored in `~/.thinkpool-pair/provider.json` (mode 0600) and
153
153
  applies to every bridge on the machine. Restart the bridge (or its launchd
154
154
  service) after changing it — the choice is read once at startup.
155
155
 
156
+ ### Context meter for non-Claude models
157
+
158
+ The context-usage meter (`ctx N%` in the room) is computed by the Agent SDK from
159
+ the model id it thinks it's talking to — for a non-Claude model bridged in over
160
+ an Anthropic-compatible endpoint, that assumed window is wrong, so the meter can
161
+ run past 100%. The bridge corrects it for common models automatically (GLM-4.5/4.6,
162
+ DeepSeek V3.x, Kimi K2, Qwen3-Coder — matched by model id).
163
+
164
+ For any model not in that list, set the real window explicitly per bridge:
165
+
166
+ ```bash
167
+ export TP_CONTEXT_MAX=131072 # your model's context window, in tokens
168
+ ```
169
+
170
+ `TP_CONTEXT_MAX` overrides the built-in map (and applies even to Claude ids if
171
+ you set it). Unset it to fall back to the automatic correction.
172
+
156
173
  Public anon creds are embedded (the same ones the web app ships). Override with
157
174
  `TP_SUPABASE_URL` / `TP_SUPABASE_ANON` if needed. Set `TP_NAME` to label yourself.
package/bridge.mjs CHANGED
@@ -39,7 +39,7 @@ import { randomUUID } from 'node:crypto'
39
39
  import { createClient } from '@supabase/supabase-js'
40
40
  import { createSdkMcpServer, tool } from '@anthropic-ai/claude-agent-sdk'
41
41
  import { z } from 'zod'
42
- import { startClaudeSession, oneShotSummary } from './claude-session.mjs'
42
+ import { startClaudeSession } from './claude-session.mjs'
43
43
  import { FLOW_CONDUCTOR_PROMPT, FLOW_LANE_PROMPT, buildConductorEnv, assembleCrossWaveContext, buildLanePrompt } from './flow-conductor.mjs'
44
44
  import { normalizePlanOutput, laneModelFor } from './flow-task-graph.mjs' // FL-B1 — validate the conductor's submit_flow_plan task-graph; laneModelFor — per-slice model tier (2026-07-03-flow-lane-model-tiers)
45
45
  // S1 (context-offload) — durable digest store. mark_flow_done digests a closed slice in;
@@ -84,7 +84,7 @@ const flowRedispatch = new Map()
84
84
  const flowBudgets = new Map()
85
85
  import { formatPeek, PEEK, siblingsOf, resolveSibling, crossPostDecision, CROSSPOST, spawnDecision, SPAWN, CROSSROOM, formatPairRoster, crossRoomPostDecision, formatRoomNow } from './cross-terminal.mjs'
86
86
  import { turnInFlight } from './update-gate.mjs'
87
- import { saveSession, flushSession, deleteSession, loadAll, canResume, loadPtyId, savePtyId, loadNames, saveNames, appendDurableEvents, seedDurableEvents, readDurablePage, readDurableOldestSeq, loadSummary, saveSummary } from './session-store.mjs'
87
+ import { saveSession, flushSession, deleteSession, loadAll, canResume, loadPtyId, savePtyId, loadNames, saveNames, appendDurableEvents, seedDurableEvents, readDurablePage, readDurableOldestSeq } from './session-store.mjs'
88
88
  import { stampEvent, makeSeqCounter, maxSeq, seqable, capReplayEvents, chunkReplayEvents, boundEventForBroadcast, inlineImageBlocks, usageReportLine } from './event-id.mjs'
89
89
  import { makeThrottledTrack } from './presence.mjs'
90
90
 
@@ -506,6 +506,13 @@ if (process.stdin.isTTY && !headless && process.env.THINKPOOL_PAIR_AUTOUPDATE !=
506
506
  }
507
507
  }
508
508
  const name = process.env.TP_NAME || os.userInfo().username || 'host'
509
+ // host: this machine's short label — os.hostname() with any DNS/domain suffix
510
+ // stripped and capped ~24 chars. A /code room can be served by different bridges
511
+ // over time (Max's Mac vs Conrad's Linux box); the client uses this to show WHICH
512
+ // machine currently serves the room and attribute dormant terminals to their home
513
+ // box. Distinct from `name` (the driver's username). Announced additively —
514
+ // top-level only (one bridge per announce); older clients ignore the unknown key.
515
+ const host = (os.hostname() || 'host').split('.')[0].slice(0, 24)
509
516
 
510
517
  // Repo awareness — the room shows which project this machine is sharing.
511
518
  // Cheap reads, no subprocess: directory name + .git/HEAD.
@@ -788,6 +795,11 @@ const announce = () =>
788
795
  // cwd + version: the host's working dir + thinkpool-pair version, shown in
789
796
  // the room's welcome banner. Re-sent per announce so late joiners get them.
790
797
  cwd, version: VERSION,
798
+ // host: short machine label (see const `host`) so the room shows which box
799
+ // currently serves it + attributes dormant terminals to their home machine.
800
+ // Additive top-level field; consumed by src/pages/code/room.jsx onAnnounce in
801
+ // a later lane. Older clients ignore it.
802
+ host,
791
803
  // provider: the LLM endpoint this bridge drives Claude Code through —
792
804
  // 'anthropic' (the default, the host's regular Claude login) or 'custom'
793
805
  // (any Anthropic-compatible base url: GLM/Z.ai, OpenRouter, a proxy). Shown
@@ -2465,94 +2477,6 @@ channel
2465
2477
  saveNames(room, termNames)
2466
2478
  announce()
2467
2479
  })
2468
- // Session Info panel (2026-07-02, plain-English recap 2026-07-02) — a cheap one-shot
2469
- // recap on demand. Writes a plain-English "Done so far" / "Working on now" bullet
2470
- // summary (Markdown) of what the whole session accomplished + what's live — NOT a
2471
- // per-terminal technical list (Max's ask: clean, readable, no jargon/terminal names).
2472
- // Runs on the room's own subscription auth via oneShotSummary (cheap haiku →
2473
- // provider-default fallback). Never auto — only on the panel button. The web caches
2474
- // the result in React state (no re-charge on re-open within a page load).
2475
- .on('broadcast', { event: 'code-summarize' }, async ({ payload }) => {
2476
- const by = payload?.by || null
2477
- // Peek — any device opening the panel fetches the cached recap (cross-device +
2478
- // cross-restart persistence) WITHOUT spending a model call. No cache → silent.
2479
- if (payload?.peek) {
2480
- const cached = loadSummary(room)
2481
- if (cached && cached.text) bcast('code-summary', { status: 'done', ...cached })
2482
- return
2483
- }
2484
- bcast('code-summary', { status: 'generating', by })
2485
- try {
2486
- // Pull assistant/user text out of a transcript-event array, keep the tail.
2487
- const logDigest = (events, max = 1200) => {
2488
- const parts = []
2489
- for (const e of (events || [])) {
2490
- if (e?.kind === 'assistant') { for (const b of (e.blocks || [])) if (b?.type === 'text' && b.text) parts.push(b.text) }
2491
- else if (e?.kind === 'you' && e.text) parts.push(`user: ${e.text}`)
2492
- }
2493
- return parts.join('\n').replace(/\n{3,}/g, '\n\n').slice(-max)
2494
- }
2495
- // LIVE structured terminals — named, doing/done, with a recent digest.
2496
- const live = []
2497
- let env = null, cwd = null
2498
- for (const [id, e] of sessions) {
2499
- if (e.kind !== 'structured') continue
2500
- live.push({ name: termNames[id] || 'Terminal', status: e.session?.turnActive ? 'working' : 'done', digest: logDigest(e.log, 1100) })
2501
- if (!env) { env = { ...process.env }; cwd = e.cwd || process.cwd() }
2502
- }
2503
- // CLOSED terminals — named but not live. Aggregate only; cap the read to bound cost.
2504
- const liveIdSet = new Set(sessions.keys())
2505
- const closedIds = Object.keys(termNames).filter(id => !liveIdSet.has(id))
2506
- const closedDigests = []
2507
- for (const id of closedIds.slice(0, 4)) {
2508
- try {
2509
- const page = readDurablePage(room, id, Number.MAX_SAFE_INTEGER, 30)
2510
- const evs = Array.isArray(page) ? page : (Array.isArray(page?.events) ? page.events : [])
2511
- const d = logDigest(evs, 450)
2512
- if (d) closedDigests.push(d)
2513
- } catch { /* no retained detail for this closed lane */ }
2514
- }
2515
- if (!live.length && !closedIds.length) { bcast('code-summary', { status: 'error', error: 'Nothing to summarize yet.', by }); return }
2516
-
2517
- const liveBlock = live.length
2518
- ? live.map(t => `LIVE TERMINAL "${t.name}" [${t.status}]:\n${t.digest || '(no recent activity)'}`).join('\n\n')
2519
- : '(no live terminals)'
2520
- const closedBlock = closedIds.length
2521
- ? `\n\nThere are also ${closedIds.length} CLOSED terminal(s). Recent excerpts (do NOT name them individually):\n${(closedDigests.join('\n---\n') || '(no retained detail)').slice(0, 1400)}`
2522
- : ''
2523
-
2524
- const prompt = `Write a short, friendly recap of this coding session for someone non-technical who just wants to know what's going on. Plain English only — no jargon, no file names, no branch names, no terminal names or IDs, no code, no counts of terminals. Explain what the work MEANS and why it matters, not the mechanics.
2525
-
2526
- Output ONLY this Markdown shape, nothing else:
2527
-
2528
- **Done so far**
2529
- - <one plain-English thing that got finished>
2530
- - <another>
2531
-
2532
- **Working on now**
2533
- - <what's actively being worked on, in plain English>
2534
-
2535
- Rules:
2536
- - Merge ALL the finished work — live and closed terminals alike — into "Done so far". Combine similar items; do NOT organise by terminal or list them. Aim for 3–6 bullets.
2537
- - Put only genuinely in-progress work under "Working on now". If nothing is active, write exactly: "- Nothing in progress right now — everything's wrapped up."
2538
- - Each bullet is one clear sentence, ~16 words max, no trailing period, and no technical term a normal person wouldn't understand.
2539
-
2540
- Here's the raw session activity to summarize:
2541
-
2542
- ${liveBlock}${closedBlock}
2543
-
2544
- Recap:`
2545
-
2546
- const res = await oneShotSummary({ prompt, env: env || process.env, cwd: cwd || process.cwd() })
2547
- if (!res || !res.text) { bcast('code-summary', { status: 'error', error: 'Summary model unavailable on this provider.', by }); return }
2548
- const summary = { text: res.text, model: res.model, generatedAt: Date.now() }
2549
- try { saveSummary(room, summary) } catch { /* cache is best-effort */ }
2550
- bcast('code-summary', { status: 'done', ...summary, by })
2551
- process.stderr.write(`\n ◆ session summary generated (${res.model}) for ${by || 'someone'}.\n`)
2552
- } catch (e) {
2553
- bcast('code-summary', { status: 'error', error: String(e?.message || e).slice(0, 140), by })
2554
- }
2555
- })
2556
2480
  .on('broadcast', { event: 'who' }, announce)
2557
2481
  .subscribe(status => {
2558
2482
  if (status === 'SUBSCRIBED') {
@@ -17,6 +17,7 @@ import { query } from '@anthropic-ai/claude-agent-sdk'
17
17
  import { sanitizeSession } from './transcript-sanitize.mjs'
18
18
  import { reviewGatePreToolDecision } from './flow-review-gate.mjs'
19
19
  import { crossPostNeedsCard } from './cross-terminal.mjs'
20
+ import { correctContext } from './context-windows.mjs'
20
21
 
21
22
  // ── risk classification — the accent/danger tier of the permission card ──
22
23
  // low (read-only) · medium (writes/runs) · network (leaves the machine) ·
@@ -145,48 +146,6 @@ const TP_ROOM_REMINDER = [
145
146
  'Verify before claiming done — show runtime evidence you produced, not "should work, go test it".',
146
147
  ].join(' ')
147
148
 
148
- // ── One-shot summary (2026-07-02, Session Info panel) ────────────────────────
149
- // A bare, cheap, cold model call — same shape as the haikuSuggest fallback: runs
150
- // on the room's own subscription auth (env carries the OAuth — no API key), NO
151
- // settingSources / MCP / tools, maxTurns 1. Used by the bridge's code-summarize
152
- // handler to write a session recap on demand. Model fallback (Max's ask): prefer a
153
- // cheap haiku-class model; if the provider (GLM / OpenRouter / custom base-url) has
154
- // no such model, retry with the provider default. Returns { text, model } or null.
155
- export async function oneShotSummary({ prompt, env, cwd, timeoutMs = 25000 }) {
156
- const attempt = async (model) => {
157
- const ac = new AbortController()
158
- const t = setTimeout(() => { try { ac.abort() } catch { /* noop */ } }, timeoutMs)
159
- try {
160
- const hq = query({
161
- prompt,
162
- options: {
163
- ...(model ? { model } : {}),
164
- ...(cwd ? { cwd } : {}),
165
- env,
166
- maxTurns: 1,
167
- permissionMode: 'bypassPermissions',
168
- settingSources: [],
169
- strictMcpConfig: true,
170
- mcpServers: {},
171
- abortController: ac,
172
- },
173
- })
174
- let out = ''
175
- for await (const mm of hq) {
176
- if (mm.type === 'assistant') for (const b of (mm.message?.content || [])) if (b.type === 'text') out += b.text
177
- if (mm.type === 'result') break
178
- }
179
- return out.trim()
180
- } finally { clearTimeout(t) }
181
- }
182
- // Prefer cheap haiku; fall back to the provider's default model if it's absent.
183
- try { const out = await attempt('claude-haiku-4-5'); if (out) return { text: out, model: 'claude-haiku-4-5' } }
184
- catch { /* provider has no haiku — fall through to default */ }
185
- try { const out = await attempt(undefined); if (out) return { text: out, model: 'default' } }
186
- catch { /* provider default also failed */ }
187
- return null
188
- }
189
-
190
149
  export function startClaudeSession({ cwd, model, resume, env, mode: initialMode = 'default', onEvent, requestPermission, mcpServers, crossPostGate, crossRoomPostGate, didSpawnTarget = null, rolePrompt, blockSubagents = false, onSubmitPlan = null, onLaneDone = null, onReviewVerdict = null, reviewGate = null, lazy = false, roomContext = null }) {
191
150
  // Per-turn reminder + live ROOM NOW tail. roomContext (bridge-supplied) returns the
192
151
  // room's CURRENT state — sibling lanes, active git worktrees — or null. The static
@@ -788,7 +747,7 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
788
747
  ;(async () => {
789
748
  try {
790
749
  const c = await q?.getContextUsage?.()
791
- if (c) { emit({ kind: 'usage', ctx: { used: c.totalTokens, max: c.maxTokens, pct: Math.round(c.percentage), model: c.model } }); if (c.model) curModel = c.model }
750
+ if (c) { emit({ kind: 'usage', ctx: correctContext({ used: c.totalTokens, max: c.maxTokens, pct: Math.round(c.percentage), model: c.model }) }); if (c.model) curModel = c.model }
792
751
  } catch { /* control req may be unavailable */ }
793
752
  })()
794
753
  break
@@ -932,7 +891,7 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
932
891
  let ctx = null
933
892
  try {
934
893
  const c = await q?.getContextUsage?.()
935
- if (c) { ctx = { used: c.totalTokens, max: c.maxTokens, pct: Math.round(c.percentage), model: c.model }; if (c.model) curModel = c.model }
894
+ if (c) { ctx = correctContext({ used: c.totalTokens, max: c.maxTokens, pct: Math.round(c.percentage), model: c.model }); if (c.model) curModel = c.model }
936
895
  } catch { /* control req may be unavailable */ }
937
896
  const u = m.usage || {}
938
897
  emit({ kind: 'usage', costUsd: m.total_cost_usd ?? null, tokens: (u.input_tokens || 0) + (u.output_tokens || 0), ctx })
@@ -0,0 +1,101 @@
1
+ /* ─────────────────────────────────────────────────────────────
2
+ context-windows — correct the /code context meter for non-Claude BYOK models.
3
+
4
+ The Claude Agent SDK computes its context-usage meter
5
+ (getContextUsage → { totalTokens, maxTokens, percentage }) from the model
6
+ STRING it believes it is talking to. When a user bridges a NON-Claude model
7
+ in over an Anthropic-compatible endpoint (ANTHROPIC_BASE_URL — GLM, DeepSeek,
8
+ Kimi, Qwen, … via OpenRouter or a provider's own compat endpoint), the SDK's
9
+ maxTokens is for the WRONG model, so the meter fills at the wrong rate and can
10
+ shoot past 100% (Max, 2026-07-04: GLM-4.6 "runs up to 100 and goes past pretty
11
+ fast").
12
+
13
+ This is a DISPLAY-ONLY correction applied on the bridge's EMIT path: we
14
+ override the emitted { max, pct, over } using the real window for the reported
15
+ model id. The SDK's own context management is untouched — no behavior change,
16
+ no racy control request (H42-safe by construction).
17
+
18
+ Numbers are the provider-HEADLINED context windows (verified per row below).
19
+ The meter is inherently approximate, so we match what each provider's own docs
20
+ advertise rather than power-of-two token counts.
21
+ ───────────────────────────────────────────────────────────── */
22
+
23
+ // id substring (case-insensitive) → context window in tokens.
24
+ // Ordered MOST-SPECIFIC first (glm-4.6 before glm-4.5) — first hit wins.
25
+ // Sources verified 2026-07-04:
26
+ // glm-4.6 200K — https://openrouter.ai/z-ai/glm-4.6 · https://docs.z.ai
27
+ // glm-4.5 128K — https://github.com/zai-org/GLM-4.5 (README)
28
+ // deepseek v3.x 128K — https://api-docs.deepseek.com (deepseek-chat/-reasoner, V3.1)
29
+ // kimi-k2 256K — https://huggingface.co/moonshotai/Kimi-K2-Instruct-0905
30
+ // qwen3-coder 256K — https://huggingface.co/Qwen/Qwen3-Coder-Next (native)
31
+ const KNOWN_WINDOWS = [
32
+ { match: 'glm-4.6', max: 200_000 },
33
+ { match: 'glm-4.5', max: 128_000 },
34
+ { match: 'deepseek', max: 128_000 },
35
+ { match: 'kimi-k2', max: 256_000 },
36
+ { match: 'qwen3-coder', max: 256_000 },
37
+ ]
38
+
39
+ // A Claude model id — the ONLY ids whose SDK-reported window we trust verbatim.
40
+ // Everything else is a candidate for correction. (Kept broad on purpose: any
41
+ // real Claude id matches, so the 1M/200k Claude paths never regress.)
42
+ export function isClaudeModel(id) {
43
+ return /claude|anthropic|opus|sonnet|haiku/i.test(String(id || ''))
44
+ }
45
+
46
+ // Look up a corrected window (tokens) for a model id. Returns null if unknown.
47
+ export function windowForModel(modelId) {
48
+ const id = String(modelId || '').toLowerCase()
49
+ if (!id) return null
50
+ for (const w of KNOWN_WINDOWS) {
51
+ if (id.includes(w.match)) return w.max
52
+ }
53
+ return null
54
+ }
55
+
56
+ // Parse TP_CONTEXT_MAX (a positive integer token count) from the environment.
57
+ // Per-bridge override for models not in the map. null if unset/non-numeric/<=0.
58
+ export function envContextMax(env = process.env) {
59
+ const raw = env?.TP_CONTEXT_MAX
60
+ if (raw == null || raw === '') return null
61
+ const n = Number(raw)
62
+ return Number.isFinite(n) && n > 0 ? Math.floor(n) : null
63
+ }
64
+
65
+ // Correct a raw ctx object from the SDK's getContextUsage().
66
+ //
67
+ // raw: { used, max, pct, model } (any field may be missing)
68
+ // env: process.env-like (for the TP_CONTEXT_MAX override)
69
+ //
70
+ // Returns a NEW ctx object. Precedence:
71
+ // 1. TP_CONTEXT_MAX override — wins for ANY model (incl. ids not in the map).
72
+ // 2. A known non-Claude window from the map.
73
+ // 3. Otherwise passthrough — Claude models and unknown ids keep the SDK's max.
74
+ // When a correction applies, max is replaced, pct is recomputed and clamped to
75
+ // 0..100, and `over: true` is set when used exceeds the corrected window.
76
+ // In every case pct is clamped so a lying >100% never reaches the UI.
77
+ export function correctContext(raw, env = process.env) {
78
+ if (!raw || typeof raw !== 'object') return raw
79
+ const used = Number(raw.used)
80
+ const override = envContextMax(env)
81
+ // The map fires only for non-Claude ids (point 4: Claude untouched); the
82
+ // env override is explicit and deliberate, so it applies even to Claude.
83
+ const mapped = override == null && !isClaudeModel(raw.model)
84
+ ? windowForModel(raw.model)
85
+ : null
86
+ const correctedMax = override ?? mapped
87
+
88
+ if (correctedMax == null) {
89
+ // No correction available — but never ship a >100% pct.
90
+ const pct = Number(raw.pct)
91
+ if (Number.isFinite(pct) && pct > 100) return { ...raw, pct: 100, over: true }
92
+ return raw
93
+ }
94
+
95
+ const max = correctedMax
96
+ const pct = Number.isFinite(used) && max > 0
97
+ ? Math.min(100, Math.max(0, Math.round((used / max) * 100)))
98
+ : raw.pct
99
+ const over = Number.isFinite(used) && used > max
100
+ return { ...raw, max, pct, over }
101
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "thinkpool-pair",
3
- "version": "0.7.155",
3
+ "version": "0.7.157",
4
4
  "description": "Share a local coding-agent CLI (Claude Code, Codex, Gemini, Aider, …) into a ThinkPool Code room, live.",
5
5
  "type": "module",
6
6
  "bin": {
@@ -11,6 +11,7 @@
11
11
  "sdk-smoke.mjs",
12
12
  "launcher.mjs",
13
13
  "byok-detect.mjs",
14
+ "context-windows.mjs",
14
15
  "claude-session.mjs",
15
16
  "update-gate.mjs",
16
17
  "event-id.mjs",
package/session-store.mjs CHANGED
@@ -215,14 +215,3 @@ export function loadNames(room) {
215
215
  export function saveNames(room, names) {
216
216
  try { ensureDir(room); fs.writeFileSync(namesFile(room), JSON.stringify(names || {})) } catch { /* noop */ }
217
217
  }
218
-
219
- // Session recap cache (2026-07-02, Session Info panel). The last on-demand summary,
220
- // stored as a dotfile (NOT *.json → never in loadAll) so it survives a bridge restart
221
- // and any device that opens the panel can fetch it (persists across devices). One per room.
222
- const summaryFile = (room) => path.join(dir(room), '.summary')
223
- export function loadSummary(room) {
224
- try { return JSON.parse(fs.readFileSync(summaryFile(room), 'utf8')) || null } catch { return null }
225
- }
226
- export function saveSummary(room, summary) {
227
- try { ensureDir(room); fs.writeFileSync(summaryFile(room), JSON.stringify(summary || null)) } catch { /* noop */ }
228
- }