thinkpool-pair 0.7.154 → 0.7.156
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -0
- package/bridge.mjs +73 -96
- package/claude-session.mjs +3 -44
- package/context-windows.mjs +101 -0
- package/event-id.mjs +57 -0
- package/flow-review-reflect.mjs +115 -0
- package/flow-review.mjs +55 -2
- package/package.json +3 -1
- package/session-store.mjs +0 -11
package/README.md
CHANGED
|
@@ -153,5 +153,22 @@ Provider config is stored in `~/.thinkpool-pair/provider.json` (mode 0600) and
|
|
|
153
153
|
applies to every bridge on the machine. Restart the bridge (or its launchd
|
|
154
154
|
service) after changing it — the choice is read once at startup.
|
|
155
155
|
|
|
156
|
+
### Context meter for non-Claude models
|
|
157
|
+
|
|
158
|
+
The context-usage meter (`ctx N%` in the room) is computed by the Agent SDK from
|
|
159
|
+
the model id it thinks it's talking to — for a non-Claude model bridged in over
|
|
160
|
+
an Anthropic-compatible endpoint, that assumed window is wrong, so the meter can
|
|
161
|
+
run past 100%. The bridge corrects it for common models automatically (GLM-4.5/4.6,
|
|
162
|
+
DeepSeek V3.x, Kimi K2, Qwen3-Coder — matched by model id).
|
|
163
|
+
|
|
164
|
+
For any model not in that list, set the real window explicitly per bridge:
|
|
165
|
+
|
|
166
|
+
```bash
|
|
167
|
+
export TP_CONTEXT_MAX=131072 # your model's context window, in tokens
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
`TP_CONTEXT_MAX` overrides the built-in map (and applies even to Claude ids if
|
|
171
|
+
you set it). Unset it to fall back to the automatic correction.
|
|
172
|
+
|
|
156
173
|
Public anon creds are embedded (the same ones the web app ships). Override with
|
|
157
174
|
`TP_SUPABASE_URL` / `TP_SUPABASE_ANON` if needed. Set `TP_NAME` to label yourself.
|
package/bridge.mjs
CHANGED
|
@@ -39,7 +39,7 @@ import { randomUUID } from 'node:crypto'
|
|
|
39
39
|
import { createClient } from '@supabase/supabase-js'
|
|
40
40
|
import { createSdkMcpServer, tool } from '@anthropic-ai/claude-agent-sdk'
|
|
41
41
|
import { z } from 'zod'
|
|
42
|
-
import { startClaudeSession
|
|
42
|
+
import { startClaudeSession } from './claude-session.mjs'
|
|
43
43
|
import { FLOW_CONDUCTOR_PROMPT, FLOW_LANE_PROMPT, buildConductorEnv, assembleCrossWaveContext, buildLanePrompt } from './flow-conductor.mjs'
|
|
44
44
|
import { normalizePlanOutput, laneModelFor } from './flow-task-graph.mjs' // FL-B1 — validate the conductor's submit_flow_plan task-graph; laneModelFor — per-slice model tier (2026-07-03-flow-lane-model-tiers)
|
|
45
45
|
// S1 (context-offload) — durable digest store. mark_flow_done digests a closed slice in;
|
|
@@ -57,8 +57,9 @@ function stopFlowPreviews (flowId, laneId = null) {
|
|
|
57
57
|
if (laneId ? key === `lane:${flowId}:${laneId}` : key.startsWith(`lane:${flowId}:`)) { try { h.stop() } catch { /* noop */ } }
|
|
58
58
|
}
|
|
59
59
|
}
|
|
60
|
-
import { FLOW_REVIEWER_PROMPT, revertLane, parseReviewVerdict } from './flow-review.mjs'
|
|
60
|
+
import { FLOW_REVIEWER_PROMPT, revertLane, parseReviewVerdict, reviewVerdictToReflection } from './flow-review.mjs'
|
|
61
61
|
import { reviewGateDecision } from './flow-review-gate.mjs'
|
|
62
|
+
import { pairAdjudicationPrompt, reviewReflectionDecision, REVIEW_DEFAULTS } from './flow-review-reflect.mjs'
|
|
62
63
|
import { mergeWorktrees, inlineSingleHtml, initRepo } from './flow-assembly.mjs'
|
|
63
64
|
import { canDispatch, FLOW_LIMITS, makeBudget, recordSpend, killSwitchEnv } from './flow-budget.mjs'
|
|
64
65
|
// S4 slice 2 — clean re-dispatch. When a lane is killed mid-tool-call (budget/review/restart)
|
|
@@ -83,8 +84,8 @@ const flowRedispatch = new Map()
|
|
|
83
84
|
const flowBudgets = new Map()
|
|
84
85
|
import { formatPeek, PEEK, siblingsOf, resolveSibling, crossPostDecision, CROSSPOST, spawnDecision, SPAWN, CROSSROOM, formatPairRoster, crossRoomPostDecision, formatRoomNow } from './cross-terminal.mjs'
|
|
85
86
|
import { turnInFlight } from './update-gate.mjs'
|
|
86
|
-
import { saveSession, flushSession, deleteSession, loadAll, canResume, loadPtyId, savePtyId, loadNames, saveNames, appendDurableEvents, seedDurableEvents, readDurablePage, readDurableOldestSeq
|
|
87
|
-
import { stampEvent, makeSeqCounter, maxSeq, seqable, capReplayEvents, chunkReplayEvents, boundEventForBroadcast, inlineImageBlocks } from './event-id.mjs'
|
|
87
|
+
import { saveSession, flushSession, deleteSession, loadAll, canResume, loadPtyId, savePtyId, loadNames, saveNames, appendDurableEvents, seedDurableEvents, readDurablePage, readDurableOldestSeq } from './session-store.mjs'
|
|
88
|
+
import { stampEvent, makeSeqCounter, maxSeq, seqable, capReplayEvents, chunkReplayEvents, boundEventForBroadcast, inlineImageBlocks, usageReportLine } from './event-id.mjs'
|
|
88
89
|
import { makeThrottledTrack } from './presence.mjs'
|
|
89
90
|
|
|
90
91
|
// Public client creds (the same anon values the web app ships — safe to embed).
|
|
@@ -1376,14 +1377,71 @@ function openStructured({ id, model, resume, log, commands, mode, spawnedBy, rol
|
|
|
1376
1377
|
} catch (e) {
|
|
1377
1378
|
return { ok: false, message: `Review verdict REJECTED: ${e?.message || e}. Re-Write FLOW_REVIEW.json with {"pass":<boolean>,"reasons":["<specific finding>"],"taskKey":"<the slice you reviewed>"}.` }
|
|
1378
1379
|
}
|
|
1379
|
-
|
|
1380
|
+
// E1 A1/A2 — the BOUNDED reviewer loop, live side. Each FLOW_REVIEW.json write is ONE
|
|
1381
|
+
// hunt round; the governor decides continue-vs-stop from the round count + the lane's
|
|
1382
|
+
// REAL budget (flowBudgets ledger). A bare pass:true (happy path held, not yet exhausted)
|
|
1383
|
+
// under both ceilings → dig ONE more round instead of concluding; a concrete failure, an
|
|
1384
|
+
// exhausted pass, or a ceiling (rounds OR budget) → terminal, surfaced to the pair (P3).
|
|
1385
|
+
// Fail-closed: a lane at/over its cap concludes (surface) and is never re-prompted. Held
|
|
1386
|
+
// for live-room verification (bridge is inert until published); the decision logic is
|
|
1387
|
+
// unit-proven in flow-review.loop.test.mjs + flow-review-reflect.test.mjs.
|
|
1388
|
+
const round = (entry.flowReviewRound = (entry.flowReviewRound || 0) + 1)
|
|
1389
|
+
const b = flowBudgets.get(entry.flowSessionId) || null
|
|
1390
|
+
const decision = reviewReflectionDecision({
|
|
1391
|
+
round,
|
|
1392
|
+
maxRounds: REVIEW_DEFAULTS.maxRounds,
|
|
1393
|
+
spentTokens: b ? b.spentTokens : 0,
|
|
1394
|
+
budgetCap: b ? b.capTokens : null,
|
|
1395
|
+
...reviewVerdictToReflection(v),
|
|
1396
|
+
})
|
|
1397
|
+
|
|
1398
|
+
// Not concluded, under both ceilings → send the reviewer back for another bounded round.
|
|
1399
|
+
// No revert, no verdict broadcast, no markFlowDone — the loop stays open.
|
|
1400
|
+
if (!decision.stop) {
|
|
1401
|
+
const next = round + 1
|
|
1402
|
+
try {
|
|
1403
|
+
entry.session?.sendTurn(
|
|
1404
|
+
`[Flow review — round ${next}/${REVIEW_DEFAULTS.maxRounds}] The happy path held, but you have NOT reported your checks exhausted and you are under both the round and budget ceilings. Dig one more round: hunt the edge cases, the reload, the second click, concurrent use, the error path — the places the builder didn't. Then re-Write FLOW_REVIEW.json — pass:false with a specific reason if you break it, or pass:true AND exhausted:true if you genuinely have nothing left to check.`,
|
|
1405
|
+
)
|
|
1406
|
+
} catch { /* lane may have closed mid-verdict */ }
|
|
1407
|
+
process.stderr.write(`\n ${A.dim}◆ review round ${round} inconclusive — digging again (${next}/${REVIEW_DEFAULTS.maxRounds}) on ${target || entry.flowTaskKey}${A.rst}\n`)
|
|
1408
|
+
return { ok: true, message: `Round ${round} recorded (pass, not yet exhausted). Keep hunting — asked you for round ${next}/${REVIEW_DEFAULTS.maxRounds}.` }
|
|
1409
|
+
}
|
|
1410
|
+
|
|
1411
|
+
// Terminal outcome. reject → revert the reviewed slice; surface → hand the inconclusive
|
|
1412
|
+
// result to the pair WITHOUT reverting (no failure was reproduced); pass → accept.
|
|
1413
|
+
if (decision.action === 'reject' && target) {
|
|
1380
1414
|
bcast('flow-revert', { term: id, flowId: entry.flowSessionId, taskKey: target }, flowChannel)
|
|
1381
|
-
process.stderr.write(`\n ${A.yel}◆ review
|
|
1415
|
+
process.stderr.write(`\n ${A.yel}◆ review REJECT (round ${round}) — reverting ${target}: ${v.reasons.join('; ').slice(0, 120)}${A.rst}\n`)
|
|
1382
1416
|
} else {
|
|
1383
|
-
process.stderr.write(`\n ${A.cyan}◆ review
|
|
1417
|
+
process.stderr.write(`\n ${A.cyan}◆ review ${decision.action.toUpperCase()} (round ${round}) — ${target || entry.flowTaskKey}${A.rst}\n`)
|
|
1384
1418
|
}
|
|
1419
|
+
// P3 (pair co-adjudication) — surface the TERMINAL verdict into the ROOM as a
|
|
1420
|
+
// challengeable prompt, ALONGSIDE the gate action above. A solo Bugbot's verdict is
|
|
1421
|
+
// final; ours is a prompt for the two humans + Pool to argue. Additive: an extra
|
|
1422
|
+
// broadcast, does NOT change the revert/gate flow.
|
|
1423
|
+
bcast('flow-review-verdict', {
|
|
1424
|
+
term: id, flowId: entry.flowSessionId,
|
|
1425
|
+
taskKey: target || entry.flowTaskKey,
|
|
1426
|
+
pass: v.pass,
|
|
1427
|
+
reasons: v.reasons,
|
|
1428
|
+
action: decision.action,
|
|
1429
|
+
rounds: round,
|
|
1430
|
+
surfaceToPair: decision.surfaceToPair,
|
|
1431
|
+
prompt: pairAdjudicationPrompt({
|
|
1432
|
+
action: decision.action,
|
|
1433
|
+
reason: decision.reason,
|
|
1434
|
+
taskKey: target || entry.flowTaskKey,
|
|
1435
|
+
findings: v.reasons,
|
|
1436
|
+
}),
|
|
1437
|
+
}, flowChannel)
|
|
1385
1438
|
const doneMsg = await markFlowDone()
|
|
1386
|
-
|
|
1439
|
+
const label = decision.action === 'reject'
|
|
1440
|
+
? `REJECT (round ${round}) — reverting ${target || '(no target)'} (${v.reasons.join('; ').slice(0, 140)})`
|
|
1441
|
+
: decision.action === 'surface'
|
|
1442
|
+
? `SURFACED to the pair after ${round} round${round === 1 ? '' : 's'} — ${decision.reason}`
|
|
1443
|
+
: `PASS (round ${round})`
|
|
1444
|
+
return { ok: true, message: `Review verdict recorded: ${label}. ${doneMsg}` }
|
|
1387
1445
|
}
|
|
1388
1446
|
// Identity for the durable archive — pushLog appends every new transcript event to
|
|
1389
1447
|
// <room>/<id>.events.jsonl keyed off these. Seed the archive once from the restored
|
|
@@ -2334,6 +2392,13 @@ channel
|
|
|
2334
2392
|
s.session.sendTurn(text)
|
|
2335
2393
|
return
|
|
2336
2394
|
}
|
|
2395
|
+
// /usage → report this session's token + cost totals WITHOUT burning an agent turn.
|
|
2396
|
+
// Computed by REPLAYING the session's own retained result events (each completed turn
|
|
2397
|
+
// persists kind:'result' with usage + costUsd + model — see onEvent). No new accumulator
|
|
2398
|
+
// state: the log is durable, so this survives a bridge restart for free. Reported as ◆
|
|
2399
|
+
// control lines (like /model) so both members AND late-joiners see it. Scope is honest —
|
|
2400
|
+
// the Agent SDK exposes tokens + cost, not plan/rate-limit meters, and the card says so.
|
|
2401
|
+
if (/^\/usage\s*$/.test(text)) { ctlLine(usageReportLine(s.model, s.log)); return }
|
|
2337
2402
|
// Pre-flight budget gate (room scope) — refuse the billable agent turn if the room hit
|
|
2338
2403
|
// its cap. The slash controls above (/model, /clear, /compact) are intentionally exempt:
|
|
2339
2404
|
// they let a user manage or REDUCE spend even when capped. Legible refusal, no silent stall.
|
|
@@ -2400,94 +2465,6 @@ channel
|
|
|
2400
2465
|
saveNames(room, termNames)
|
|
2401
2466
|
announce()
|
|
2402
2467
|
})
|
|
2403
|
-
// Session Info panel (2026-07-02, plain-English recap 2026-07-02) — a cheap one-shot
|
|
2404
|
-
// recap on demand. Writes a plain-English "Done so far" / "Working on now" bullet
|
|
2405
|
-
// summary (Markdown) of what the whole session accomplished + what's live — NOT a
|
|
2406
|
-
// per-terminal technical list (Max's ask: clean, readable, no jargon/terminal names).
|
|
2407
|
-
// Runs on the room's own subscription auth via oneShotSummary (cheap haiku →
|
|
2408
|
-
// provider-default fallback). Never auto — only on the panel button. The web caches
|
|
2409
|
-
// the result in React state (no re-charge on re-open within a page load).
|
|
2410
|
-
.on('broadcast', { event: 'code-summarize' }, async ({ payload }) => {
|
|
2411
|
-
const by = payload?.by || null
|
|
2412
|
-
// Peek — any device opening the panel fetches the cached recap (cross-device +
|
|
2413
|
-
// cross-restart persistence) WITHOUT spending a model call. No cache → silent.
|
|
2414
|
-
if (payload?.peek) {
|
|
2415
|
-
const cached = loadSummary(room)
|
|
2416
|
-
if (cached && cached.text) bcast('code-summary', { status: 'done', ...cached })
|
|
2417
|
-
return
|
|
2418
|
-
}
|
|
2419
|
-
bcast('code-summary', { status: 'generating', by })
|
|
2420
|
-
try {
|
|
2421
|
-
// Pull assistant/user text out of a transcript-event array, keep the tail.
|
|
2422
|
-
const logDigest = (events, max = 1200) => {
|
|
2423
|
-
const parts = []
|
|
2424
|
-
for (const e of (events || [])) {
|
|
2425
|
-
if (e?.kind === 'assistant') { for (const b of (e.blocks || [])) if (b?.type === 'text' && b.text) parts.push(b.text) }
|
|
2426
|
-
else if (e?.kind === 'you' && e.text) parts.push(`user: ${e.text}`)
|
|
2427
|
-
}
|
|
2428
|
-
return parts.join('\n').replace(/\n{3,}/g, '\n\n').slice(-max)
|
|
2429
|
-
}
|
|
2430
|
-
// LIVE structured terminals — named, doing/done, with a recent digest.
|
|
2431
|
-
const live = []
|
|
2432
|
-
let env = null, cwd = null
|
|
2433
|
-
for (const [id, e] of sessions) {
|
|
2434
|
-
if (e.kind !== 'structured') continue
|
|
2435
|
-
live.push({ name: termNames[id] || 'Terminal', status: e.session?.turnActive ? 'working' : 'done', digest: logDigest(e.log, 1100) })
|
|
2436
|
-
if (!env) { env = { ...process.env }; cwd = e.cwd || process.cwd() }
|
|
2437
|
-
}
|
|
2438
|
-
// CLOSED terminals — named but not live. Aggregate only; cap the read to bound cost.
|
|
2439
|
-
const liveIdSet = new Set(sessions.keys())
|
|
2440
|
-
const closedIds = Object.keys(termNames).filter(id => !liveIdSet.has(id))
|
|
2441
|
-
const closedDigests = []
|
|
2442
|
-
for (const id of closedIds.slice(0, 4)) {
|
|
2443
|
-
try {
|
|
2444
|
-
const page = readDurablePage(room, id, Number.MAX_SAFE_INTEGER, 30)
|
|
2445
|
-
const evs = Array.isArray(page) ? page : (Array.isArray(page?.events) ? page.events : [])
|
|
2446
|
-
const d = logDigest(evs, 450)
|
|
2447
|
-
if (d) closedDigests.push(d)
|
|
2448
|
-
} catch { /* no retained detail for this closed lane */ }
|
|
2449
|
-
}
|
|
2450
|
-
if (!live.length && !closedIds.length) { bcast('code-summary', { status: 'error', error: 'Nothing to summarize yet.', by }); return }
|
|
2451
|
-
|
|
2452
|
-
const liveBlock = live.length
|
|
2453
|
-
? live.map(t => `LIVE TERMINAL "${t.name}" [${t.status}]:\n${t.digest || '(no recent activity)'}`).join('\n\n')
|
|
2454
|
-
: '(no live terminals)'
|
|
2455
|
-
const closedBlock = closedIds.length
|
|
2456
|
-
? `\n\nThere are also ${closedIds.length} CLOSED terminal(s). Recent excerpts (do NOT name them individually):\n${(closedDigests.join('\n---\n') || '(no retained detail)').slice(0, 1400)}`
|
|
2457
|
-
: ''
|
|
2458
|
-
|
|
2459
|
-
const prompt = `Write a short, friendly recap of this coding session for someone non-technical who just wants to know what's going on. Plain English only — no jargon, no file names, no branch names, no terminal names or IDs, no code, no counts of terminals. Explain what the work MEANS and why it matters, not the mechanics.
|
|
2460
|
-
|
|
2461
|
-
Output ONLY this Markdown shape, nothing else:
|
|
2462
|
-
|
|
2463
|
-
**Done so far**
|
|
2464
|
-
- <one plain-English thing that got finished>
|
|
2465
|
-
- <another>
|
|
2466
|
-
|
|
2467
|
-
**Working on now**
|
|
2468
|
-
- <what's actively being worked on, in plain English>
|
|
2469
|
-
|
|
2470
|
-
Rules:
|
|
2471
|
-
- Merge ALL the finished work — live and closed terminals alike — into "Done so far". Combine similar items; do NOT organise by terminal or list them. Aim for 3–6 bullets.
|
|
2472
|
-
- Put only genuinely in-progress work under "Working on now". If nothing is active, write exactly: "- Nothing in progress right now — everything's wrapped up."
|
|
2473
|
-
- Each bullet is one clear sentence, ~16 words max, no trailing period, and no technical term a normal person wouldn't understand.
|
|
2474
|
-
|
|
2475
|
-
Here's the raw session activity to summarize:
|
|
2476
|
-
|
|
2477
|
-
${liveBlock}${closedBlock}
|
|
2478
|
-
|
|
2479
|
-
Recap:`
|
|
2480
|
-
|
|
2481
|
-
const res = await oneShotSummary({ prompt, env: env || process.env, cwd: cwd || process.cwd() })
|
|
2482
|
-
if (!res || !res.text) { bcast('code-summary', { status: 'error', error: 'Summary model unavailable on this provider.', by }); return }
|
|
2483
|
-
const summary = { text: res.text, model: res.model, generatedAt: Date.now() }
|
|
2484
|
-
try { saveSummary(room, summary) } catch { /* cache is best-effort */ }
|
|
2485
|
-
bcast('code-summary', { status: 'done', ...summary, by })
|
|
2486
|
-
process.stderr.write(`\n ◆ session summary generated (${res.model}) for ${by || 'someone'}.\n`)
|
|
2487
|
-
} catch (e) {
|
|
2488
|
-
bcast('code-summary', { status: 'error', error: String(e?.message || e).slice(0, 140), by })
|
|
2489
|
-
}
|
|
2490
|
-
})
|
|
2491
2468
|
.on('broadcast', { event: 'who' }, announce)
|
|
2492
2469
|
.subscribe(status => {
|
|
2493
2470
|
if (status === 'SUBSCRIBED') {
|
package/claude-session.mjs
CHANGED
|
@@ -17,6 +17,7 @@ import { query } from '@anthropic-ai/claude-agent-sdk'
|
|
|
17
17
|
import { sanitizeSession } from './transcript-sanitize.mjs'
|
|
18
18
|
import { reviewGatePreToolDecision } from './flow-review-gate.mjs'
|
|
19
19
|
import { crossPostNeedsCard } from './cross-terminal.mjs'
|
|
20
|
+
import { correctContext } from './context-windows.mjs'
|
|
20
21
|
|
|
21
22
|
// ── risk classification — the accent/danger tier of the permission card ──
|
|
22
23
|
// low (read-only) · medium (writes/runs) · network (leaves the machine) ·
|
|
@@ -145,48 +146,6 @@ const TP_ROOM_REMINDER = [
|
|
|
145
146
|
'Verify before claiming done — show runtime evidence you produced, not "should work, go test it".',
|
|
146
147
|
].join(' ')
|
|
147
148
|
|
|
148
|
-
// ── One-shot summary (2026-07-02, Session Info panel) ────────────────────────
|
|
149
|
-
// A bare, cheap, cold model call — same shape as the haikuSuggest fallback: runs
|
|
150
|
-
// on the room's own subscription auth (env carries the OAuth — no API key), NO
|
|
151
|
-
// settingSources / MCP / tools, maxTurns 1. Used by the bridge's code-summarize
|
|
152
|
-
// handler to write a session recap on demand. Model fallback (Max's ask): prefer a
|
|
153
|
-
// cheap haiku-class model; if the provider (GLM / OpenRouter / custom base-url) has
|
|
154
|
-
// no such model, retry with the provider default. Returns { text, model } or null.
|
|
155
|
-
export async function oneShotSummary({ prompt, env, cwd, timeoutMs = 25000 }) {
|
|
156
|
-
const attempt = async (model) => {
|
|
157
|
-
const ac = new AbortController()
|
|
158
|
-
const t = setTimeout(() => { try { ac.abort() } catch { /* noop */ } }, timeoutMs)
|
|
159
|
-
try {
|
|
160
|
-
const hq = query({
|
|
161
|
-
prompt,
|
|
162
|
-
options: {
|
|
163
|
-
...(model ? { model } : {}),
|
|
164
|
-
...(cwd ? { cwd } : {}),
|
|
165
|
-
env,
|
|
166
|
-
maxTurns: 1,
|
|
167
|
-
permissionMode: 'bypassPermissions',
|
|
168
|
-
settingSources: [],
|
|
169
|
-
strictMcpConfig: true,
|
|
170
|
-
mcpServers: {},
|
|
171
|
-
abortController: ac,
|
|
172
|
-
},
|
|
173
|
-
})
|
|
174
|
-
let out = ''
|
|
175
|
-
for await (const mm of hq) {
|
|
176
|
-
if (mm.type === 'assistant') for (const b of (mm.message?.content || [])) if (b.type === 'text') out += b.text
|
|
177
|
-
if (mm.type === 'result') break
|
|
178
|
-
}
|
|
179
|
-
return out.trim()
|
|
180
|
-
} finally { clearTimeout(t) }
|
|
181
|
-
}
|
|
182
|
-
// Prefer cheap haiku; fall back to the provider's default model if it's absent.
|
|
183
|
-
try { const out = await attempt('claude-haiku-4-5'); if (out) return { text: out, model: 'claude-haiku-4-5' } }
|
|
184
|
-
catch { /* provider has no haiku — fall through to default */ }
|
|
185
|
-
try { const out = await attempt(undefined); if (out) return { text: out, model: 'default' } }
|
|
186
|
-
catch { /* provider default also failed */ }
|
|
187
|
-
return null
|
|
188
|
-
}
|
|
189
|
-
|
|
190
149
|
export function startClaudeSession({ cwd, model, resume, env, mode: initialMode = 'default', onEvent, requestPermission, mcpServers, crossPostGate, crossRoomPostGate, didSpawnTarget = null, rolePrompt, blockSubagents = false, onSubmitPlan = null, onLaneDone = null, onReviewVerdict = null, reviewGate = null, lazy = false, roomContext = null }) {
|
|
191
150
|
// Per-turn reminder + live ROOM NOW tail. roomContext (bridge-supplied) returns the
|
|
192
151
|
// room's CURRENT state — sibling lanes, active git worktrees — or null. The static
|
|
@@ -788,7 +747,7 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
|
|
|
788
747
|
;(async () => {
|
|
789
748
|
try {
|
|
790
749
|
const c = await q?.getContextUsage?.()
|
|
791
|
-
if (c) { emit({ kind: 'usage', ctx: { used: c.totalTokens, max: c.maxTokens, pct: Math.round(c.percentage), model: c.model } }); if (c.model) curModel = c.model }
|
|
750
|
+
if (c) { emit({ kind: 'usage', ctx: correctContext({ used: c.totalTokens, max: c.maxTokens, pct: Math.round(c.percentage), model: c.model }) }); if (c.model) curModel = c.model }
|
|
792
751
|
} catch { /* control req may be unavailable */ }
|
|
793
752
|
})()
|
|
794
753
|
break
|
|
@@ -932,7 +891,7 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
|
|
|
932
891
|
let ctx = null
|
|
933
892
|
try {
|
|
934
893
|
const c = await q?.getContextUsage?.()
|
|
935
|
-
if (c) { ctx = { used: c.totalTokens, max: c.maxTokens, pct: Math.round(c.percentage), model: c.model }; if (c.model) curModel = c.model }
|
|
894
|
+
if (c) { ctx = correctContext({ used: c.totalTokens, max: c.maxTokens, pct: Math.round(c.percentage), model: c.model }); if (c.model) curModel = c.model }
|
|
936
895
|
} catch { /* control req may be unavailable */ }
|
|
937
896
|
const u = m.usage || {}
|
|
938
897
|
emit({ kind: 'usage', costUsd: m.total_cost_usd ?? null, tokens: (u.input_tokens || 0) + (u.output_tokens || 0), ctx })
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
/* ─────────────────────────────────────────────────────────────
|
|
2
|
+
context-windows — correct the /code context meter for non-Claude BYOK models.
|
|
3
|
+
|
|
4
|
+
The Claude Agent SDK computes its context-usage meter
|
|
5
|
+
(getContextUsage → { totalTokens, maxTokens, percentage }) from the model
|
|
6
|
+
STRING it believes it is talking to. When a user bridges a NON-Claude model
|
|
7
|
+
in over an Anthropic-compatible endpoint (ANTHROPIC_BASE_URL — GLM, DeepSeek,
|
|
8
|
+
Kimi, Qwen, … via OpenRouter or a provider's own compat endpoint), the SDK's
|
|
9
|
+
maxTokens is for the WRONG model, so the meter fills at the wrong rate and can
|
|
10
|
+
shoot past 100% (Max, 2026-07-04: GLM-4.6 "runs up to 100 and goes past pretty
|
|
11
|
+
fast").
|
|
12
|
+
|
|
13
|
+
This is a DISPLAY-ONLY correction applied on the bridge's EMIT path: we
|
|
14
|
+
override the emitted { max, pct, over } using the real window for the reported
|
|
15
|
+
model id. The SDK's own context management is untouched — no behavior change,
|
|
16
|
+
no racy control request (H42-safe by construction).
|
|
17
|
+
|
|
18
|
+
Numbers are the provider-HEADLINED context windows (verified per row below).
|
|
19
|
+
The meter is inherently approximate, so we match what each provider's own docs
|
|
20
|
+
advertise rather than power-of-two token counts.
|
|
21
|
+
───────────────────────────────────────────────────────────── */
|
|
22
|
+
|
|
23
|
+
// id substring (case-insensitive) → context window in tokens.
|
|
24
|
+
// Ordered MOST-SPECIFIC first (glm-4.6 before glm-4.5) — first hit wins.
|
|
25
|
+
// Sources verified 2026-07-04:
|
|
26
|
+
// glm-4.6 200K — https://openrouter.ai/z-ai/glm-4.6 · https://docs.z.ai
|
|
27
|
+
// glm-4.5 128K — https://github.com/zai-org/GLM-4.5 (README)
|
|
28
|
+
// deepseek v3.x 128K — https://api-docs.deepseek.com (deepseek-chat/-reasoner, V3.1)
|
|
29
|
+
// kimi-k2 256K — https://huggingface.co/moonshotai/Kimi-K2-Instruct-0905
|
|
30
|
+
// qwen3-coder 256K — https://huggingface.co/Qwen/Qwen3-Coder-Next (native)
|
|
31
|
+
const KNOWN_WINDOWS = [
|
|
32
|
+
{ match: 'glm-4.6', max: 200_000 },
|
|
33
|
+
{ match: 'glm-4.5', max: 128_000 },
|
|
34
|
+
{ match: 'deepseek', max: 128_000 },
|
|
35
|
+
{ match: 'kimi-k2', max: 256_000 },
|
|
36
|
+
{ match: 'qwen3-coder', max: 256_000 },
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
// A Claude model id — the ONLY ids whose SDK-reported window we trust verbatim.
|
|
40
|
+
// Everything else is a candidate for correction. (Kept broad on purpose: any
|
|
41
|
+
// real Claude id matches, so the 1M/200k Claude paths never regress.)
|
|
42
|
+
export function isClaudeModel(id) {
|
|
43
|
+
return /claude|anthropic|opus|sonnet|haiku/i.test(String(id || ''))
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
// Look up a corrected window (tokens) for a model id. Returns null if unknown.
|
|
47
|
+
export function windowForModel(modelId) {
|
|
48
|
+
const id = String(modelId || '').toLowerCase()
|
|
49
|
+
if (!id) return null
|
|
50
|
+
for (const w of KNOWN_WINDOWS) {
|
|
51
|
+
if (id.includes(w.match)) return w.max
|
|
52
|
+
}
|
|
53
|
+
return null
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
// Parse TP_CONTEXT_MAX (a positive integer token count) from the environment.
|
|
57
|
+
// Per-bridge override for models not in the map. null if unset/non-numeric/<=0.
|
|
58
|
+
export function envContextMax(env = process.env) {
|
|
59
|
+
const raw = env?.TP_CONTEXT_MAX
|
|
60
|
+
if (raw == null || raw === '') return null
|
|
61
|
+
const n = Number(raw)
|
|
62
|
+
return Number.isFinite(n) && n > 0 ? Math.floor(n) : null
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
// Correct a raw ctx object from the SDK's getContextUsage().
|
|
66
|
+
//
|
|
67
|
+
// raw: { used, max, pct, model } (any field may be missing)
|
|
68
|
+
// env: process.env-like (for the TP_CONTEXT_MAX override)
|
|
69
|
+
//
|
|
70
|
+
// Returns a NEW ctx object. Precedence:
|
|
71
|
+
// 1. TP_CONTEXT_MAX override — wins for ANY model (incl. ids not in the map).
|
|
72
|
+
// 2. A known non-Claude window from the map.
|
|
73
|
+
// 3. Otherwise passthrough — Claude models and unknown ids keep the SDK's max.
|
|
74
|
+
// When a correction applies, max is replaced, pct is recomputed and clamped to
|
|
75
|
+
// 0..100, and `over: true` is set when used exceeds the corrected window.
|
|
76
|
+
// In every case pct is clamped so a lying >100% never reaches the UI.
|
|
77
|
+
export function correctContext(raw, env = process.env) {
|
|
78
|
+
if (!raw || typeof raw !== 'object') return raw
|
|
79
|
+
const used = Number(raw.used)
|
|
80
|
+
const override = envContextMax(env)
|
|
81
|
+
// The map fires only for non-Claude ids (point 4: Claude untouched); the
|
|
82
|
+
// env override is explicit and deliberate, so it applies even to Claude.
|
|
83
|
+
const mapped = override == null && !isClaudeModel(raw.model)
|
|
84
|
+
? windowForModel(raw.model)
|
|
85
|
+
: null
|
|
86
|
+
const correctedMax = override ?? mapped
|
|
87
|
+
|
|
88
|
+
if (correctedMax == null) {
|
|
89
|
+
// No correction available — but never ship a >100% pct.
|
|
90
|
+
const pct = Number(raw.pct)
|
|
91
|
+
if (Number.isFinite(pct) && pct > 100) return { ...raw, pct: 100, over: true }
|
|
92
|
+
return raw
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
const max = correctedMax
|
|
96
|
+
const pct = Number.isFinite(used) && max > 0
|
|
97
|
+
? Math.min(100, Math.max(0, Math.round((used / max) * 100)))
|
|
98
|
+
: raw.pct
|
|
99
|
+
const over = Number.isFinite(used) && used > max
|
|
100
|
+
return { ...raw, max, pct, over }
|
|
101
|
+
}
|
package/event-id.mjs
CHANGED
|
@@ -191,3 +191,60 @@ export function inlineImageBlocks(evt) {
|
|
|
191
191
|
})
|
|
192
192
|
return out
|
|
193
193
|
}
|
|
194
|
+
|
|
195
|
+
// Pure aggregation for the /usage report — replay a session's retained log events and
|
|
196
|
+
// fold every completed turn (kind:'result' with usage) into per-model + total token/cost
|
|
197
|
+
// sums. No side effects, import-safe (bridge/test-usage-summary.mjs exercises it), so the
|
|
198
|
+
// bridge computes /usage by replaying its own durable log — no separate accumulator that a
|
|
199
|
+
// restart would drop. `costUsd` is the SDK's total_cost_usd per turn (BYOK estimate).
|
|
200
|
+
export function summarizeUsage(logEvents) {
|
|
201
|
+
const results = (Array.isArray(logEvents) ? logEvents : []).filter((e) => e && e.kind === 'result' && e.usage)
|
|
202
|
+
const per = new Map()
|
|
203
|
+
let totalIn = 0, totalOut = 0, totalCost = 0, haveCost = false
|
|
204
|
+
for (const e of results) {
|
|
205
|
+
const u = e.usage || {}
|
|
206
|
+
const inTok = (u.input_tokens || 0) + (u.cache_creation_input_tokens || 0) + (u.cache_read_input_tokens || 0)
|
|
207
|
+
const out = u.output_tokens || 0
|
|
208
|
+
const cached = u.cache_read_input_tokens || 0
|
|
209
|
+
const mid = e.model || 'unknown'
|
|
210
|
+
const r = per.get(mid) || { model: mid, in: 0, out: 0, cached: 0 }
|
|
211
|
+
r.in += inTok; r.out += out; r.cached += cached; per.set(mid, r)
|
|
212
|
+
totalIn += inTok; totalOut += out
|
|
213
|
+
if (typeof e.costUsd === 'number') { totalCost += e.costUsd; haveCost = true }
|
|
214
|
+
}
|
|
215
|
+
return { turns: results.length, per: [...per.values()], totalIn, totalOut, totalCost, haveCost }
|
|
216
|
+
}
|
|
217
|
+
|
|
218
|
+
// model id → friendly label for the /usage report: claude-opus-4-8 → "Opus 4.8".
|
|
219
|
+
// Bridge-local twin of the web's prettyModel (structured.jsx); falls back to the raw id.
|
|
220
|
+
export function prettyModelId (m) {
|
|
221
|
+
if (!m) return 'unknown'
|
|
222
|
+
const mm = String(m).match(/(opus|sonnet|haiku|fable)-(\d+)(?:-(\d+))?/i)
|
|
223
|
+
return mm ? `${mm[1][0].toUpperCase()}${mm[1].slice(1)} ${mm[2]}${mm[3] ? '.' + mm[3] : ''}` : String(m)
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
// Build the ONE ◆ control line the /usage command emits, from the session's model +
|
|
227
|
+
// retained log. Pure so the whole command — gate, empty-state, and per-model/total/cost
|
|
228
|
+
// formatting — is exercised deterministically (bridge/test-usage-summary.mjs). The bridge
|
|
229
|
+
// handler is a thin wrapper: ctlLine(usageReportLine(s.model, s.log)). `sessionModel` is
|
|
230
|
+
// the resolved model id the /model state tracks; `logEvents` the session's retained log.
|
|
231
|
+
export function usageReportLine (sessionModel, logEvents) {
|
|
232
|
+
const claudeish = (m) => /claude|anthropic|opus|sonnet|haiku|fable/i.test(String(m || ''))
|
|
233
|
+
const sum = summarizeUsage(logEvents)
|
|
234
|
+
// Gate (Max's ask: "make /usage work if users bridge is on claude llms"). The Code bridge
|
|
235
|
+
// runs the Claude Agent SDK, so a normal lane IS a Claude model — but a session pointed at a
|
|
236
|
+
// non-Claude base URL/model gets a legible line, not a bogus token report. Only assert this
|
|
237
|
+
// when we actually know a model (else a fresh empty session would wrongly trip the gate).
|
|
238
|
+
const anyClaude = claudeish(sessionModel) || sum.per.some((r) => claudeish(r.model))
|
|
239
|
+
if (!anyClaude && (sessionModel || sum.turns)) return 'usage reporting is available on Claude models only'
|
|
240
|
+
if (!sum.turns) return 'usage · no completed turns yet this session'
|
|
241
|
+
const fmtTok = (n) => n >= 1e6 ? (n / 1e6).toFixed(1) + 'M' : n >= 1e3 ? (n / 1e3).toFixed(1) + 'k' : String(n)
|
|
242
|
+
const perStr = sum.per.map((r) => {
|
|
243
|
+
const cch = r.cached ? ` (${fmtTok(r.cached)} cached)` : ''
|
|
244
|
+
return `${prettyModelId(r.model)}: ${fmtTok(r.in)} in${cch} / ${fmtTok(r.out)} out`
|
|
245
|
+
}).join(' · ')
|
|
246
|
+
const total = sum.per.length > 1 ? ` · total ${fmtTok(sum.totalIn)} in / ${fmtTok(sum.totalOut)} out` : ''
|
|
247
|
+
const cost = sum.haveCost ? ` · ~$${sum.totalCost.toFixed(2)} est` : ''
|
|
248
|
+
const turns = `${sum.turns} turn${sum.turns === 1 ? '' : 's'}`
|
|
249
|
+
return `usage · ${turns} · ${perStr}${total}${cost} · tokens + cost only (no plan/rate-limit meters)`
|
|
250
|
+
}
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
/* ─────────────────────────────────────────────────────────────
|
|
2
|
+
flow-review-reflect.mjs — the BOUNDED reflection governor for the
|
|
3
|
+
verifying reviewer lane (heist E1 + P3).
|
|
4
|
+
Spec: docs/specs/2026-07-02-e1p3-verifying-reviewer.md
|
|
5
|
+
Parents: Emergent steal E1 (self-test loop) + Cursor steal P3 (Bugbot).
|
|
6
|
+
|
|
7
|
+
The existing adversarial reviewer (bridge/flow-review.mjs) runs the slice
|
|
8
|
+
once and emits a verdict. E1 makes that a bounded, budget-capped SELF-TEST
|
|
9
|
+
loop: the reviewer may dig across several rounds (re-run, hit edge cases,
|
|
10
|
+
reproduce the acceptance proof harder) — but it MUST stop at a hard ceiling
|
|
11
|
+
and hand off to the humans, never loop unbounded. That unbounded self-heal
|
|
12
|
+
loop is exactly Emergent's most-hated wound (billing a runaway fix-loop back
|
|
13
|
+
to the user); this governor is the structural guarantee we never have it.
|
|
14
|
+
|
|
15
|
+
Two independent stop conditions, both enforced here (not prompted):
|
|
16
|
+
• ROUND ceiling — never more than maxRounds hunt rounds (anti-Emergent).
|
|
17
|
+
• BUDGET ceiling — never spend past the room/lane cap (nexos steal #1 /
|
|
18
|
+
cost-guard; the review lane SPENDS inference, so it inherits N1).
|
|
19
|
+
|
|
20
|
+
Pure + deterministic — the whole loop-control decision in one function so it
|
|
21
|
+
is unit-testable without a live Flow room (the reviewable seam; the live
|
|
22
|
+
wiring into flow-review.mjs + bridge.mjs is the held integration slice,
|
|
23
|
+
same split that shipped S5's reviewGateDecision).
|
|
24
|
+
|
|
25
|
+
P3 (pair co-adjudication): when the loop stops, the outcome is not the end —
|
|
26
|
+
`surfaceToPair` says the verdict/findings must go to the ROOM for the two
|
|
27
|
+
humans + Pool to challenge, rather than silently auto-driving the gate. A
|
|
28
|
+
solo Bugbot's verdict is final; ours is a prompt for the pair to argue.
|
|
29
|
+
───────────────────────────────────────────────────────────── */
|
|
30
|
+
|
|
31
|
+
export const REVIEW_DEFAULTS = Object.freeze({
|
|
32
|
+
maxRounds: 3, // hard hunt ceiling — bounded, never unbounded (anti-Emergent)
|
|
33
|
+
})
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Decide whether the reviewer's self-test loop continues or stops, and why.
|
|
37
|
+
*
|
|
38
|
+
* @param {object} s
|
|
39
|
+
* @param {number} s.round 1-based current hunt round (1 = first pass)
|
|
40
|
+
* @param {number} [s.maxRounds] hard ceiling (default REVIEW_DEFAULTS.maxRounds)
|
|
41
|
+
* @param {number} [s.spentTokens] tokens the review lane has spent so far
|
|
42
|
+
* @param {number|null} [s.budgetCap] the room/lane token cap (null = uncapped)
|
|
43
|
+
* @param {boolean} s.foundFailure this round produced a CONCRETE, reproduced failure
|
|
44
|
+
* @param {boolean} [s.exhausted] the reviewer reports it has nothing left to check
|
|
45
|
+
* @returns {{ stop: boolean, action: 'reject'|'pass'|'surface'|'continue',
|
|
46
|
+
* surfaceToPair: boolean, reason: string }}
|
|
47
|
+
* action:
|
|
48
|
+
* 'reject' — a concrete failure was found; stop, verdict = REJECT
|
|
49
|
+
* 'pass' — checks exhausted with no failure; stop, verdict = PASS
|
|
50
|
+
* 'surface' — a ceiling (budget or rounds) was hit inconclusively; stop,
|
|
51
|
+
* hand the partial findings to the humans (no guess, no loop)
|
|
52
|
+
* 'continue' — keep hunting (still under both ceilings, not yet concluded)
|
|
53
|
+
* surfaceToPair — true whenever the loop STOPS (P3: every terminal outcome is
|
|
54
|
+
* a prompt for the pair, not an automatic gate action).
|
|
55
|
+
*/
|
|
56
|
+
export function reviewReflectionDecision ({
|
|
57
|
+
round,
|
|
58
|
+
maxRounds = REVIEW_DEFAULTS.maxRounds,
|
|
59
|
+
spentTokens = 0,
|
|
60
|
+
budgetCap = null,
|
|
61
|
+
foundFailure = false,
|
|
62
|
+
exhausted = false,
|
|
63
|
+
} = {}) {
|
|
64
|
+
const r = Number.isFinite(round) ? round : 1
|
|
65
|
+
const cap = Number.isFinite(maxRounds) && maxRounds > 0 ? maxRounds : REVIEW_DEFAULTS.maxRounds
|
|
66
|
+
|
|
67
|
+
// 1. A concrete, reproduced failure ends the hunt immediately — default-to-
|
|
68
|
+
// reject is the whole point; no value in more rounds once it's broken.
|
|
69
|
+
if (foundFailure) {
|
|
70
|
+
return { stop: true, action: 'reject', surfaceToPair: true,
|
|
71
|
+
reason: 'concrete failure reproduced — verdict REJECT' }
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
// 2. BUDGET ceiling (nexos #1 / cost-guard). Fail-closed: the moment spend
|
|
75
|
+
// reaches the cap, STOP and hand off — never spend the next unit. Checked
|
|
76
|
+
// before the round ceiling because money is the harder limit.
|
|
77
|
+
if (budgetCap != null && Number.isFinite(budgetCap) && spentTokens >= budgetCap) {
|
|
78
|
+
return { stop: true, action: 'surface', surfaceToPair: true,
|
|
79
|
+
reason: `review budget cap reached (${spentTokens}/${budgetCap} tokens) — surfacing partial findings to the pair` }
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// 3. Reviewer says it has checked everything and found nothing — verdict PASS.
|
|
83
|
+
if (exhausted) {
|
|
84
|
+
return { stop: true, action: 'pass', surfaceToPair: true,
|
|
85
|
+
reason: 'checks exhausted, no failure reproduced — verdict PASS' }
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
// 4. ROUND ceiling (anti-Emergent). Reached the last allowed round without a
|
|
89
|
+
// conclusion → surface to the humans rather than guess or loop forever.
|
|
90
|
+
if (r >= cap) {
|
|
91
|
+
return { stop: true, action: 'surface', surfaceToPair: true,
|
|
92
|
+
reason: `reached the ${cap}-round hunt ceiling without a conclusion — surfacing to the pair (not looping)` }
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
// 5. Under both ceilings, not yet concluded → dig another round.
|
|
96
|
+
return { stop: false, action: 'continue', surfaceToPair: false,
|
|
97
|
+
reason: `round ${r}/${cap}, under budget — continue hunting` }
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* Format a stopped review outcome as the room-facing co-adjudication prompt
|
|
102
|
+
* (P3). Not the gate action itself — a message that invites the two humans +
|
|
103
|
+
* Pool to challenge the reviewer before the verdict is treated as final.
|
|
104
|
+
*/
|
|
105
|
+
export function pairAdjudicationPrompt ({ action, reason, findings = [], taskKey = '' } = {}) {
|
|
106
|
+
const head = {
|
|
107
|
+
reject: `The reviewer REJECTED ${taskKey || 'this slice'}.`,
|
|
108
|
+
pass: `The reviewer PASSED ${taskKey || 'this slice'}.`,
|
|
109
|
+
surface: `The reviewer could not conclude on ${taskKey || 'this slice'}.`,
|
|
110
|
+
}[action] || `Review outcome for ${taskKey || 'this slice'}.`
|
|
111
|
+
const body = findings.length
|
|
112
|
+
? '\n' + findings.map((f) => ` • ${f}`).join('\n')
|
|
113
|
+
: (reason ? `\n ${reason}` : '')
|
|
114
|
+
return `${head}${body}\n\nYou two decide — do you agree? Challenge it, or accept.`
|
|
115
|
+
}
|
package/flow-review.mjs
CHANGED
|
@@ -12,6 +12,7 @@
|
|
|
12
12
|
import { execFileSync } from 'node:child_process'
|
|
13
13
|
import { TASK_STATUS } from './flow-task-graph.mjs'
|
|
14
14
|
import { removeFlowWorktree } from './flow-worktree.mjs'
|
|
15
|
+
import { reviewReflectionDecision, REVIEW_DEFAULTS } from './flow-review-reflect.mjs'
|
|
15
16
|
|
|
16
17
|
// The reviewer lane's rolePrompt (via startClaudeSession). Mirrors FLOW_LANE_PROMPT's
|
|
17
18
|
// join(' ') style. This lane is ADVERSARIAL — its job is to disprove "done", not to build.
|
|
@@ -26,7 +27,9 @@ export const FLOW_REVIEWER_PROMPT = [
|
|
|
26
27
|
|
|
27
28
|
'BE SPECIFIC. Your reasons must name exactly WHAT failed and HOW you found it — the command you ran, the output you got, the acceptance criterion it violated. "Doesn\'t work" is useless. "GET /api/todos returned 500 with `column todos.user_id does not exist`; acceptance required 200 + []" is a usable verdict.',
|
|
28
29
|
|
|
29
|
-
'
|
|
30
|
+
'YOU GET BOUNDED HUNT ROUNDS, NOT ONE GLANCE. You may be asked to dig again — up to a hard ceiling of rounds. A green happy-path on round 1 is NOT a pass; use the next round to hunt harder (edge cases, the reload, the second click, concurrent use, the error path). You only stop early two ways: you reproduce a concrete FAILURE (emit pass:false — that ends it, the slice reverts), OR you have genuinely EXHAUSTED your checks and found nothing (emit pass:true AND exhausted:true — that ends it, the slice passes). If you still have angles left to try, emit pass:true and leave exhausted false/absent — you will be asked to dig one more round until the ceiling, at which point the inconclusive result is handed to the two humans to decide (never an unbounded self-loop).',
|
|
31
|
+
|
|
32
|
+
'EMIT YOUR VERDICT BY WRITING FLOW_REVIEW.json. Do NOT call mark_flow_done or ExitPlanMode (they hang here). As your LAST action, use the Write tool with file_path "FLOW_REVIEW.json" and content = a single JSON object { "pass": <boolean>, "reasons": ["<specific finding>", ...], "taskKey": "<the slice you reviewed>", "exhausted": <boolean, optional — true only when you have nothing left to check> }. `pass: false` reverts the reviewed slice so it rebuilds. The content of FLOW_REVIEW.json must be ONLY that JSON object (no prose, no markdown fences). Do not edit the slice — you review, you do not fix.',
|
|
30
33
|
].join(' ')
|
|
31
34
|
|
|
32
35
|
// Parse a reviewer's raw verdict. Accepts an object OR a JSON string (optionally
|
|
@@ -43,7 +46,57 @@ export function parseReviewVerdict (raw) {
|
|
|
43
46
|
if (!obj || typeof obj !== 'object') throw new Error('review verdict is not an object')
|
|
44
47
|
if (typeof obj.pass !== 'boolean') throw new Error('review verdict missing boolean `pass`')
|
|
45
48
|
const reasons = Array.isArray(obj.reasons) ? obj.reasons.map(String) : []
|
|
46
|
-
|
|
49
|
+
// E1 (bounded self-test): `exhausted` is an OPTIONAL third signal — true only when the
|
|
50
|
+
// reviewer reports it has nothing left to check. Absent/non-boolean → false (keep hunting
|
|
51
|
+
// until the round ceiling). Never throws on its absence; it's additive to the old contract.
|
|
52
|
+
const exhausted = obj.exhausted === true
|
|
53
|
+
return { pass: obj.pass, reasons, exhausted }
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
// E1 — map a parsed review verdict into the two governor inputs. A concrete FAILURE
|
|
57
|
+
// (pass:false) is a reproduced break → foundFailure. A pass is only conclusive when the
|
|
58
|
+
// reviewer says it EXHAUSTED its checks; a bare pass:true means "happy path held, more to
|
|
59
|
+
// hunt" → keep going until the round ceiling. Pure; the seam the loop drives on.
|
|
60
|
+
export function reviewVerdictToReflection (verdict) {
|
|
61
|
+
const v = verdict || {}
|
|
62
|
+
return {
|
|
63
|
+
foundFailure: v.pass === false,
|
|
64
|
+
exhausted: v.pass === true && v.exhausted === true,
|
|
65
|
+
}
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
// E1+A1/A2 — the BOUNDED reviewer loop. Drives the adversarial reviewer across up to
|
|
69
|
+
// `maxRounds` hunt rounds, consulting reviewReflectionDecision between rounds, and stops at
|
|
70
|
+
// the FIRST terminal outcome (reject on a concrete failure, pass on exhaustion, or SURFACE
|
|
71
|
+
// when a ceiling — rounds or budget — is hit inconclusively). Pure over two injected effects
|
|
72
|
+
// so it's unit-testable without a live Flow room (the S5 split):
|
|
73
|
+
// • runRound(round) → Promise<parsed verdict {pass,reasons,exhausted}> — spend one round
|
|
74
|
+
// • getSpent() → number — the review lane's tokens spent so far (flow-budget ledger)
|
|
75
|
+
// N1 fail-closed: the budget is checked BEFORE each round is spent, so a review lane at/over
|
|
76
|
+
// its cap SURFACES without ever spending the next round (never runs the bill up, anti-Emergent).
|
|
77
|
+
// Returns { action, surfaceToPair, reason, rounds, verdict } — `rounds` = rounds actually run.
|
|
78
|
+
export async function runBoundedReview ({ maxRounds, budgetCap = null, getSpent = () => 0, runRound }) {
|
|
79
|
+
if (typeof runRound !== 'function') throw new Error('runBoundedReview needs a runRound(round) function')
|
|
80
|
+
const cap = Number.isFinite(maxRounds) && maxRounds > 0 ? maxRounds : REVIEW_DEFAULTS.maxRounds
|
|
81
|
+
const capped = budgetCap != null && Number.isFinite(budgetCap)
|
|
82
|
+
for (let round = 1; ; round++) {
|
|
83
|
+
// N1 fail-closed pre-gate: never SPEND another review round once the cap is reached.
|
|
84
|
+
const spentBefore = getSpent()
|
|
85
|
+
if (capped && spentBefore >= budgetCap) {
|
|
86
|
+
const d = reviewReflectionDecision({
|
|
87
|
+
round, maxRounds: cap, spentTokens: spentBefore, budgetCap,
|
|
88
|
+
foundFailure: false, exhausted: false,
|
|
89
|
+
})
|
|
90
|
+
return { ...d, rounds: round - 1, verdict: null }
|
|
91
|
+
}
|
|
92
|
+
const verdict = await runRound(round)
|
|
93
|
+
const decision = reviewReflectionDecision({
|
|
94
|
+
round, maxRounds: cap, spentTokens: getSpent(), budgetCap,
|
|
95
|
+
...reviewVerdictToReflection(verdict),
|
|
96
|
+
})
|
|
97
|
+
if (decision.stop) return { ...decision, rounds: round, verdict }
|
|
98
|
+
// else action==='continue' — under both ceilings, not yet concluded → dig another round.
|
|
99
|
+
}
|
|
47
100
|
}
|
|
48
101
|
|
|
49
102
|
// Gate decision over a task's verdicts (≥1 reviewer). Accept iff EVERY verdict passes;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "thinkpool-pair",
|
|
3
|
-
"version": "0.7.
|
|
3
|
+
"version": "0.7.156",
|
|
4
4
|
"description": "Share a local coding-agent CLI (Claude Code, Codex, Gemini, Aider, …) into a ThinkPool Code room, live.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
"sdk-smoke.mjs",
|
|
12
12
|
"launcher.mjs",
|
|
13
13
|
"byok-detect.mjs",
|
|
14
|
+
"context-windows.mjs",
|
|
14
15
|
"claude-session.mjs",
|
|
15
16
|
"update-gate.mjs",
|
|
16
17
|
"event-id.mjs",
|
|
@@ -23,6 +24,7 @@
|
|
|
23
24
|
"flow-preview.mjs",
|
|
24
25
|
"flow-review.mjs",
|
|
25
26
|
"flow-review-gate.mjs",
|
|
27
|
+
"flow-review-reflect.mjs",
|
|
26
28
|
"flow-assembly.mjs",
|
|
27
29
|
"flow-budget.mjs",
|
|
28
30
|
"flow-context-store.mjs",
|
package/session-store.mjs
CHANGED
|
@@ -215,14 +215,3 @@ export function loadNames(room) {
|
|
|
215
215
|
export function saveNames(room, names) {
|
|
216
216
|
try { ensureDir(room); fs.writeFileSync(namesFile(room), JSON.stringify(names || {})) } catch { /* noop */ }
|
|
217
217
|
}
|
|
218
|
-
|
|
219
|
-
// Session recap cache (2026-07-02, Session Info panel). The last on-demand summary,
|
|
220
|
-
// stored as a dotfile (NOT *.json → never in loadAll) so it survives a bridge restart
|
|
221
|
-
// and any device that opens the panel can fetch it (persists across devices). One per room.
|
|
222
|
-
const summaryFile = (room) => path.join(dir(room), '.summary')
|
|
223
|
-
export function loadSummary(room) {
|
|
224
|
-
try { return JSON.parse(fs.readFileSync(summaryFile(room), 'utf8')) || null } catch { return null }
|
|
225
|
-
}
|
|
226
|
-
export function saveSummary(room, summary) {
|
|
227
|
-
try { ensureDir(room); fs.writeFileSync(summaryFile(room), JSON.stringify(summary || null)) } catch { /* noop */ }
|
|
228
|
-
}
|