thinkpool-pair 0.7.153 → 0.7.155

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bridge.mjs CHANGED
@@ -57,8 +57,9 @@ function stopFlowPreviews (flowId, laneId = null) {
57
57
  if (laneId ? key === `lane:${flowId}:${laneId}` : key.startsWith(`lane:${flowId}:`)) { try { h.stop() } catch { /* noop */ } }
58
58
  }
59
59
  }
60
- import { FLOW_REVIEWER_PROMPT, revertLane, parseReviewVerdict } from './flow-review.mjs'
60
+ import { FLOW_REVIEWER_PROMPT, revertLane, parseReviewVerdict, reviewVerdictToReflection } from './flow-review.mjs'
61
61
  import { reviewGateDecision } from './flow-review-gate.mjs'
62
+ import { pairAdjudicationPrompt, reviewReflectionDecision, REVIEW_DEFAULTS } from './flow-review-reflect.mjs'
62
63
  import { mergeWorktrees, inlineSingleHtml, initRepo } from './flow-assembly.mjs'
63
64
  import { canDispatch, FLOW_LIMITS, makeBudget, recordSpend, killSwitchEnv } from './flow-budget.mjs'
64
65
  // S4 slice 2 — clean re-dispatch. When a lane is killed mid-tool-call (budget/review/restart)
@@ -81,10 +82,10 @@ const flowRedispatch = new Map()
81
82
  // wave BEFORE overrun. Lives bridge-side because waves dispatch across separate
82
83
  // broadcasts; without persistent state the cap can never bite.
83
84
  const flowBudgets = new Map()
84
- import { formatPeek, PEEK, siblingsOf, resolveSibling, crossPostDecision, CROSSPOST, spawnDecision, CROSSROOM, formatPairRoster, crossRoomPostDecision, formatRoomNow } from './cross-terminal.mjs'
85
+ import { formatPeek, PEEK, siblingsOf, resolveSibling, crossPostDecision, CROSSPOST, spawnDecision, SPAWN, CROSSROOM, formatPairRoster, crossRoomPostDecision, formatRoomNow } from './cross-terminal.mjs'
85
86
  import { turnInFlight } from './update-gate.mjs'
86
87
  import { saveSession, flushSession, deleteSession, loadAll, canResume, loadPtyId, savePtyId, loadNames, saveNames, appendDurableEvents, seedDurableEvents, readDurablePage, readDurableOldestSeq, loadSummary, saveSummary } from './session-store.mjs'
87
- import { stampEvent, makeSeqCounter, maxSeq, seqable, capReplayEvents, chunkReplayEvents, boundEventForBroadcast, inlineImageBlocks } from './event-id.mjs'
88
+ import { stampEvent, makeSeqCounter, maxSeq, seqable, capReplayEvents, chunkReplayEvents, boundEventForBroadcast, inlineImageBlocks, usageReportLine } from './event-id.mjs'
88
89
  import { makeThrottledTrack } from './presence.mjs'
89
90
 
90
91
  // Public client creds (the same anon values the web app ships — safe to embed).
@@ -1376,14 +1377,71 @@ function openStructured({ id, model, resume, log, commands, mode, spawnedBy, rol
1376
1377
  } catch (e) {
1377
1378
  return { ok: false, message: `Review verdict REJECTED: ${e?.message || e}. Re-Write FLOW_REVIEW.json with {"pass":<boolean>,"reasons":["<specific finding>"],"taskKey":"<the slice you reviewed>"}.` }
1378
1379
  }
1379
- if (!v.pass && target) {
1380
+ // E1 A1/A2 — the BOUNDED reviewer loop, live side. Each FLOW_REVIEW.json write is ONE
1381
+ // hunt round; the governor decides continue-vs-stop from the round count + the lane's
1382
+ // REAL budget (flowBudgets ledger). A bare pass:true (happy path held, not yet exhausted)
1383
+ // under both ceilings → dig ONE more round instead of concluding; a concrete failure, an
1384
+ // exhausted pass, or a ceiling (rounds OR budget) → terminal, surfaced to the pair (P3).
1385
+ // Fail-closed: a lane at/over its cap concludes (surface) and is never re-prompted. Held
1386
+ // for live-room verification (bridge is inert until published); the decision logic is
1387
+ // unit-proven in flow-review.loop.test.mjs + flow-review-reflect.test.mjs.
1388
+ const round = (entry.flowReviewRound = (entry.flowReviewRound || 0) + 1)
1389
+ const b = flowBudgets.get(entry.flowSessionId) || null
1390
+ const decision = reviewReflectionDecision({
1391
+ round,
1392
+ maxRounds: REVIEW_DEFAULTS.maxRounds,
1393
+ spentTokens: b ? b.spentTokens : 0,
1394
+ budgetCap: b ? b.capTokens : null,
1395
+ ...reviewVerdictToReflection(v),
1396
+ })
1397
+
1398
+ // Not concluded, under both ceilings → send the reviewer back for another bounded round.
1399
+ // No revert, no verdict broadcast, no markFlowDone — the loop stays open.
1400
+ if (!decision.stop) {
1401
+ const next = round + 1
1402
+ try {
1403
+ entry.session?.sendTurn(
1404
+ `[Flow review — round ${next}/${REVIEW_DEFAULTS.maxRounds}] The happy path held, but you have NOT reported your checks exhausted and you are under both the round and budget ceilings. Dig one more round: hunt the edge cases, the reload, the second click, concurrent use, the error path — the places the builder didn't. Then re-Write FLOW_REVIEW.json — pass:false with a specific reason if you break it, or pass:true AND exhausted:true if you genuinely have nothing left to check.`,
1405
+ )
1406
+ } catch { /* lane may have closed mid-verdict */ }
1407
+ process.stderr.write(`\n ${A.dim}◆ review round ${round} inconclusive — digging again (${next}/${REVIEW_DEFAULTS.maxRounds}) on ${target || entry.flowTaskKey}${A.rst}\n`)
1408
+ return { ok: true, message: `Round ${round} recorded (pass, not yet exhausted). Keep hunting — asked you for round ${next}/${REVIEW_DEFAULTS.maxRounds}.` }
1409
+ }
1410
+
1411
+ // Terminal outcome. reject → revert the reviewed slice; surface → hand the inconclusive
1412
+ // result to the pair WITHOUT reverting (no failure was reproduced); pass → accept.
1413
+ if (decision.action === 'reject' && target) {
1380
1414
  bcast('flow-revert', { term: id, flowId: entry.flowSessionId, taskKey: target }, flowChannel)
1381
- process.stderr.write(`\n ${A.yel}◆ review FAIL — reverting ${target}: ${v.reasons.join('; ').slice(0, 120)}${A.rst}\n`)
1415
+ process.stderr.write(`\n ${A.yel}◆ review REJECT (round ${round}) — reverting ${target}: ${v.reasons.join('; ').slice(0, 120)}${A.rst}\n`)
1382
1416
  } else {
1383
- process.stderr.write(`\n ${A.cyan}◆ review PASS — ${target || entry.flowTaskKey}${A.rst}\n`)
1417
+ process.stderr.write(`\n ${A.cyan}◆ review ${decision.action.toUpperCase()} (round ${round}) — ${target || entry.flowTaskKey}${A.rst}\n`)
1384
1418
  }
1419
+ // P3 (pair co-adjudication) — surface the TERMINAL verdict into the ROOM as a
1420
+ // challengeable prompt, ALONGSIDE the gate action above. A solo Bugbot's verdict is
1421
+ // final; ours is a prompt for the two humans + Pool to argue. Additive: an extra
1422
+ // broadcast, does NOT change the revert/gate flow.
1423
+ bcast('flow-review-verdict', {
1424
+ term: id, flowId: entry.flowSessionId,
1425
+ taskKey: target || entry.flowTaskKey,
1426
+ pass: v.pass,
1427
+ reasons: v.reasons,
1428
+ action: decision.action,
1429
+ rounds: round,
1430
+ surfaceToPair: decision.surfaceToPair,
1431
+ prompt: pairAdjudicationPrompt({
1432
+ action: decision.action,
1433
+ reason: decision.reason,
1434
+ taskKey: target || entry.flowTaskKey,
1435
+ findings: v.reasons,
1436
+ }),
1437
+ }, flowChannel)
1385
1438
  const doneMsg = await markFlowDone()
1386
- return { ok: true, message: `Review verdict recorded: ${v.pass ? 'PASS' : `FAIL — reverting ${target || '(no target)'} (${v.reasons.join('; ').slice(0, 160)})`}. ${doneMsg}` }
1439
+ const label = decision.action === 'reject'
1440
+ ? `REJECT (round ${round}) — reverting ${target || '(no target)'} (${v.reasons.join('; ').slice(0, 140)})`
1441
+ : decision.action === 'surface'
1442
+ ? `SURFACED to the pair after ${round} round${round === 1 ? '' : 's'} — ${decision.reason}`
1443
+ : `PASS (round ${round})`
1444
+ return { ok: true, message: `Review verdict recorded: ${label}. ${doneMsg}` }
1387
1445
  }
1388
1446
  // Identity for the durable archive — pushLog appends every new transcript event to
1389
1447
  // <room>/<id>.events.jsonl keyed off these. Seed the archive once from the restored
@@ -1593,16 +1651,20 @@ function openStructured({ id, model, resume, log, commands, mode, spawnedBy, rol
1593
1651
  },
1594
1652
  async (args) => {
1595
1653
  const okText = (t) => ({ content: [{ type: 'text', text: t }] })
1654
+ const now = Date.now()
1596
1655
  const gate = spawnDecision({
1597
1656
  hop: entry.hop || 0,
1598
- spawnCount: entry.spawnCount || 0,
1657
+ spawnTimes: entry.spawnTimes || [], // sliding-window burst breaker (replenishes on wall-clock, not on a human turn)
1658
+ now,
1599
1659
  spawnedLive: [...sessions.values()].filter((s) => s.spawnedBy).length, // dispatch budget counts ONLY spawned lanes
1600
1660
  totalLive: sessions.size + terms.size, // machine cap counts everything
1601
1661
  plan: ownerPlan, // Free 3 / Plus 6 dispatch ceiling
1602
1662
  disabled: process.env.TP_SPAWN_OFF === '1',
1603
1663
  })
1604
1664
  if (!gate.ok) return okText(gate.reason)
1605
- entry.spawnCount = (entry.spawnCount || 0) + 1
1665
+ // Record this spawn, pruning timestamps that have already left the window
1666
+ // (identical predicate to recentSpawnCount) so the array can't grow unbounded.
1667
+ entry.spawnTimes = [...(entry.spawnTimes || []).filter((t) => Number.isFinite(t) && t > now - SPAWN.windowMs), now]
1606
1668
  const newId = randomUUID()
1607
1669
  const newRef = String(newId).slice(0, 8)
1608
1670
  const fromRef = String(id).slice(0, 8)
@@ -1623,7 +1685,7 @@ function openStructured({ id, model, resume, log, commands, mode, spawnedBy, rol
1623
1685
  if (args?.task) {
1624
1686
  // One hop deep: the spawned lane cannot spawn/post onward until a person
1625
1687
  // speaks to it (code-turn resets hop to 0). Mirrors the Tier C injection.
1626
- ne.hop = 1; ne.peekCount = 0; ne.postCount = 0; ne.spawnCount = 0
1688
+ ne.hop = 1; ne.peekCount = 0; ne.postCount = 0; ne.spawnTimes = []
1627
1689
  const msg = `[Task from terminal ${fromRef}'s agent — relayed via ThinkPool cross-terminal; you are a fresh lane it opened for you]\n${args.task}`
1628
1690
  const evt = { kind: 'you', text: msg, by: `terminal ${fromRef} (agent)`, crosspost: true }
1629
1691
  stampEvent(evt); pushLog(ne, evt); bcast('code-event', { term: newId, evt })
@@ -2266,7 +2328,7 @@ channel
2266
2328
  s.postCount = 0
2267
2329
  s.pairPeekCount = 0 // cross-room read budget resets with the in-room ones (Tier 1)
2268
2330
  s.crossRoomPostCount = 0 // Tier 3 cross-room post budget resets on a real human turn
2269
- s.spawnCount = 0 // EN-M3 — spawn budget MUST reset too, else turn 2+ can never spawn a lane
2331
+ s.spawnTimes = [] // EN-M3 — a real human turn clears the spawn burst window instantly (the window also self-replenishes on wall-clock for UNATTENDED cascades that never reach this handler)
2270
2332
  s.hop = 0
2271
2333
  s.roomHop = 0 // a human turn is room-hop 0 — clears any injected cross-room hop depth
2272
2334
  const text = String(payload.text)
@@ -2330,6 +2392,13 @@ channel
2330
2392
  s.session.sendTurn(text)
2331
2393
  return
2332
2394
  }
2395
+ // /usage → report this session's token + cost totals WITHOUT burning an agent turn.
2396
+ // Computed by REPLAYING the session's own retained result events (each completed turn
2397
+ // persists kind:'result' with usage + costUsd + model — see onEvent). No new accumulator
2398
+ // state: the log is durable, so this survives a bridge restart for free. Reported as ◆
2399
+ // control lines (like /model) so both members AND late-joiners see it. Scope is honest —
2400
+ // the Agent SDK exposes tokens + cost, not plan/rate-limit meters, and the card says so.
2401
+ if (/^\/usage\s*$/.test(text)) { ctlLine(usageReportLine(s.model, s.log)); return }
2333
2402
  // Pre-flight budget gate (room scope) — refuse the billable agent turn if the room hit
2334
2403
  // its cap. The slash controls above (/model, /clear, /compact) are intentionally exempt:
2335
2404
  // they let a user manage or REDUCE spend even when capped. Legible refusal, no silent stall.
@@ -227,15 +227,25 @@ export const crossPostNeedsCard = ({ mode = 'default', spawnedByMe = false } = {
227
227
  // own terminals, and the human's terminals never eat the dispatch budget. Under
228
228
  // BYOK-only the tokens are the user's, so this is feature-shaping + machine
229
229
  // safety, NOT a ThinkPool-COGS control.
230
- // • perTurnCap — a per-turn runaway breaker (mirrors PEEK/CROSSPOST)
230
+ // • burst window — a SLIDING-window runaway breaker: ≤ perWindowCap spawns per
231
+ // rolling windowMs. Replaces the old per-HUMAN-turn cap, which starved an
232
+ // UNATTENDED conductor: agent turns woken by Monitor/task notifications never
233
+ // traverse the 'code-turn' handler that reset the per-turn counter, so a cascade
234
+ // that spent its 4 spawns could NEVER spawn again until a person typed. The
235
+ // window replenishes on wall-clock alone — no turn boundary required — while a
236
+ // real human turn still clears it instantly (the reset point is unchanged).
231
237
  // The caller measures: spawnedLive (live sessions with spawnedBy set), totalLive
232
- // (sessions.size + terms.size), and resolves `plan` from the room owner.
238
+ // (sessions.size + terms.size), resolves `plan` from the room owner, and passes the
239
+ // lane's recent spawn timestamps (`spawnTimes`) plus the current clock (`now`) — so
240
+ // this stays a PURE function (no Date.now inside; the clock is injected, like
241
+ // spawnCount was before).
233
242
  export const SPAWN = {
234
- maxHop: 1, // human turn = hop 0; a spawned/injected lane = hop ≥ 1, CANNOT spawn onward
235
- perTurnCap: 4, // spawn_terminal calls allowed per turn
236
- freeMaxSpawned: 3, // Free plan dispatch ceiling (concurrent spawned lanes)
237
- plusMaxSpawned: 6, // Plus plan dispatch ceiling
238
- machineMax: 10, // hard cap on TOTAL live lanes (the laptop/runaway backstop)
243
+ maxHop: 1, // human turn = hop 0; a spawned/injected lane = hop ≥ 1, CANNOT spawn onward
244
+ perWindowCap: 4, // spawn_terminal calls allowed per rolling window (burst breaker)
245
+ windowMs: 10 * 60 * 1000, // sliding window over which perWindowCap applies (10 min)
246
+ freeMaxSpawned: 3, // Free plan dispatch ceiling (concurrent spawned lanes)
247
+ plusMaxSpawned: 6, // Plus plan dispatch ceiling
248
+ machineMax: 10, // hard cap on TOTAL live lanes (the laptop/runaway backstop)
239
249
  }
240
250
 
241
251
  // Plan → dispatch ceiling. Anything not 'plus' is treated as Free (the safe default).
@@ -244,14 +254,24 @@ export const maxSpawnedFor = (plan, limits = SPAWN) => {
244
254
  return plan === 'plus' ? o.plusMaxSpawned : o.freeMaxSpawned
245
255
  }
246
256
 
247
- export const spawnDecision = ({ hop = 0, spawnCount = 0, spawnedLive = 0, totalLive = 0, plan = 'free', disabled = false } = {}, limits = SPAWN) => {
257
+ // Count spawn timestamps still inside the sliding window ending at `now`. Timestamps
258
+ // strictly older than `now - windowMs` have replenished (fallen out of the window).
259
+ // Non-finite entries are ignored defensively. Exported so the bridge can prune its
260
+ // stored spawnTimes with the identical predicate (no unbounded growth).
261
+ export const recentSpawnCount = (spawnTimes, now, windowMs = SPAWN.windowMs) =>
262
+ (Array.isArray(spawnTimes) ? spawnTimes : []).filter((t) => Number.isFinite(t) && t > now - windowMs).length
263
+
264
+ export const spawnDecision = ({ hop = 0, spawnTimes = [], now = 0, spawnedLive = 0, totalLive = 0, plan = 'free', disabled = false } = {}, limits = SPAWN) => {
248
265
  const o = { ...SPAWN, ...(limits || {}) }
249
266
  if (disabled) return { ok: false, reason: 'Spawning terminals is disabled in this room.' }
250
267
  if (hop >= o.maxHop) return { ok: false, reason: 'Spawn limit reached — a terminal reached via cross-terminal cannot itself open further terminals; a person must open the next lane.' }
251
268
  if (totalLive >= o.machineMax) return { ok: false, reason: `Room is at the machine cap (${o.machineMax} live lanes) — close one before opening another.` }
252
269
  const cap = maxSpawnedFor(plan, o)
253
270
  if (spawnedLive >= cap) return { ok: false, reason: `Dispatch ceiling reached (${cap} lane${cap === 1 ? '' : 's'} on the ${plan} plan)${plan !== 'plus' ? ` — Plus raises it to ${o.plusMaxSpawned}.` : '.'} Close a dispatched lane to open another.` }
254
- if (spawnCount >= o.perTurnCap) return { ok: false, reason: `Spawn limit reached for this turn (${o.perTurnCap}).` }
271
+ if (recentSpawnCount(spawnTimes, now, o.windowMs) >= o.perWindowCap) {
272
+ const mins = Math.round(o.windowMs / 60000)
273
+ return { ok: false, reason: `Spawn burst limit reached (${o.perWindowCap} in ${mins} min) — this replenishes as the window slides forward, or instantly when a person types into this lane.` }
274
+ }
255
275
  return { ok: true }
256
276
  }
257
277
 
package/event-id.mjs CHANGED
@@ -191,3 +191,60 @@ export function inlineImageBlocks(evt) {
191
191
  })
192
192
  return out
193
193
  }
194
+
195
+ // Pure aggregation for the /usage report — replay a session's retained log events and
196
+ // fold every completed turn (kind:'result' with usage) into per-model + total token/cost
197
+ // sums. No side effects, import-safe (bridge/test-usage-summary.mjs exercises it), so the
198
+ // bridge computes /usage by replaying its own durable log — no separate accumulator that a
199
+ // restart would drop. `costUsd` is the SDK's total_cost_usd per turn (BYOK estimate).
200
+ export function summarizeUsage(logEvents) {
201
+ const results = (Array.isArray(logEvents) ? logEvents : []).filter((e) => e && e.kind === 'result' && e.usage)
202
+ const per = new Map()
203
+ let totalIn = 0, totalOut = 0, totalCost = 0, haveCost = false
204
+ for (const e of results) {
205
+ const u = e.usage || {}
206
+ const inTok = (u.input_tokens || 0) + (u.cache_creation_input_tokens || 0) + (u.cache_read_input_tokens || 0)
207
+ const out = u.output_tokens || 0
208
+ const cached = u.cache_read_input_tokens || 0
209
+ const mid = e.model || 'unknown'
210
+ const r = per.get(mid) || { model: mid, in: 0, out: 0, cached: 0 }
211
+ r.in += inTok; r.out += out; r.cached += cached; per.set(mid, r)
212
+ totalIn += inTok; totalOut += out
213
+ if (typeof e.costUsd === 'number') { totalCost += e.costUsd; haveCost = true }
214
+ }
215
+ return { turns: results.length, per: [...per.values()], totalIn, totalOut, totalCost, haveCost }
216
+ }
217
+
218
+ // model id → friendly label for the /usage report: claude-opus-4-8 → "Opus 4.8".
219
+ // Bridge-local twin of the web's prettyModel (structured.jsx); falls back to the raw id.
220
+ export function prettyModelId (m) {
221
+ if (!m) return 'unknown'
222
+ const mm = String(m).match(/(opus|sonnet|haiku|fable)-(\d+)(?:-(\d+))?/i)
223
+ return mm ? `${mm[1][0].toUpperCase()}${mm[1].slice(1)} ${mm[2]}${mm[3] ? '.' + mm[3] : ''}` : String(m)
224
+ }
225
+
226
+ // Build the ONE ◆ control line the /usage command emits, from the session's model +
227
+ // retained log. Pure so the whole command — gate, empty-state, and per-model/total/cost
228
+ // formatting — is exercised deterministically (bridge/test-usage-summary.mjs). The bridge
229
+ // handler is a thin wrapper: ctlLine(usageReportLine(s.model, s.log)). `sessionModel` is
230
+ // the resolved model id the /model state tracks; `logEvents` the session's retained log.
231
+ export function usageReportLine (sessionModel, logEvents) {
232
+ const claudeish = (m) => /claude|anthropic|opus|sonnet|haiku|fable/i.test(String(m || ''))
233
+ const sum = summarizeUsage(logEvents)
234
+ // Gate (Max's ask: "make /usage work if users bridge is on claude llms"). The Code bridge
235
+ // runs the Claude Agent SDK, so a normal lane IS a Claude model — but a session pointed at a
236
+ // non-Claude base URL/model gets a legible line, not a bogus token report. Only assert this
237
+ // when we actually know a model (else a fresh empty session would wrongly trip the gate).
238
+ const anyClaude = claudeish(sessionModel) || sum.per.some((r) => claudeish(r.model))
239
+ if (!anyClaude && (sessionModel || sum.turns)) return 'usage reporting is available on Claude models only'
240
+ if (!sum.turns) return 'usage · no completed turns yet this session'
241
+ const fmtTok = (n) => n >= 1e6 ? (n / 1e6).toFixed(1) + 'M' : n >= 1e3 ? (n / 1e3).toFixed(1) + 'k' : String(n)
242
+ const perStr = sum.per.map((r) => {
243
+ const cch = r.cached ? ` (${fmtTok(r.cached)} cached)` : ''
244
+ return `${prettyModelId(r.model)}: ${fmtTok(r.in)} in${cch} / ${fmtTok(r.out)} out`
245
+ }).join(' · ')
246
+ const total = sum.per.length > 1 ? ` · total ${fmtTok(sum.totalIn)} in / ${fmtTok(sum.totalOut)} out` : ''
247
+ const cost = sum.haveCost ? ` · ~$${sum.totalCost.toFixed(2)} est` : ''
248
+ const turns = `${sum.turns} turn${sum.turns === 1 ? '' : 's'}`
249
+ return `usage · ${turns} · ${perStr}${total}${cost} · tokens + cost only (no plan/rate-limit meters)`
250
+ }
@@ -0,0 +1,115 @@
1
+ /* ─────────────────────────────────────────────────────────────
2
+ flow-review-reflect.mjs — the BOUNDED reflection governor for the
3
+ verifying reviewer lane (heist E1 + P3).
4
+ Spec: docs/specs/2026-07-02-e1p3-verifying-reviewer.md
5
+ Parents: Emergent steal E1 (self-test loop) + Cursor steal P3 (Bugbot).
6
+
7
+ The existing adversarial reviewer (bridge/flow-review.mjs) runs the slice
8
+ once and emits a verdict. E1 makes that a bounded, budget-capped SELF-TEST
9
+ loop: the reviewer may dig across several rounds (re-run, hit edge cases,
10
+ reproduce the acceptance proof harder) — but it MUST stop at a hard ceiling
11
+ and hand off to the humans, never loop unbounded. That unbounded self-heal
12
+ loop is exactly Emergent's most-hated wound (billing a runaway fix-loop back
13
+ to the user); this governor is the structural guarantee we never have it.
14
+
15
+ Two independent stop conditions, both enforced here (not prompted):
16
+ • ROUND ceiling — never more than maxRounds hunt rounds (anti-Emergent).
17
+ • BUDGET ceiling — never spend past the room/lane cap (nexos steal #1 /
18
+ cost-guard; the review lane SPENDS inference, so it inherits N1).
19
+
20
+ Pure + deterministic — the whole loop-control decision in one function so it
21
+ is unit-testable without a live Flow room (the reviewable seam; the live
22
+ wiring into flow-review.mjs + bridge.mjs is the held integration slice,
23
+ same split that shipped S5's reviewGateDecision).
24
+
25
+ P3 (pair co-adjudication): when the loop stops, the outcome is not the end —
26
+ `surfaceToPair` says the verdict/findings must go to the ROOM for the two
27
+ humans + Pool to challenge, rather than silently auto-driving the gate. A
28
+ solo Bugbot's verdict is final; ours is a prompt for the pair to argue.
29
+ ───────────────────────────────────────────────────────────── */
30
+
31
+ export const REVIEW_DEFAULTS = Object.freeze({
32
+ maxRounds: 3, // hard hunt ceiling — bounded, never unbounded (anti-Emergent)
33
+ })
34
+
35
+ /**
36
+ * Decide whether the reviewer's self-test loop continues or stops, and why.
37
+ *
38
+ * @param {object} s
39
+ * @param {number} s.round 1-based current hunt round (1 = first pass)
40
+ * @param {number} [s.maxRounds] hard ceiling (default REVIEW_DEFAULTS.maxRounds)
41
+ * @param {number} [s.spentTokens] tokens the review lane has spent so far
42
+ * @param {number|null} [s.budgetCap] the room/lane token cap (null = uncapped)
43
+ * @param {boolean} s.foundFailure this round produced a CONCRETE, reproduced failure
44
+ * @param {boolean} [s.exhausted] the reviewer reports it has nothing left to check
45
+ * @returns {{ stop: boolean, action: 'reject'|'pass'|'surface'|'continue',
46
+ * surfaceToPair: boolean, reason: string }}
47
+ * action:
48
+ * 'reject' — a concrete failure was found; stop, verdict = REJECT
49
+ * 'pass' — checks exhausted with no failure; stop, verdict = PASS
50
+ * 'surface' — a ceiling (budget or rounds) was hit inconclusively; stop,
51
+ * hand the partial findings to the humans (no guess, no loop)
52
+ * 'continue' — keep hunting (still under both ceilings, not yet concluded)
53
+ * surfaceToPair — true whenever the loop STOPS (P3: every terminal outcome is
54
+ * a prompt for the pair, not an automatic gate action).
55
+ */
56
+ export function reviewReflectionDecision ({
57
+ round,
58
+ maxRounds = REVIEW_DEFAULTS.maxRounds,
59
+ spentTokens = 0,
60
+ budgetCap = null,
61
+ foundFailure = false,
62
+ exhausted = false,
63
+ } = {}) {
64
+ const r = Number.isFinite(round) ? round : 1
65
+ const cap = Number.isFinite(maxRounds) && maxRounds > 0 ? maxRounds : REVIEW_DEFAULTS.maxRounds
66
+
67
+ // 1. A concrete, reproduced failure ends the hunt immediately — default-to-
68
+ // reject is the whole point; no value in more rounds once it's broken.
69
+ if (foundFailure) {
70
+ return { stop: true, action: 'reject', surfaceToPair: true,
71
+ reason: 'concrete failure reproduced — verdict REJECT' }
72
+ }
73
+
74
+ // 2. BUDGET ceiling (nexos #1 / cost-guard). Fail-closed: the moment spend
75
+ // reaches the cap, STOP and hand off — never spend the next unit. Checked
76
+ // before the round ceiling because money is the harder limit.
77
+ if (budgetCap != null && Number.isFinite(budgetCap) && spentTokens >= budgetCap) {
78
+ return { stop: true, action: 'surface', surfaceToPair: true,
79
+ reason: `review budget cap reached (${spentTokens}/${budgetCap} tokens) — surfacing partial findings to the pair` }
80
+ }
81
+
82
+ // 3. Reviewer says it has checked everything and found nothing — verdict PASS.
83
+ if (exhausted) {
84
+ return { stop: true, action: 'pass', surfaceToPair: true,
85
+ reason: 'checks exhausted, no failure reproduced — verdict PASS' }
86
+ }
87
+
88
+ // 4. ROUND ceiling (anti-Emergent). Reached the last allowed round without a
89
+ // conclusion → surface to the humans rather than guess or loop forever.
90
+ if (r >= cap) {
91
+ return { stop: true, action: 'surface', surfaceToPair: true,
92
+ reason: `reached the ${cap}-round hunt ceiling without a conclusion — surfacing to the pair (not looping)` }
93
+ }
94
+
95
+ // 5. Under both ceilings, not yet concluded → dig another round.
96
+ return { stop: false, action: 'continue', surfaceToPair: false,
97
+ reason: `round ${r}/${cap}, under budget — continue hunting` }
98
+ }
99
+
100
+ /**
101
+ * Format a stopped review outcome as the room-facing co-adjudication prompt
102
+ * (P3). Not the gate action itself — a message that invites the two humans +
103
+ * Pool to challenge the reviewer before the verdict is treated as final.
104
+ */
105
+ export function pairAdjudicationPrompt ({ action, reason, findings = [], taskKey = '' } = {}) {
106
+ const head = {
107
+ reject: `The reviewer REJECTED ${taskKey || 'this slice'}.`,
108
+ pass: `The reviewer PASSED ${taskKey || 'this slice'}.`,
109
+ surface: `The reviewer could not conclude on ${taskKey || 'this slice'}.`,
110
+ }[action] || `Review outcome for ${taskKey || 'this slice'}.`
111
+ const body = findings.length
112
+ ? '\n' + findings.map((f) => ` • ${f}`).join('\n')
113
+ : (reason ? `\n ${reason}` : '')
114
+ return `${head}${body}\n\nYou two decide — do you agree? Challenge it, or accept.`
115
+ }
package/flow-review.mjs CHANGED
@@ -12,6 +12,7 @@
12
12
  import { execFileSync } from 'node:child_process'
13
13
  import { TASK_STATUS } from './flow-task-graph.mjs'
14
14
  import { removeFlowWorktree } from './flow-worktree.mjs'
15
+ import { reviewReflectionDecision, REVIEW_DEFAULTS } from './flow-review-reflect.mjs'
15
16
 
16
17
  // The reviewer lane's rolePrompt (via startClaudeSession). Mirrors FLOW_LANE_PROMPT's
17
18
  // join(' ') style. This lane is ADVERSARIAL — its job is to disprove "done", not to build.
@@ -26,7 +27,9 @@ export const FLOW_REVIEWER_PROMPT = [
26
27
 
27
28
  'BE SPECIFIC. Your reasons must name exactly WHAT failed and HOW you found it — the command you ran, the output you got, the acceptance criterion it violated. "Doesn\'t work" is useless. "GET /api/todos returned 500 with `column todos.user_id does not exist`; acceptance required 200 + []" is a usable verdict.',
28
29
 
29
- 'EMIT YOUR VERDICT BY WRITING FLOW_REVIEW.json. Do NOT call mark_flow_done or ExitPlanMode (they hang here). As your LAST action, use the Write tool with file_path "FLOW_REVIEW.json" and content = a single JSON object { "pass": <boolean>, "reasons": ["<specific finding>", ...], "taskKey": "<the slice you reviewed>" }. `pass: false` reverts the reviewed slice so it rebuilds. The content of FLOW_REVIEW.json must be ONLY that JSON object (no prose, no markdown fences). Do not edit the slice — you review, you do not fix.',
30
+ 'YOU GET BOUNDED HUNT ROUNDS, NOT ONE GLANCE. You may be asked to dig again — up to a hard ceiling of rounds. A green happy-path on round 1 is NOT a pass; use the next round to hunt harder (edge cases, the reload, the second click, concurrent use, the error path). You only stop early two ways: you reproduce a concrete FAILURE (emit pass:false — that ends it, the slice reverts), OR you have genuinely EXHAUSTED your checks and found nothing (emit pass:true AND exhausted:true — that ends it, the slice passes). If you still have angles left to try, emit pass:true and leave exhausted false/absent — you will be asked to dig one more round until the ceiling, at which point the inconclusive result is handed to the two humans to decide (never an unbounded self-loop).',
31
+
32
+ 'EMIT YOUR VERDICT BY WRITING FLOW_REVIEW.json. Do NOT call mark_flow_done or ExitPlanMode (they hang here). As your LAST action, use the Write tool with file_path "FLOW_REVIEW.json" and content = a single JSON object { "pass": <boolean>, "reasons": ["<specific finding>", ...], "taskKey": "<the slice you reviewed>", "exhausted": <boolean, optional — true only when you have nothing left to check> }. `pass: false` reverts the reviewed slice so it rebuilds. The content of FLOW_REVIEW.json must be ONLY that JSON object (no prose, no markdown fences). Do not edit the slice — you review, you do not fix.',
30
33
  ].join(' ')
31
34
 
32
35
  // Parse a reviewer's raw verdict. Accepts an object OR a JSON string (optionally
@@ -43,7 +46,57 @@ export function parseReviewVerdict (raw) {
43
46
  if (!obj || typeof obj !== 'object') throw new Error('review verdict is not an object')
44
47
  if (typeof obj.pass !== 'boolean') throw new Error('review verdict missing boolean `pass`')
45
48
  const reasons = Array.isArray(obj.reasons) ? obj.reasons.map(String) : []
46
- return { pass: obj.pass, reasons }
49
+ // E1 (bounded self-test): `exhausted` is an OPTIONAL third signal — true only when the
50
+ // reviewer reports it has nothing left to check. Absent/non-boolean → false (keep hunting
51
+ // until the round ceiling). Never throws on its absence; it's additive to the old contract.
52
+ const exhausted = obj.exhausted === true
53
+ return { pass: obj.pass, reasons, exhausted }
54
+ }
55
+
56
+ // E1 — map a parsed review verdict into the two governor inputs. A concrete FAILURE
57
+ // (pass:false) is a reproduced break → foundFailure. A pass is only conclusive when the
58
+ // reviewer says it EXHAUSTED its checks; a bare pass:true means "happy path held, more to
59
+ // hunt" → keep going until the round ceiling. Pure; the seam the loop drives on.
60
+ export function reviewVerdictToReflection (verdict) {
61
+ const v = verdict || {}
62
+ return {
63
+ foundFailure: v.pass === false,
64
+ exhausted: v.pass === true && v.exhausted === true,
65
+ }
66
+ }
67
+
68
+ // E1+A1/A2 — the BOUNDED reviewer loop. Drives the adversarial reviewer across up to
69
+ // `maxRounds` hunt rounds, consulting reviewReflectionDecision between rounds, and stops at
70
+ // the FIRST terminal outcome (reject on a concrete failure, pass on exhaustion, or SURFACE
71
+ // when a ceiling — rounds or budget — is hit inconclusively). Pure over two injected effects
72
+ // so it's unit-testable without a live Flow room (the S5 split):
73
+ // • runRound(round) → Promise<parsed verdict {pass,reasons,exhausted}> — spend one round
74
+ // • getSpent() → number — the review lane's tokens spent so far (flow-budget ledger)
75
+ // N1 fail-closed: the budget is checked BEFORE each round is spent, so a review lane at/over
76
+ // its cap SURFACES without ever spending the next round (never runs the bill up, anti-Emergent).
77
+ // Returns { action, surfaceToPair, reason, rounds, verdict } — `rounds` = rounds actually run.
78
+ export async function runBoundedReview ({ maxRounds, budgetCap = null, getSpent = () => 0, runRound }) {
79
+ if (typeof runRound !== 'function') throw new Error('runBoundedReview needs a runRound(round) function')
80
+ const cap = Number.isFinite(maxRounds) && maxRounds > 0 ? maxRounds : REVIEW_DEFAULTS.maxRounds
81
+ const capped = budgetCap != null && Number.isFinite(budgetCap)
82
+ for (let round = 1; ; round++) {
83
+ // N1 fail-closed pre-gate: never SPEND another review round once the cap is reached.
84
+ const spentBefore = getSpent()
85
+ if (capped && spentBefore >= budgetCap) {
86
+ const d = reviewReflectionDecision({
87
+ round, maxRounds: cap, spentTokens: spentBefore, budgetCap,
88
+ foundFailure: false, exhausted: false,
89
+ })
90
+ return { ...d, rounds: round - 1, verdict: null }
91
+ }
92
+ const verdict = await runRound(round)
93
+ const decision = reviewReflectionDecision({
94
+ round, maxRounds: cap, spentTokens: getSpent(), budgetCap,
95
+ ...reviewVerdictToReflection(verdict),
96
+ })
97
+ if (decision.stop) return { ...decision, rounds: round, verdict }
98
+ // else action==='continue' — under both ceilings, not yet concluded → dig another round.
99
+ }
47
100
  }
48
101
 
49
102
  // Gate decision over a task's verdicts (≥1 reviewer). Accept iff EVERY verdict passes;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "thinkpool-pair",
3
- "version": "0.7.153",
3
+ "version": "0.7.155",
4
4
  "description": "Share a local coding-agent CLI (Claude Code, Codex, Gemini, Aider, …) into a ThinkPool Code room, live.",
5
5
  "type": "module",
6
6
  "bin": {
@@ -23,6 +23,7 @@
23
23
  "flow-preview.mjs",
24
24
  "flow-review.mjs",
25
25
  "flow-review-gate.mjs",
26
+ "flow-review-reflect.mjs",
26
27
  "flow-assembly.mjs",
27
28
  "flow-budget.mjs",
28
29
  "flow-context-store.mjs",