thinkpool-pair 0.7.171 → 0.7.172
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/claude-session.mjs +153 -32
- package/package.json +2 -1
- package/turn-stall.mjs +43 -0
package/claude-session.mjs
CHANGED
|
@@ -13,11 +13,30 @@
|
|
|
13
13
|
───────────────────────────────────────────────────────────── */
|
|
14
14
|
|
|
15
15
|
import { randomUUID } from 'node:crypto'
|
|
16
|
+
import { createRequire } from 'node:module'
|
|
17
|
+
import { readFileSync } from 'node:fs'
|
|
18
|
+
import { dirname, join } from 'node:path'
|
|
16
19
|
import { query } from '@anthropic-ai/claude-agent-sdk'
|
|
17
20
|
import { sanitizeSession } from './transcript-sanitize.mjs'
|
|
18
21
|
import { reviewGatePreToolDecision } from './flow-review-gate.mjs'
|
|
19
22
|
import { crossPostNeedsCard } from './cross-terminal.mjs'
|
|
20
23
|
import { correctContext } from './context-windows.mjs'
|
|
24
|
+
import { stallDecision, stallEvent } from './turn-stall.mjs'
|
|
25
|
+
|
|
26
|
+
// The caret-pulled SDK's real version (^0.3.x auto-upgrades on restart). Resolved
|
|
27
|
+
// once at import by walking up from the package entry to its own package.json.
|
|
28
|
+
// Named in the [SDK-REGRESSION] guard below so a silent gate-change is attributable.
|
|
29
|
+
const SDK_VERSION = (() => {
|
|
30
|
+
try {
|
|
31
|
+
const req = createRequire(import.meta.url)
|
|
32
|
+
let d = dirname(req.resolve('@anthropic-ai/claude-agent-sdk'))
|
|
33
|
+
for (let i = 0; i < 8; i++) {
|
|
34
|
+
try { const p = JSON.parse(readFileSync(join(d, 'package.json'), 'utf8')); if (p.name === '@anthropic-ai/claude-agent-sdk') return p.version } catch { /* keep walking */ }
|
|
35
|
+
const up = dirname(d); if (up === d) break; d = up
|
|
36
|
+
}
|
|
37
|
+
} catch { /* unresolved */ }
|
|
38
|
+
return 'unknown'
|
|
39
|
+
})()
|
|
21
40
|
|
|
22
41
|
// ── risk classification — the accent/danger tier of the permission card ──
|
|
23
42
|
// low (read-only) · medium (writes/runs) · network (leaves the machine) ·
|
|
@@ -73,6 +92,44 @@ export function autoAllow({ toolName, input, mode = 'default', alwaysAllow = new
|
|
|
73
92
|
)
|
|
74
93
|
}
|
|
75
94
|
|
|
95
|
+
// AskUserQuestion answer path — pure mapping from the room's requestPermission
|
|
96
|
+
// result to the PreToolUse decision fed back to the model. The card handler
|
|
97
|
+
// denies (PreToolUse can't inject a tool_result) and puts the human's pick in
|
|
98
|
+
// the deny reason, which IS what the model receives. Exported so the feedback
|
|
99
|
+
// contract is locked in a unit test (mock requestPermission → this output).
|
|
100
|
+
// `decision` is the requestPermission return: 'answer:<pick>' when the person
|
|
101
|
+
// selected, anything else (including '' / dismissal / a broken path) is treated
|
|
102
|
+
// as "no selection".
|
|
103
|
+
export function askUserQuestionHookOutput(decision) {
|
|
104
|
+
const ans = (typeof decision === 'string' && decision.startsWith('answer:')) ? decision.slice(7) : ''
|
|
105
|
+
return {
|
|
106
|
+
continue: true,
|
|
107
|
+
hookSpecificOutput: {
|
|
108
|
+
hookEventName: 'PreToolUse',
|
|
109
|
+
permissionDecision: 'deny',
|
|
110
|
+
permissionDecisionReason: ans
|
|
111
|
+
? `The user answered in the ThinkPool room — ${ans}. Treat this as their selection and continue; do not call AskUserQuestion again for the same question.`
|
|
112
|
+
: 'The user dismissed the question in the ThinkPool room without selecting. Ask in plain prose, or proceed with a sensible default.',
|
|
113
|
+
},
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
// Regression guard for the AskUserQuestion gate. SDK 0.3.173 enabled the tool
|
|
118
|
+
// unconditionally; a later bump in the 0.3.185→0.3.204 window flipped it to require
|
|
119
|
+
// a `canUseTool` callback, and the ^0.3.x caret silently killed the feature on a
|
|
120
|
+
// routine restart (2026-07-08). The bridge DOES pass canUseTool, so the SDK's init
|
|
121
|
+
// `tools[]` MUST advertise AskUserQuestion — checked against the real init message
|
|
122
|
+
// the live session already receives (no boot-time probe: cold-CLI init is flaky).
|
|
123
|
+
// If the message carries a tools[] that lacks AskUserQuestion → return the loud
|
|
124
|
+
// [SDK-REGRESSION] reason (logged + surfaced in the room). If it carries no tools[]
|
|
125
|
+
// (older/other SDK message shape) → fail OPEN (null), never a false alarm. Pure +
|
|
126
|
+
// exported so both branches are locked by a unit test.
|
|
127
|
+
export function askUserQuestionRegression(tools, version) {
|
|
128
|
+
if (!Array.isArray(tools)) return null
|
|
129
|
+
if (tools.includes('AskUserQuestion')) return null
|
|
130
|
+
return `[SDK-REGRESSION] agent SDK v${version || 'unknown'} init tools[] is missing 'AskUserQuestion' — the multiple-choice question card is disabled. A caret SDK bump likely changed the canUseTool gate; pin a known-good SDK in bridge/package.json + republish.`
|
|
131
|
+
}
|
|
132
|
+
|
|
76
133
|
// ── input stream — a generator we keep open and feed turns into ──
|
|
77
134
|
function makeInputStream() {
|
|
78
135
|
const queue = []
|
|
@@ -183,6 +240,7 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
|
|
|
183
240
|
const spawnT0 = Date.now()
|
|
184
241
|
let readyLogged = false
|
|
185
242
|
let modelsSent = false // one-shot: emit the SDK's supported-model list on first init
|
|
243
|
+
let sdkToolsChecked = false // one-shot: run the AskUserQuestion regression guard on the first init carrying tools[]
|
|
186
244
|
let q = null // the live Query — control requests (interrupt /
|
|
187
245
|
// setPermissionMode) route through it once streaming.
|
|
188
246
|
// The session is CREATED in the caller's chosen mode (not hard-coded default):
|
|
@@ -223,8 +281,10 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
|
|
|
223
281
|
// 3+ min) or abort silently with no `result`. Track turn liveness + last event
|
|
224
282
|
// time; if an active turn goes quiet past STALL_MS, emit one `stalled` chrome
|
|
225
283
|
// event so the room can stop showing a frozen "working" forever. Cleared on the
|
|
226
|
-
// next event / `result`. Refs: TS SDK #44, claude-code #38905.
|
|
227
|
-
|
|
284
|
+
// next event / `result`. Refs: TS SDK #44, claude-code #38905. The decision layer
|
|
285
|
+
// (stallDecision/stallEvent, above) is pure + unit-tested; this owns the state +
|
|
286
|
+
// side effects. 90s status default (2026-07-08 hardening: the 529-overload hang).
|
|
287
|
+
const STALL_MS = Math.max(30000, parseInt(process.env.TP_STALL_MS, 10) || 90000)
|
|
228
288
|
let turnActive = false // true between a sent turn and its `result`
|
|
229
289
|
let lastEvtTs = Date.now() // wall-clock of the most recent emitted event
|
|
230
290
|
let stalledSent = false // one `stalled` per stall, not a storm
|
|
@@ -240,12 +300,16 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
|
|
|
240
300
|
let restartCount = 0
|
|
241
301
|
let restartTimer = null // the pending auto-restart backoff — cancelled by end()
|
|
242
302
|
// Force-stop a true wedge (item 3): no result, no error, just silence past
|
|
243
|
-
// FORCE_STOP_MS.
|
|
244
|
-
//
|
|
245
|
-
//
|
|
246
|
-
// the
|
|
303
|
+
// FORCE_STOP_MS. As of the 2026-07-08 hardening we no longer just surface an error
|
|
304
|
+
// and leave the hung loop in place — retryStalledTurn ABORTS the wedged SDK call (via
|
|
305
|
+
// its AbortController, the same unblock path abort()/recreateForSwitch use) and
|
|
306
|
+
// AUTO-RETRIES the turn ONCE, preserving the prompt. If that one retry ALSO stalls
|
|
307
|
+
// past FORCE_STOP_MS, we fall back to the old behavior: force-stop + "send again".
|
|
308
|
+
// stallRetried bounds it to a single auto-retry per turn (reset on each new turn /
|
|
309
|
+
// settled result), so a sustained upstream outage can't thundering-herd retries.
|
|
247
310
|
const FORCE_STOP_MS = Math.max(STALL_MS * 3, parseInt(process.env.TP_FORCE_STOP_MS, 10) || 300000)
|
|
248
311
|
let forceStopped = false
|
|
312
|
+
let stallRetried = false // one auto-retry per turn (the abort+re-run path below)
|
|
249
313
|
// Waiting on a HUMAN decision (permission card / plan gate / AskUserQuestion)
|
|
250
314
|
// is NOT a wedge — the turn is correctly idle until the person answers. Every
|
|
251
315
|
// interactive gate goes through requestPermission, so wrap it once to bump a
|
|
@@ -288,22 +352,29 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
|
|
|
288
352
|
let sugTimer = null // pending fallback timer
|
|
289
353
|
const SUGGEST_FALLBACK_MS = Math.max(1200, parseInt(process.env.TP_SUGGEST_FALLBACK_MS, 10) || 2500)
|
|
290
354
|
|
|
291
|
-
|
|
355
|
+
// emitRaw forwards an event WITHOUT stamping lastEvtTs — used for the watchdog's OWN
|
|
356
|
+
// synthetic status/error events, which are not SDK activity and must not reset the
|
|
357
|
+
// silence clock (else the 5-min abort would measure from our 90s status emit, not from
|
|
358
|
+
// the last real SDK event). emit() is the normal path: it stamps liveness + clears the
|
|
359
|
+
// one-shot stall flag on any real event other than a `stalled` re-emit.
|
|
360
|
+
const emitRaw = (evt) => { try { onEvent?.(evt) } catch { /* never let a consumer throw into the loop */ } }
|
|
361
|
+
const emit = (evt) => { lastEvtTs = Date.now(); if (evt && evt.kind !== 'stalled') stalledSent = false; emitRaw(evt) }
|
|
292
362
|
const stallTimer = setInterval(() => {
|
|
293
|
-
if (!turnActive || awaitingUser > 0) return // awaiting a human decision ≠ wedged
|
|
294
363
|
const quiet = Date.now() - lastEvtTs
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
}
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
364
|
+
const action = stallDecision({ turnActive, awaitingUser, quietMs: quiet, stallMs: STALL_MS, forceStopMs: FORCE_STOP_MS, stalledSent, stallRetried })
|
|
365
|
+
if (action === 'none') return
|
|
366
|
+
// ALWAYS surface — a stall is never swallowed. emitRaw so this synthetic event
|
|
367
|
+
// doesn't reset the silence clock (see emitRaw above).
|
|
368
|
+
const ev = stallEvent(action, quiet)
|
|
369
|
+
if (ev) emitRaw(ev)
|
|
370
|
+
if (action === 'status') { stalledSent = true; return }
|
|
371
|
+
if (action === 'retry') { forceStopped = true; retryStalledTurn(quiet); return }
|
|
372
|
+
// 'giveup' — the one auto-retry ALSO stalled past FORCE_STOP_MS. Fall back to the
|
|
373
|
+
// pre-2026-07-08 behavior: force-stop the turn so between-turns updates unblock, and
|
|
374
|
+
// let the human resend. The wedged loop is left in place; if it later throws, the
|
|
375
|
+
// catch self-heals.
|
|
376
|
+
forceStopped = true
|
|
377
|
+
turnActive = false
|
|
307
378
|
}, 5000)
|
|
308
379
|
stallTimer.unref?.()
|
|
309
380
|
|
|
@@ -527,17 +598,7 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
|
|
|
527
598
|
let decision = ''
|
|
528
599
|
try { decision = await requestPermission?.({ id: randomUUID(), toolName, input: toolInput, risk: 'ask', questions: toolInput?.questions || [] }) ?? '' }
|
|
529
600
|
catch { decision = '' }
|
|
530
|
-
|
|
531
|
-
return {
|
|
532
|
-
continue: true,
|
|
533
|
-
hookSpecificOutput: {
|
|
534
|
-
hookEventName: 'PreToolUse',
|
|
535
|
-
permissionDecision: 'deny',
|
|
536
|
-
permissionDecisionReason: ans
|
|
537
|
-
? `The user answered in the ThinkPool room — ${ans}. Treat this as their selection and continue; do not call AskUserQuestion again for the same question.`
|
|
538
|
-
: 'The user dismissed the question in the ThinkPool room without selecting. Ask in plain prose, or proceed with a sensible default.',
|
|
539
|
-
},
|
|
540
|
-
}
|
|
601
|
+
return askUserQuestionHookOutput(decision)
|
|
541
602
|
}
|
|
542
603
|
const risk = classifyRisk(toolName, toolInput)
|
|
543
604
|
// "Don't ask again" is keyed by tool + risk tier, so allowing medium Bash
|
|
@@ -574,6 +635,21 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
|
|
|
574
635
|
abortController: ac,
|
|
575
636
|
permissionMode: mode,
|
|
576
637
|
hooks: { PreToolUse: [{ hooks: [preTool] }] },
|
|
638
|
+
// Enable AskUserQuestion (the agent's multiple-choice card) in EVERY mode.
|
|
639
|
+
// The CLI only adds AskUserQuestion / EnterPlanMode / ExitPlanMode to a
|
|
640
|
+
// session's toolset when the host declares it can service an interactive
|
|
641
|
+
// permission — i.e. when a `canUseTool` callback is present. Without it the
|
|
642
|
+
// tool is absent and the model gets "AskUserQuestion exists but is not
|
|
643
|
+
// enabled in this context", so the room's card handler above was dead code.
|
|
644
|
+
// Its PRESENCE is the whole point: in bypassPermissions this callback is
|
|
645
|
+
// shadowed (never invoked — bypass auto-approves before it, per the SDK's
|
|
646
|
+
// CLAUDE_SDK_CAN_USE_TOOL_SHADOWED warning), and in gated modes the
|
|
647
|
+
// PreToolUse hook runs first and stays authoritative (a hook deny — which is
|
|
648
|
+
// exactly how the AskUserQuestion card feeds its answer back — short-circuits
|
|
649
|
+
// canUseTool). So the human-gate holds in all modes: an ask-card always waits
|
|
650
|
+
// for a person via requestPermission inside the hook. This allow is a no-op
|
|
651
|
+
// relative to the hook, never a second gate that could override a denial.
|
|
652
|
+
canUseTool: async (_toolName, input) => ({ behavior: 'allow', updatedInput: input }),
|
|
577
653
|
// Load the host's REAL Claude environment — user + project + local settings —
|
|
578
654
|
// so custom slash commands (.claude/commands/*.md), CLAUDE.md and agents work
|
|
579
655
|
// in the room exactly as in the user's own CLI. The Agent SDK isolates by
|
|
@@ -780,6 +856,19 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
|
|
|
780
856
|
// correct it). opts.model reflects what the session will actually run, so trust it.
|
|
781
857
|
curModel = opts.model || m.model || model || curModel
|
|
782
858
|
emit({ kind: 'system', sessionId, model: opts.model || m.model || model || null, commands: Array.isArray(m.slash_commands) ? m.slash_commands : undefined })
|
|
859
|
+
// AskUserQuestion regression guard — the init message carries the session's
|
|
860
|
+
// real tool list. If a caret SDK bump silently dropped AskUserQuestion from
|
|
861
|
+
// it (the 2026-07-08 outage), surface it loudly instead of a dead card. One-
|
|
862
|
+
// shot per session (resumed sessions replay init messages). Fails open when
|
|
863
|
+
// the message shape carries no tools[].
|
|
864
|
+
if (!sdkToolsChecked && Array.isArray(m.tools)) {
|
|
865
|
+
sdkToolsChecked = true
|
|
866
|
+
const regression = askUserQuestionRegression(m.tools, SDK_VERSION)
|
|
867
|
+
if (regression) {
|
|
868
|
+
process.stderr.write(`\n ⚠⚠ ${regression}\n`)
|
|
869
|
+
emit({ kind: 'error', message: regression })
|
|
870
|
+
}
|
|
871
|
+
}
|
|
783
872
|
// Dynamic model list — ask the SDK for its OWN supported models and surface
|
|
784
873
|
// them to the room so the /model picker renders live options (incl. new
|
|
785
874
|
// models like Fable) instead of a hardcoded three-item list. Fire-and-forget:
|
|
@@ -875,6 +964,7 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
|
|
|
875
964
|
// drops refills every cycle past RESTART_MAX) — and marks this session id durably
|
|
876
965
|
// persisted so a later model-switch re-create can safely resume it.
|
|
877
966
|
forceStopped = false
|
|
967
|
+
stallRetried = false // a turn settled → the next turn gets a fresh auto-retry budget
|
|
878
968
|
// Ground-truth the model from THIS turn's modelUsage (the models actually billed
|
|
879
969
|
// this turn), not the init/system messages — a RESUMED session REPLAYS the prior
|
|
880
970
|
// transcript's init messages, so curModel got polluted back to the pre-switch model
|
|
@@ -991,12 +1081,43 @@ export function startClaudeSession({ cwd, model, resume, env, mode: initialMode
|
|
|
991
1081
|
sessionId = persistedSessionId
|
|
992
1082
|
if (!closed) runQuery() // relaunch on opts.model + resume
|
|
993
1083
|
}
|
|
1084
|
+
// Abort a wedged turn (no event, no throw — the SDK's `for await` blocked inside an
|
|
1085
|
+
// internal retry, e.g. an Anthropic 529-overload window) and re-run it ONCE, preserving
|
|
1086
|
+
// the prompt. Mirrors recreateForSwitch's teardown-then-relaunch (abort qAc → swap the
|
|
1087
|
+
// input stream → await the dying query's disposal so the two processes never race on the
|
|
1088
|
+
// session file lock → resume the durable id) and then RE-PUSHES lastTurnText so the turn
|
|
1089
|
+
// continues instead of being lost. The stall error was already surfaced by the watchdog
|
|
1090
|
+
// (emitRaw of stallEvent('retry', …)); this adds a 'note' once the fresh turn is armed.
|
|
1091
|
+
// Bounded to one call per turn by stallRetried (set in the interval before we're called).
|
|
1092
|
+
const retryStalledTurn = async (quietMs) => {
|
|
1093
|
+
if (closed) return
|
|
1094
|
+
stallRetried = true
|
|
1095
|
+
const wasActive = turnActive
|
|
1096
|
+
const replay = wasActive ? lastTurnText : null
|
|
1097
|
+
turnActive = false // unblock between-turns updates while we tear down
|
|
1098
|
+
const dying = qDone
|
|
1099
|
+
const oldInput = input
|
|
1100
|
+
input = makeInputStream() // fresh stream for the relaunched query
|
|
1101
|
+
if (qAc) { try { qAc.abort() } catch { /* noop */ } } // unblock the wedged for-await → its catch disposes the process
|
|
1102
|
+
try { oldInput.end() } catch { /* noop */ }
|
|
1103
|
+
try { await dying } catch { /* noop */ } // wait until the old query is actually dead (avoids the lock race)
|
|
1104
|
+
await sleep(300) // brief lock-release margin
|
|
1105
|
+
sessionId = persistedSessionId // resume the durably-persisted id (recreateForSwitch invariant)
|
|
1106
|
+
if (closed) return
|
|
1107
|
+
runQuery()
|
|
1108
|
+
if (replay != null) {
|
|
1109
|
+
turnActive = true; lastEvtTs = Date.now(); stalledSent = false // re-arm liveness for the retried turn
|
|
1110
|
+
// Match sendTurn's block shape: a slash command goes clean; a normal turn keeps the reminder.
|
|
1111
|
+
input.push(/^\s*\//.test(replay) ? [{ type: 'text', text: replay }] : [{ type: 'text', text: replay }, { type: 'text', text: roomReminder() }])
|
|
1112
|
+
emitRaw({ kind: 'note', text: 'retrying the stalled turn on a fresh connection' })
|
|
1113
|
+
}
|
|
1114
|
+
}
|
|
994
1115
|
if (!lazy) runQuery()
|
|
995
1116
|
|
|
996
1117
|
return {
|
|
997
1118
|
// A lazy (restored-idle) session cold-boots the query on its first turn. input.push
|
|
998
1119
|
// is queue-backed, so the pushed turn buffers and runs once the query is ready.
|
|
999
|
-
sendTurn(text) { if (!closed) { if (!started) runQuery(); turnActive = true; sawSuggestion = false; lastTurnText = String(text); if (sugTimer) { clearTimeout(sugTimer); sugTimer = null } lastEvtTs = Date.now(); stalledSent = false;
|
|
1120
|
+
sendTurn(text) { if (!closed) { if (!started) runQuery(); turnActive = true; sawSuggestion = false; lastTurnText = String(text); if (sugTimer) { clearTimeout(sugTimer); sugTimer = null } lastEvtTs = Date.now(); stalledSent = false; stallRetried = false; forceStopped = false;
|
|
1000
1121
|
const t = String(text)
|
|
1001
1122
|
// SDK 0.3.198 only recognizes a slash command (/compact, /clear, custom /cmds) when the
|
|
1002
1123
|
// user message is a SINGLE text block. Appending the per-turn ROOM NOW <system-reminder>
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "thinkpool-pair",
|
|
3
|
-
"version": "0.7.
|
|
3
|
+
"version": "0.7.172",
|
|
4
4
|
"description": "Share a local coding-agent CLI (Claude Code, Codex, Gemini, Aider, …) into a ThinkPool Code room, live.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -13,6 +13,7 @@
|
|
|
13
13
|
"byok-detect.mjs",
|
|
14
14
|
"context-windows.mjs",
|
|
15
15
|
"claude-session.mjs",
|
|
16
|
+
"turn-stall.mjs",
|
|
16
17
|
"update-gate.mjs",
|
|
17
18
|
"event-id.mjs",
|
|
18
19
|
"transcript-sanitize.mjs",
|
package/turn-stall.mjs
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
/* ─────────────────────────────────────────────────────────────
|
|
2
|
+
turn-stall.mjs — the pure decision layer for the structured-turn stall watchdog.
|
|
3
|
+
|
|
4
|
+
A structured turn can stall with the SDK's `for await` blocked INSIDE an internal retry
|
|
5
|
+
— no event, no throw — so claude-session's catch never runs and nothing reaches the room.
|
|
6
|
+
The 2026-07-08 incident: an Anthropic 529-overload window hung a turn for 14 minutes with
|
|
7
|
+
zero output until it was aborted by hand. The watchdog in claude-session.mjs turns that
|
|
8
|
+
silent wedge into a visible status, then an abort + one auto-retry, then a force-stop.
|
|
9
|
+
|
|
10
|
+
These two functions are the DECISION layer (when to act / what to say). They're pure — no
|
|
11
|
+
SDK, no clock, no I/O — so the stateful loop injects state in and they unit-test against a
|
|
12
|
+
fake clock with zero deps (bridge/test-turn-stall-watchdog.mjs). Same shape + spirit as
|
|
13
|
+
account.mjs's claimLoopWedged / withTimeout. Kept in their OWN module (no SDK import) so the
|
|
14
|
+
test needs neither the agent SDK nor auth. The loop owns the flags + the abort/re-push side
|
|
15
|
+
effects; this only says WHAT to do.
|
|
16
|
+
───────────────────────────────────────────────────────────── */
|
|
17
|
+
|
|
18
|
+
// Decide the watchdog action for an ACTIVE turn gone quiet (quietMs = ms since the last
|
|
19
|
+
// emitted SDK event). Idle (no turn) or awaiting a HUMAN decision (permission card / plan
|
|
20
|
+
// gate / AskUserQuestion) is never a wedge — the turn is correctly idle until the person
|
|
21
|
+
// answers. Timeline:
|
|
22
|
+
// quietMs > stallMs (90s) → 'status' — surface a "still retrying (Ns)" banner, ONCE
|
|
23
|
+
// quietMs > forceStopMs (5min) → 'retry' — abort the hung call, re-run the turn ONCE
|
|
24
|
+
// quietMs > forceStopMs again → 'giveup' — the one retry also stalled: force-stop, resend
|
|
25
|
+
export function stallDecision({ turnActive, awaitingUser, quietMs, stallMs, forceStopMs, stalledSent, stallRetried }) {
|
|
26
|
+
if (!turnActive || awaitingUser > 0) return 'none'
|
|
27
|
+
if (quietMs > forceStopMs) return stallRetried ? 'giveup' : 'retry'
|
|
28
|
+
if (!stalledSent && quietMs > stallMs) return 'status'
|
|
29
|
+
return 'none'
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
// The room-facing event for a watchdog action (pure → the exact wording is unit-tested).
|
|
33
|
+
// 'status' rides the transient `stalled` chrome event (broadcast, never persisted); 'retry'
|
|
34
|
+
// and 'giveup' ride `error` (recoverable) so the stall is ALWAYS surfaced, never swallowed.
|
|
35
|
+
export function stallEvent(action, quietMs) {
|
|
36
|
+
const s = Math.round(quietMs / 1000)
|
|
37
|
+
switch (action) {
|
|
38
|
+
case 'status': return { kind: 'stalled', sinceMs: quietMs, retrying: true, message: `model call stalled — still retrying (${s}s)` }
|
|
39
|
+
case 'retry': return { kind: 'error', recoverable: true, message: `model call stalled ${s}s with no response (upstream overload) — aborting the hung call and retrying the turn once` }
|
|
40
|
+
case 'giveup': return { kind: 'error', recoverable: true, message: `agent went silent for ${s}s — the retry also stalled; force-stopped the turn, send again to resume` }
|
|
41
|
+
default: return null
|
|
42
|
+
}
|
|
43
|
+
}
|